Files
plantillas-proyectos/backend/api/v1/modules/a76/layouts_csv/drivers/tasks.py
2026-03-03 16:45:12 -07:00

641 lines
23 KiB
Python

"""
Tareas Celery para importacion CSV de Conductores.
Flujo: scan_file (validacion) -> insert_valid_rows (commit).
"""
import os
import base64
import csv
import json
import logging
import re
import unicodedata
from typing import Dict, Any, Optional, List
from core.celery_app import celery_app
from core.database import CoreSessionLocal
from core.paths import layout_path
from .template_config import row_from_template
logger = logging.getLogger(__name__)
DRV_IMPORT_FILE_PREFIX = "drv_import_file:"
DRV_IMPORT_META_PREFIX = "drv_import_meta:"
DRV_IMPORT_ERROR_LINES_PREFIX = "drv_import_error_lines:"
DRV_IMPORT_STATUS_PREFIX = "drv_import_status:"
DRV_IMPORT_REDIS_TTL = 3600
def _get_redis():
import redis
url = os.getenv("VALKEY_URL", os.getenv("REDIS_URL", "redis://valkey:6379/0"))
return redis.Redis.from_url(url, decode_responses=False)
def _worker_upload_dir() -> str:
return layout_path("imports", "temp")
def _ensure_worker_has_file_from_redis(job_id: str) -> Optional[str]:
r = _get_redis()
data = r.get(f"{DRV_IMPORT_FILE_PREFIX}{job_id}")
if not data:
return None
try:
raw = base64.b64decode(data)
except Exception as e:
logger.warning(f"Drivers import: failed to decode file from Redis: {e}")
return None
upload_dir = _worker_upload_dir()
os.makedirs(upload_dir, exist_ok=True)
file_path = os.path.join(upload_dir, f"drv_{job_id}.csv")
with open(file_path, "wb") as f:
f.write(raw)
return file_path
def _ensure_worker_has_meta_from_redis(job_id: str, file_path: str) -> bool:
r = _get_redis()
data = r.get(f"{DRV_IMPORT_META_PREFIX}{job_id}")
if not data:
return False
try:
meta = json.loads(data.decode("utf-8"))
except Exception as e:
logger.warning(f"Drivers import: failed to decode meta from Redis: {e}")
return False
meta_path = file_path.replace(".csv", ".meta.json")
with open(meta_path, "w", encoding="utf-8") as f:
json.dump(meta, f)
return True
def _delete_import_from_redis(job_id: str) -> None:
try:
r = _get_redis()
r.delete(
f"{DRV_IMPORT_FILE_PREFIX}{job_id}",
f"{DRV_IMPORT_META_PREFIX}{job_id}",
f"{DRV_IMPORT_ERROR_LINES_PREFIX}{job_id}",
f"{DRV_IMPORT_STATUS_PREFIX}{job_id}",
)
except Exception as e:
logger.warning(f"Drivers import: failed to delete Redis keys: {e}")
def normalize_header(name: Optional[str]) -> str:
if not name:
return ""
name = unicodedata.normalize("NFKD", str(name)).upper()
name = "".join(ch for ch in name if not unicodedata.combining(ch))
name = re.sub(r"[^A-Z0-9]+", " ", name)
return re.sub(r"\s+", " ", name).strip()
_MAX = {
"transporter_key": 5,
"driver_name": 80,
"license_number": 29,
"express_line_id": 17,
"ace_id": 20,
"gender": 1,
"birth_country": 3,
"hazardous_material_auth": 2,
"hazardous_material_state": 30,
"first_name": 20,
"last_name": 20,
"id_key1": 40,
"id_number1": 20,
"id_state1": 30,
"id_country1": 3,
"id_key2": 40,
"id_number2": 20,
"id_state2": 30,
"id_country2": 3,
"badge_number": 20,
"class_type": 1,
"unique_badge_number": 100,
}
def _parse_int(val: Any) -> Optional[int]:
if val is None or (isinstance(val, str) and not val.strip()):
return None
s = str(val).strip()
if re.match(r"^\d+$", s):
return int(s)
if re.match(r"^\d+\.0+$", s):
try:
return int(float(s))
except ValueError:
return None
return None
def _parse_birth_date(val: Any) -> Optional[int]:
if val is None or (isinstance(val, str) and not val.strip()):
return None
s = str(val).strip()
if re.match(r"^\d{8}$", s):
try:
return int(s)
except ValueError:
return None
if re.match(r"^\d+\.0+$", s):
try:
return int(float(s))
except ValueError:
return None
for sep in ["/", "-", "."]:
if sep in s:
parts = s.split(sep)
if len(parts) == 3:
try:
a, b, c = [p.strip() for p in parts]
if len(c) == 4 and len(a) <= 2 and len(b) <= 2:
return int(c) * 10000 + int(b) * 100 + int(a)
if len(a) == 4 and len(b) <= 2 and len(c) <= 2:
return int(a) * 10000 + int(b) * 100 + int(c)
except (ValueError, TypeError):
return None
break
return None
def _str_or_none(val: Any, max_len: Optional[int] = None) -> Optional[str]:
if val is None:
return None
s = str(val).strip()
if not s:
return None
if max_len and len(s) > max_len:
return s[:max_len]
return s
def _dedupe_headers(headers: List[str]) -> List[str]:
counts: Dict[str, int] = {}
unique: List[str] = []
for header in headers:
name = str(header or "").strip() or "COL"
count = counts.get(name, 0) + 1
counts[name] = count
if count == 1:
unique.append(name)
else:
unique.append(f"{name} {count}")
return unique
def _validate_row_driver(row: Dict[str, Any], line_num: int) -> Optional[Dict[str, Any]]:
transporter_key = (row.get("TRANSPORTISTA") or "").strip()
if not transporter_key:
return {"line": line_num, "col": "TRANSPORTISTA", "msg": "Requerido"}
if len(transporter_key) > _MAX["transporter_key"]:
return {"line": line_num, "col": "TRANSPORTISTA", "msg": f"Maximo {_MAX['transporter_key']} caracteres"}
line_val = _parse_int(row.get("LINEA"))
if line_val is None:
return {"line": line_num, "col": "LINEA", "msg": "Debe ser numerico"}
if line_val <= 0:
return {"line": line_num, "col": "LINEA", "msg": "Debe ser mayor a 0"}
driver_name = (row.get("CLAVE CONDUCTOR") or "").strip()
if driver_name and len(driver_name) > _MAX["driver_name"]:
return {"line": line_num, "col": "CLAVE CONDUCTOR", "msg": f"Maximo {_MAX['driver_name']} caracteres"}
for col, max_len in [
("LICENCIA", _MAX["license_number"]),
("PERMISO LINEA EXPRESS", _MAX["express_line_id"]),
("IDENTIFICACION ACE", _MAX["ace_id"]),
("SEXO", _MAX["gender"]),
("PAIS NACIMIENTO", _MAX["birth_country"]),
("TRANSPORTA MAT. PELIGROSO?", _MAX["hazardous_material_auth"]),
("PERMISO MAT. PELIGROSO", _MAX["hazardous_material_state"]),
("NOMBRE(S)", _MAX["first_name"]),
("APELLIDO PATERNO", _MAX["last_name"]),
("FORMA IDENTIFICACION 1", _MAX["id_key1"]),
("NUM. IDENTIFICACION 1", _MAX["id_number1"]),
("ESTADO", _MAX["id_state1"]),
("PAIS", _MAX["id_country1"]),
("FORMA IDENTIFICACION 2", _MAX["id_key2"]),
("NUM. IDENTIFICACION 2", _MAX["id_number2"]),
("ESTADO 2", _MAX["id_state2"]),
("PAIS 2", _MAX["id_country2"]),
]:
val = (row.get(col) or "").strip()
if val and len(val) > max_len:
return {"line": line_num, "col": col, "msg": f"Maximo {max_len} caracteres"}
fecha = row.get("FECHA NACIMIENTO")
if fecha is not None and str(fecha).strip():
if _parse_birth_date(fecha) is None:
return {
"line": line_num,
"col": "FECHA NACIMIENTO",
"msg": "Formato de fecha invalido (use YYYYMMDD o DD/MM/YYYY)",
}
return None
def _row_to_driver_dto(row: Dict[str, Any], tenant_id: int, company_id: int) -> Dict[str, Any]:
transporter_key = _str_or_none(row.get("TRANSPORTISTA"), _MAX["transporter_key"])
line = _parse_int(row.get("LINEA"))
if not transporter_key or line is None:
return {}
data = {
"transporter_key": transporter_key,
"line": line,
"driver_name": _str_or_none(row.get("CLAVE CONDUCTOR"), _MAX["driver_name"]),
"license_number": _str_or_none(row.get("LICENCIA"), _MAX["license_number"]),
"express_line_id": _str_or_none(row.get("PERMISO LINEA EXPRESS"), _MAX["express_line_id"]),
"ace_id": _str_or_none(row.get("IDENTIFICACION ACE"), _MAX["ace_id"]),
"birth_date": _parse_birth_date(row.get("FECHA NACIMIENTO")),
"gender": _str_or_none(row.get("SEXO"), _MAX["gender"]),
"birth_country": _str_or_none(row.get("PAIS NACIMIENTO"), _MAX["birth_country"]),
"hazardous_material_auth": _str_or_none(
row.get("TRANSPORTA MAT. PELIGROSO?"), _MAX["hazardous_material_auth"]
),
"hazardous_material_state": _str_or_none(
row.get("PERMISO MAT. PELIGROSO"), _MAX["hazardous_material_state"]
),
"first_name": _str_or_none(row.get("NOMBRE(S)"), _MAX["first_name"]),
"last_name": _str_or_none(row.get("APELLIDO PATERNO"), _MAX["last_name"]),
"id_key1": _str_or_none(row.get("FORMA IDENTIFICACION 1"), _MAX["id_key1"]),
"id_number1": _str_or_none(row.get("NUM. IDENTIFICACION 1"), _MAX["id_number1"]),
"id_state1": _str_or_none(row.get("ESTADO"), _MAX["id_state1"]),
"id_country1": _str_or_none(row.get("PAIS"), _MAX["id_country1"]),
"id_key2": _str_or_none(row.get("FORMA IDENTIFICACION 2"), _MAX["id_key2"]),
"id_number2": _str_or_none(row.get("NUM. IDENTIFICACION 2"), _MAX["id_number2"]),
"id_state2": _str_or_none(row.get("ESTADO 2"), _MAX["id_state2"]),
"id_country2": _str_or_none(row.get("PAIS 2"), _MAX["id_country2"]),
"company_id": company_id,
"tenant_id": tenant_id,
}
return data
def _do_scan(job_id: str, progress_callback: Optional[Any] = None) -> Dict[str, Any]:
file_path = _ensure_worker_has_file_from_redis(job_id)
if not file_path:
return {"status": "failed", "error": "Archivo no encontrado (expirado o no subido). Sube de nuevo."}
_ensure_worker_has_meta_from_redis(job_id, file_path)
error_dir = layout_path("imports", "errors")
os.makedirs(error_dir, exist_ok=True)
error_path = os.path.join(error_dir, f"drv_{job_id}.jsonl")
total_rows = 0
try:
with open(file_path, "r", encoding="utf-8-sig") as f:
total_rows = sum(1 for _ in f) - 1
except Exception as e:
return {"status": "failed", "error": str(e)}
meta_path = file_path.replace(".csv", ".meta.json")
meta = {}
if os.path.exists(meta_path):
try:
with open(meta_path, "r", encoding="utf-8") as f:
meta = json.load(f) or {}
except Exception as e:
logger.warning(f"Drivers import: failed to read meta: {e}")
tenant_id = meta.get("tenant_id")
company_id = meta.get("company_id")
if not tenant_id or not company_id:
return {"status": "failed", "error": "Falta contexto (tenant/company)"}
error_count = 0
processed_rows = 0
errors_detail: List[Dict[str, Any]] = []
try:
with open(file_path, "r", encoding="utf-8-sig") as f_in, open(
error_path, "w", encoding="utf-8"
) as f_err:
sample = f_in.read(2048)
f_in.seek(0)
try:
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t")
except Exception:
dialect = "excel"
reader = csv.reader(f_in, dialect=dialect)
try:
headers = next(reader)
except StopIteration:
headers = []
headers = _dedupe_headers(headers)
dict_reader = csv.DictReader(f_in, fieldnames=headers, dialect=dialect)
for i, row in enumerate(dict_reader, start=1):
if progress_callback and i % 500 == 0:
progress_callback(i, total_rows, error_count)
row_norm = row_from_template(row, normalize_header)
err = _validate_row_driver(row_norm, i)
if err:
error_count += 1
f_err.write(json.dumps(err) + "\n")
if len(errors_detail) < 500:
errors_detail.append(
{"line": err["line"], "col": err.get("col", ""), "msg": err.get("msg", "")}
)
processed_rows += 1
except Exception as e:
logger.error(f"Drivers import scan failed: {e}")
return {"status": "failed", "error": str(e)}
error_lines_list = []
try:
if os.path.exists(error_path):
with open(error_path, "r", encoding="utf-8") as f:
for line in f:
try:
err = json.loads(line)
if "line" in err:
error_lines_list.append(err["line"])
except Exception:
pass
if error_lines_list:
r = _get_redis()
r.set(
f"{DRV_IMPORT_ERROR_LINES_PREFIX}{job_id}",
json.dumps(error_lines_list).encode("utf-8"),
ex=DRV_IMPORT_REDIS_TTL,
)
except Exception as e:
logger.warning(f"Drivers import: failed to store error lines: {e}")
result = {
"status": "waiting_confirmation",
"job_id": job_id,
"total_rows": processed_rows,
"error_count": error_count,
"valid_rows": processed_rows - error_count,
"errors": errors_detail,
}
return result
def run_scan_sync(job_id: str) -> Dict[str, Any]:
result = _do_scan(job_id, progress_callback=None)
try:
r = _get_redis()
r.set(
f"{DRV_IMPORT_STATUS_PREFIX}{job_id}",
json.dumps(result).encode("utf-8"),
ex=DRV_IMPORT_REDIS_TTL,
)
except Exception as e:
logger.warning(f"Drivers import: failed to store scan status in Redis: {e}")
return result
@celery_app.task(bind=True)
def scan_file(self, job_id: str, config: str = None):
logger.info(f"Drivers import: starting scan for job {job_id}")
def on_progress(current: int, total: int, errors: int) -> None:
self.update_state(
state="PROGRESS",
meta={"current": current, "total": total, "errors": errors},
)
result = _do_scan(job_id, progress_callback=on_progress)
try:
r = _get_redis()
r.set(
f"{DRV_IMPORT_STATUS_PREFIX}{job_id}",
json.dumps(result).encode("utf-8"),
ex=DRV_IMPORT_REDIS_TTL,
)
except Exception as e:
logger.warning(f"Drivers import: failed to store scan status in Redis: {e}")
return result
def _do_commit(job_id: str) -> Dict[str, Any]:
file_path = _ensure_worker_has_file_from_redis(job_id)
if not file_path:
alt_path = os.path.join(_worker_upload_dir(), f"drv_{job_id}.csv")
if not os.path.exists(alt_path):
return {
"status": "failed",
"error": "Archivo no encontrado (expirado). Sube y confirma de nuevo.",
}
file_path = alt_path
else:
_ensure_worker_has_meta_from_redis(job_id, file_path)
error_dir = layout_path("imports", "errors")
error_path = os.path.join(error_dir, f"drv_{job_id}.jsonl")
error_lines = set()
try:
r = _get_redis()
raw = r.get(f"{DRV_IMPORT_ERROR_LINES_PREFIX}{job_id}")
if raw:
error_lines = set(json.loads(raw.decode("utf-8")))
except Exception as e:
logger.debug(f"Drivers import: could not load error lines from Redis: {e}")
if not error_lines and os.path.exists(error_path):
with open(error_path, "r", encoding="utf-8") as f:
for line in f:
try:
err = json.loads(line)
error_lines.add(err["line"])
except Exception:
pass
meta_path = file_path.replace(".csv", ".meta.json")
tenant_id = None
company_id = None
meta = {}
if os.path.exists(meta_path):
try:
with open(meta_path, "r", encoding="utf-8") as f:
meta = json.load(f) or {}
tenant_id = meta.get("tenant_id")
company_id = meta.get("company_id")
except Exception:
pass
if not tenant_id or not company_id:
return {"status": "failed", "error": "Falta contexto (tenant/company)"}
from api.v1.modules.a76.transportation.drivers.services import DriverService
from api.v1.modules.a76.transportation.drivers.dto import DriverCreateDTO
inserted_count = 0
updated_count = 0
skipped_invalid = 0
skipped_duplicate = 0
skipped_details: List[Dict[str, Any]] = []
seen_keys_in_file: Dict[str, int] = {}
try:
with CoreSessionLocal() as session:
with open(file_path, "r", encoding="utf-8-sig") as f:
sample = f.read(2048)
f.seek(0)
try:
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t")
except Exception:
dialect = "excel"
reader = csv.reader(f, dialect=dialect)
try:
headers = next(reader)
except StopIteration:
headers = []
headers = _dedupe_headers(headers)
dict_reader = csv.DictReader(f, fieldnames=headers, dialect=dialect)
for i, row in enumerate(dict_reader, start=1):
if i in error_lines:
continue
row_norm = row_from_template(row, normalize_header)
err = _validate_row_driver(row_norm, i)
if err:
skipped_invalid += 1
skipped_details.append(
{
"line": i,
"driver_key": (row_norm.get("CLAVE CONDUCTOR") or "").strip()[:80] or "-",
"invoice": (row_norm.get("CLAVE CONDUCTOR") or "").strip()[:80] or "-",
"reason": f"{err.get('col', '')}: {err.get('msg', '')}",
}
)
continue
data = _row_to_driver_dto(row_norm, tenant_id, company_id)
if not data or not data.get("transporter_key") or data.get("line") is None:
skipped_invalid += 1
continue
key = f"{data['transporter_key']}:{data['line']}"
if key in seen_keys_in_file:
skipped_duplicate += 1
skipped_details.append(
{
"line": i,
"driver_key": key,
"invoice": key,
"reason": "Clave duplicada en el archivo (se usa la primera)",
}
)
continue
seen_keys_in_file[key] = i
existing = DriverService.get_driver_by_key_and_line(
session, data["transporter_key"], data["line"], str(company_id), tenant_id
)
try:
if existing:
update_fields = {k: v for k, v in data.items() if k not in ("transporter_key", "line", "company_id", "tenant_id")}
for field, value in update_fields.items():
setattr(existing, field, value)
session.add(existing)
updated_count += 1
else:
create_data = DriverCreateDTO(**data)
DriverService.create_driver(session, create_data)
inserted_count += 1
except Exception as db_err:
session.rollback()
skipped_invalid += 1
skipped_details.append(
{"line": i, "driver_key": key, "invoice": key, "reason": str(db_err)}
)
continue
try:
session.commit()
except Exception as db_err:
session.rollback()
logger.error(f"Drivers import DB error: {db_err}")
return {"status": "failed", "error": str(db_err)}
except Exception as e:
logger.error(f"Drivers import task failed: {e}")
import traceback
logger.error(traceback.format_exc())
return {"status": "failed", "error": str(e)}
try:
if file_path and os.path.exists(file_path):
os.remove(file_path)
if os.path.exists(error_path):
os.remove(error_path)
if os.path.exists(meta_path):
os.remove(meta_path)
_delete_import_from_redis(job_id)
except Exception as cleanup_err:
logger.warning(f"Drivers import cleanup failed: {cleanup_err}")
total_ok = inserted_count + updated_count
if total_ok == 0 and (skipped_invalid + skipped_duplicate) > 0:
return {
"status": "warning",
"inserted": inserted_count,
"updated": updated_count,
"skipped_invalid": skipped_invalid,
"skipped_duplicate": skipped_duplicate,
"skipped_missing_fk": 0,
"skipped_details": skipped_details,
"message": f"No se insertaron registros. {skipped_invalid + skipped_duplicate} rechazados.",
}
if total_ok == 0:
return {
"status": "failed",
"error": "No hay registros validos en el archivo CSV",
"inserted": 0,
"updated": 0,
"skipped_invalid": skipped_invalid,
"skipped_duplicate": skipped_duplicate,
"skipped_missing_fk": 0,
"skipped_details": skipped_details,
}
return {
"status": "finished",
"inserted": inserted_count,
"updated": updated_count,
"skipped_invalid": skipped_invalid,
"skipped_duplicate": skipped_duplicate,
"skipped_missing_fk": 0,
"skipped_details": skipped_details,
}
def run_commit_sync(job_id: str) -> Dict[str, Any]:
result = _do_commit(job_id)
try:
r = _get_redis()
r.set(
f"{DRV_IMPORT_STATUS_PREFIX}{job_id}",
json.dumps(result).encode("utf-8"),
ex=DRV_IMPORT_REDIS_TTL,
)
except Exception as e:
logger.warning(f"Drivers import: failed to store commit status in Redis: {e}")
return result
@celery_app.task(bind=True)
def insert_valid_rows(self, job_id: str):
logger.info(f"Drivers import: starting commit for job {job_id}")
result = _do_commit(job_id)
try:
r = _get_redis()
r.set(
f"{DRV_IMPORT_STATUS_PREFIX}{job_id}",
json.dumps(result).encode("utf-8"),
ex=DRV_IMPORT_REDIS_TTL,
)
except Exception as e:
logger.warning(f"Drivers import: failed to store commit status in Redis: {e}")
return result