""" Tareas Celery para importacion CSV de Conductores. Flujo: scan_file (validacion) -> insert_valid_rows (commit). """ import os import base64 import csv import json import logging import re import unicodedata from typing import Dict, Any, Optional, List from core.celery_app import celery_app from core.database import CoreSessionLocal from core.paths import layout_path from .template_config import row_from_template logger = logging.getLogger(__name__) DRV_IMPORT_FILE_PREFIX = "drv_import_file:" DRV_IMPORT_META_PREFIX = "drv_import_meta:" DRV_IMPORT_ERROR_LINES_PREFIX = "drv_import_error_lines:" DRV_IMPORT_STATUS_PREFIX = "drv_import_status:" DRV_IMPORT_REDIS_TTL = 3600 def _get_redis(): import redis url = os.getenv("VALKEY_URL", os.getenv("REDIS_URL", "redis://valkey:6379/0")) return redis.Redis.from_url(url, decode_responses=False) def _worker_upload_dir() -> str: return layout_path("imports", "temp") def _ensure_worker_has_file_from_redis(job_id: str) -> Optional[str]: r = _get_redis() data = r.get(f"{DRV_IMPORT_FILE_PREFIX}{job_id}") if not data: return None try: raw = base64.b64decode(data) except Exception as e: logger.warning(f"Drivers import: failed to decode file from Redis: {e}") return None upload_dir = _worker_upload_dir() os.makedirs(upload_dir, exist_ok=True) file_path = os.path.join(upload_dir, f"drv_{job_id}.csv") with open(file_path, "wb") as f: f.write(raw) return file_path def _ensure_worker_has_meta_from_redis(job_id: str, file_path: str) -> bool: r = _get_redis() data = r.get(f"{DRV_IMPORT_META_PREFIX}{job_id}") if not data: return False try: meta = json.loads(data.decode("utf-8")) except Exception as e: logger.warning(f"Drivers import: failed to decode meta from Redis: {e}") return False meta_path = file_path.replace(".csv", ".meta.json") with open(meta_path, "w", encoding="utf-8") as f: json.dump(meta, f) return True def _delete_import_from_redis(job_id: str) -> None: try: r = _get_redis() r.delete( f"{DRV_IMPORT_FILE_PREFIX}{job_id}", f"{DRV_IMPORT_META_PREFIX}{job_id}", f"{DRV_IMPORT_ERROR_LINES_PREFIX}{job_id}", f"{DRV_IMPORT_STATUS_PREFIX}{job_id}", ) except Exception as e: logger.warning(f"Drivers import: failed to delete Redis keys: {e}") def normalize_header(name: Optional[str]) -> str: if not name: return "" name = unicodedata.normalize("NFKD", str(name)).upper() name = "".join(ch for ch in name if not unicodedata.combining(ch)) name = re.sub(r"[^A-Z0-9]+", " ", name) return re.sub(r"\s+", " ", name).strip() _MAX = { "transporter_key": 5, "driver_name": 80, "license_number": 29, "express_line_id": 17, "ace_id": 20, "gender": 1, "birth_country": 3, "hazardous_material_auth": 2, "hazardous_material_state": 30, "first_name": 20, "last_name": 20, "id_key1": 40, "id_number1": 20, "id_state1": 30, "id_country1": 3, "id_key2": 40, "id_number2": 20, "id_state2": 30, "id_country2": 3, "badge_number": 20, "class_type": 1, "unique_badge_number": 100, } def _parse_int(val: Any) -> Optional[int]: if val is None or (isinstance(val, str) and not val.strip()): return None s = str(val).strip() if re.match(r"^\d+$", s): return int(s) if re.match(r"^\d+\.0+$", s): try: return int(float(s)) except ValueError: return None return None def _parse_birth_date(val: Any) -> Optional[int]: if val is None or (isinstance(val, str) and not val.strip()): return None s = str(val).strip() if re.match(r"^\d{8}$", s): try: return int(s) except ValueError: return None if re.match(r"^\d+\.0+$", s): try: return int(float(s)) except ValueError: return None for sep in ["/", "-", "."]: if sep in s: parts = s.split(sep) if len(parts) == 3: try: a, b, c = [p.strip() for p in parts] if len(c) == 4 and len(a) <= 2 and len(b) <= 2: return int(c) * 10000 + int(b) * 100 + int(a) if len(a) == 4 and len(b) <= 2 and len(c) <= 2: return int(a) * 10000 + int(b) * 100 + int(c) except (ValueError, TypeError): return None break return None def _str_or_none(val: Any, max_len: Optional[int] = None) -> Optional[str]: if val is None: return None s = str(val).strip() if not s: return None if max_len and len(s) > max_len: return s[:max_len] return s def _dedupe_headers(headers: List[str]) -> List[str]: counts: Dict[str, int] = {} unique: List[str] = [] for header in headers: name = str(header or "").strip() or "COL" count = counts.get(name, 0) + 1 counts[name] = count if count == 1: unique.append(name) else: unique.append(f"{name} {count}") return unique def _validate_row_driver(row: Dict[str, Any], line_num: int) -> Optional[Dict[str, Any]]: transporter_key = (row.get("TRANSPORTISTA") or "").strip() if not transporter_key: return {"line": line_num, "col": "TRANSPORTISTA", "msg": "Requerido"} if len(transporter_key) > _MAX["transporter_key"]: return {"line": line_num, "col": "TRANSPORTISTA", "msg": f"Maximo {_MAX['transporter_key']} caracteres"} line_val = _parse_int(row.get("LINEA")) if line_val is None: return {"line": line_num, "col": "LINEA", "msg": "Debe ser numerico"} if line_val <= 0: return {"line": line_num, "col": "LINEA", "msg": "Debe ser mayor a 0"} driver_name = (row.get("CLAVE CONDUCTOR") or "").strip() if driver_name and len(driver_name) > _MAX["driver_name"]: return {"line": line_num, "col": "CLAVE CONDUCTOR", "msg": f"Maximo {_MAX['driver_name']} caracteres"} for col, max_len in [ ("LICENCIA", _MAX["license_number"]), ("PERMISO LINEA EXPRESS", _MAX["express_line_id"]), ("IDENTIFICACION ACE", _MAX["ace_id"]), ("SEXO", _MAX["gender"]), ("PAIS NACIMIENTO", _MAX["birth_country"]), ("TRANSPORTA MAT. PELIGROSO?", _MAX["hazardous_material_auth"]), ("PERMISO MAT. PELIGROSO", _MAX["hazardous_material_state"]), ("NOMBRE(S)", _MAX["first_name"]), ("APELLIDO PATERNO", _MAX["last_name"]), ("FORMA IDENTIFICACION 1", _MAX["id_key1"]), ("NUM. IDENTIFICACION 1", _MAX["id_number1"]), ("ESTADO", _MAX["id_state1"]), ("PAIS", _MAX["id_country1"]), ("FORMA IDENTIFICACION 2", _MAX["id_key2"]), ("NUM. IDENTIFICACION 2", _MAX["id_number2"]), ("ESTADO 2", _MAX["id_state2"]), ("PAIS 2", _MAX["id_country2"]), ]: val = (row.get(col) or "").strip() if val and len(val) > max_len: return {"line": line_num, "col": col, "msg": f"Maximo {max_len} caracteres"} fecha = row.get("FECHA NACIMIENTO") if fecha is not None and str(fecha).strip(): if _parse_birth_date(fecha) is None: return { "line": line_num, "col": "FECHA NACIMIENTO", "msg": "Formato de fecha invalido (use YYYYMMDD o DD/MM/YYYY)", } return None def _row_to_driver_dto(row: Dict[str, Any], tenant_id: int, company_id: int) -> Dict[str, Any]: transporter_key = _str_or_none(row.get("TRANSPORTISTA"), _MAX["transporter_key"]) line = _parse_int(row.get("LINEA")) if not transporter_key or line is None: return {} data = { "transporter_key": transporter_key, "line": line, "driver_name": _str_or_none(row.get("CLAVE CONDUCTOR"), _MAX["driver_name"]), "license_number": _str_or_none(row.get("LICENCIA"), _MAX["license_number"]), "express_line_id": _str_or_none(row.get("PERMISO LINEA EXPRESS"), _MAX["express_line_id"]), "ace_id": _str_or_none(row.get("IDENTIFICACION ACE"), _MAX["ace_id"]), "birth_date": _parse_birth_date(row.get("FECHA NACIMIENTO")), "gender": _str_or_none(row.get("SEXO"), _MAX["gender"]), "birth_country": _str_or_none(row.get("PAIS NACIMIENTO"), _MAX["birth_country"]), "hazardous_material_auth": _str_or_none( row.get("TRANSPORTA MAT. PELIGROSO?"), _MAX["hazardous_material_auth"] ), "hazardous_material_state": _str_or_none( row.get("PERMISO MAT. PELIGROSO"), _MAX["hazardous_material_state"] ), "first_name": _str_or_none(row.get("NOMBRE(S)"), _MAX["first_name"]), "last_name": _str_or_none(row.get("APELLIDO PATERNO"), _MAX["last_name"]), "id_key1": _str_or_none(row.get("FORMA IDENTIFICACION 1"), _MAX["id_key1"]), "id_number1": _str_or_none(row.get("NUM. IDENTIFICACION 1"), _MAX["id_number1"]), "id_state1": _str_or_none(row.get("ESTADO"), _MAX["id_state1"]), "id_country1": _str_or_none(row.get("PAIS"), _MAX["id_country1"]), "id_key2": _str_or_none(row.get("FORMA IDENTIFICACION 2"), _MAX["id_key2"]), "id_number2": _str_or_none(row.get("NUM. IDENTIFICACION 2"), _MAX["id_number2"]), "id_state2": _str_or_none(row.get("ESTADO 2"), _MAX["id_state2"]), "id_country2": _str_or_none(row.get("PAIS 2"), _MAX["id_country2"]), "company_id": company_id, "tenant_id": tenant_id, } return data def _do_scan(job_id: str, progress_callback: Optional[Any] = None) -> Dict[str, Any]: file_path = _ensure_worker_has_file_from_redis(job_id) if not file_path: return {"status": "failed", "error": "Archivo no encontrado (expirado o no subido). Sube de nuevo."} _ensure_worker_has_meta_from_redis(job_id, file_path) error_dir = layout_path("imports", "errors") os.makedirs(error_dir, exist_ok=True) error_path = os.path.join(error_dir, f"drv_{job_id}.jsonl") total_rows = 0 try: with open(file_path, "r", encoding="utf-8-sig") as f: total_rows = sum(1 for _ in f) - 1 except Exception as e: return {"status": "failed", "error": str(e)} meta_path = file_path.replace(".csv", ".meta.json") meta = {} if os.path.exists(meta_path): try: with open(meta_path, "r", encoding="utf-8") as f: meta = json.load(f) or {} except Exception as e: logger.warning(f"Drivers import: failed to read meta: {e}") tenant_id = meta.get("tenant_id") company_id = meta.get("company_id") if not tenant_id or not company_id: return {"status": "failed", "error": "Falta contexto (tenant/company)"} error_count = 0 processed_rows = 0 errors_detail: List[Dict[str, Any]] = [] try: with open(file_path, "r", encoding="utf-8-sig") as f_in, open( error_path, "w", encoding="utf-8" ) as f_err: sample = f_in.read(2048) f_in.seek(0) try: dialect = csv.Sniffer().sniff(sample, delimiters=",;\t") except Exception: dialect = "excel" reader = csv.reader(f_in, dialect=dialect) try: headers = next(reader) except StopIteration: headers = [] headers = _dedupe_headers(headers) dict_reader = csv.DictReader(f_in, fieldnames=headers, dialect=dialect) for i, row in enumerate(dict_reader, start=1): if progress_callback and i % 500 == 0: progress_callback(i, total_rows, error_count) row_norm = row_from_template(row, normalize_header) err = _validate_row_driver(row_norm, i) if err: error_count += 1 f_err.write(json.dumps(err) + "\n") if len(errors_detail) < 500: errors_detail.append( {"line": err["line"], "col": err.get("col", ""), "msg": err.get("msg", "")} ) processed_rows += 1 except Exception as e: logger.error(f"Drivers import scan failed: {e}") return {"status": "failed", "error": str(e)} error_lines_list = [] try: if os.path.exists(error_path): with open(error_path, "r", encoding="utf-8") as f: for line in f: try: err = json.loads(line) if "line" in err: error_lines_list.append(err["line"]) except Exception: pass if error_lines_list: r = _get_redis() r.set( f"{DRV_IMPORT_ERROR_LINES_PREFIX}{job_id}", json.dumps(error_lines_list).encode("utf-8"), ex=DRV_IMPORT_REDIS_TTL, ) except Exception as e: logger.warning(f"Drivers import: failed to store error lines: {e}") result = { "status": "waiting_confirmation", "job_id": job_id, "total_rows": processed_rows, "error_count": error_count, "valid_rows": processed_rows - error_count, "errors": errors_detail, } return result def run_scan_sync(job_id: str) -> Dict[str, Any]: result = _do_scan(job_id, progress_callback=None) try: r = _get_redis() r.set( f"{DRV_IMPORT_STATUS_PREFIX}{job_id}", json.dumps(result).encode("utf-8"), ex=DRV_IMPORT_REDIS_TTL, ) except Exception as e: logger.warning(f"Drivers import: failed to store scan status in Redis: {e}") return result @celery_app.task(bind=True) def scan_file(self, job_id: str, config: str = None): logger.info(f"Drivers import: starting scan for job {job_id}") def on_progress(current: int, total: int, errors: int) -> None: self.update_state( state="PROGRESS", meta={"current": current, "total": total, "errors": errors}, ) result = _do_scan(job_id, progress_callback=on_progress) try: r = _get_redis() r.set( f"{DRV_IMPORT_STATUS_PREFIX}{job_id}", json.dumps(result).encode("utf-8"), ex=DRV_IMPORT_REDIS_TTL, ) except Exception as e: logger.warning(f"Drivers import: failed to store scan status in Redis: {e}") return result def _do_commit(job_id: str) -> Dict[str, Any]: file_path = _ensure_worker_has_file_from_redis(job_id) if not file_path: alt_path = os.path.join(_worker_upload_dir(), f"drv_{job_id}.csv") if not os.path.exists(alt_path): return { "status": "failed", "error": "Archivo no encontrado (expirado). Sube y confirma de nuevo.", } file_path = alt_path else: _ensure_worker_has_meta_from_redis(job_id, file_path) error_dir = layout_path("imports", "errors") error_path = os.path.join(error_dir, f"drv_{job_id}.jsonl") error_lines = set() try: r = _get_redis() raw = r.get(f"{DRV_IMPORT_ERROR_LINES_PREFIX}{job_id}") if raw: error_lines = set(json.loads(raw.decode("utf-8"))) except Exception as e: logger.debug(f"Drivers import: could not load error lines from Redis: {e}") if not error_lines and os.path.exists(error_path): with open(error_path, "r", encoding="utf-8") as f: for line in f: try: err = json.loads(line) error_lines.add(err["line"]) except Exception: pass meta_path = file_path.replace(".csv", ".meta.json") tenant_id = None company_id = None meta = {} if os.path.exists(meta_path): try: with open(meta_path, "r", encoding="utf-8") as f: meta = json.load(f) or {} tenant_id = meta.get("tenant_id") company_id = meta.get("company_id") except Exception: pass if not tenant_id or not company_id: return {"status": "failed", "error": "Falta contexto (tenant/company)"} from api.v1.modules.a76.transportation.drivers.services import DriverService from api.v1.modules.a76.transportation.drivers.dto import DriverCreateDTO inserted_count = 0 updated_count = 0 skipped_invalid = 0 skipped_duplicate = 0 skipped_details: List[Dict[str, Any]] = [] seen_keys_in_file: Dict[str, int] = {} try: with CoreSessionLocal() as session: with open(file_path, "r", encoding="utf-8-sig") as f: sample = f.read(2048) f.seek(0) try: dialect = csv.Sniffer().sniff(sample, delimiters=",;\t") except Exception: dialect = "excel" reader = csv.reader(f, dialect=dialect) try: headers = next(reader) except StopIteration: headers = [] headers = _dedupe_headers(headers) dict_reader = csv.DictReader(f, fieldnames=headers, dialect=dialect) for i, row in enumerate(dict_reader, start=1): if i in error_lines: continue row_norm = row_from_template(row, normalize_header) err = _validate_row_driver(row_norm, i) if err: skipped_invalid += 1 skipped_details.append( { "line": i, "driver_key": (row_norm.get("CLAVE CONDUCTOR") or "").strip()[:80] or "-", "invoice": (row_norm.get("CLAVE CONDUCTOR") or "").strip()[:80] or "-", "reason": f"{err.get('col', '')}: {err.get('msg', '')}", } ) continue data = _row_to_driver_dto(row_norm, tenant_id, company_id) if not data or not data.get("transporter_key") or data.get("line") is None: skipped_invalid += 1 continue key = f"{data['transporter_key']}:{data['line']}" if key in seen_keys_in_file: skipped_duplicate += 1 skipped_details.append( { "line": i, "driver_key": key, "invoice": key, "reason": "Clave duplicada en el archivo (se usa la primera)", } ) continue seen_keys_in_file[key] = i existing = DriverService.get_driver_by_key_and_line( session, data["transporter_key"], data["line"], str(company_id), tenant_id ) try: if existing: update_fields = {k: v for k, v in data.items() if k not in ("transporter_key", "line", "company_id", "tenant_id")} for field, value in update_fields.items(): setattr(existing, field, value) session.add(existing) updated_count += 1 else: create_data = DriverCreateDTO(**data) DriverService.create_driver(session, create_data) inserted_count += 1 except Exception as db_err: session.rollback() skipped_invalid += 1 skipped_details.append( {"line": i, "driver_key": key, "invoice": key, "reason": str(db_err)} ) continue try: session.commit() except Exception as db_err: session.rollback() logger.error(f"Drivers import DB error: {db_err}") return {"status": "failed", "error": str(db_err)} except Exception as e: logger.error(f"Drivers import task failed: {e}") import traceback logger.error(traceback.format_exc()) return {"status": "failed", "error": str(e)} try: if file_path and os.path.exists(file_path): os.remove(file_path) if os.path.exists(error_path): os.remove(error_path) if os.path.exists(meta_path): os.remove(meta_path) _delete_import_from_redis(job_id) except Exception as cleanup_err: logger.warning(f"Drivers import cleanup failed: {cleanup_err}") total_ok = inserted_count + updated_count if total_ok == 0 and (skipped_invalid + skipped_duplicate) > 0: return { "status": "warning", "inserted": inserted_count, "updated": updated_count, "skipped_invalid": skipped_invalid, "skipped_duplicate": skipped_duplicate, "skipped_missing_fk": 0, "skipped_details": skipped_details, "message": f"No se insertaron registros. {skipped_invalid + skipped_duplicate} rechazados.", } if total_ok == 0: return { "status": "failed", "error": "No hay registros validos en el archivo CSV", "inserted": 0, "updated": 0, "skipped_invalid": skipped_invalid, "skipped_duplicate": skipped_duplicate, "skipped_missing_fk": 0, "skipped_details": skipped_details, } return { "status": "finished", "inserted": inserted_count, "updated": updated_count, "skipped_invalid": skipped_invalid, "skipped_duplicate": skipped_duplicate, "skipped_missing_fk": 0, "skipped_details": skipped_details, } def run_commit_sync(job_id: str) -> Dict[str, Any]: result = _do_commit(job_id) try: r = _get_redis() r.set( f"{DRV_IMPORT_STATUS_PREFIX}{job_id}", json.dumps(result).encode("utf-8"), ex=DRV_IMPORT_REDIS_TTL, ) except Exception as e: logger.warning(f"Drivers import: failed to store commit status in Redis: {e}") return result @celery_app.task(bind=True) def insert_valid_rows(self, job_id: str): logger.info(f"Drivers import: starting commit for job {job_id}") result = _do_commit(job_id) try: r = _get_redis() r.set( f"{DRV_IMPORT_STATUS_PREFIX}{job_id}", json.dumps(result).encode("utf-8"), ex=DRV_IMPORT_REDIS_TTL, ) except Exception as e: logger.warning(f"Drivers import: failed to store commit status in Redis: {e}") return result