""" Tareas Celery para importación CSV de Clases de Materiales. Flujo: scan_file (validación) → insert_valid_rows (commit). """ import os import base64 import csv import json import logging import re import unicodedata from typing import Dict, Any, Optional, List, Set from core.celery_app import celery_app from core.database import CoreSessionLocal from core.paths import layout_path from .template_config import row_from_template logger = logging.getLogger(__name__) CLS_IMPORT_FILE_PREFIX = "cls_import_file:" CLS_IMPORT_META_PREFIX = "cls_import_meta:" CLS_IMPORT_ERROR_LINES_PREFIX = "cls_import_error_lines:" CLS_IMPORT_REDIS_TTL = 3600 def _get_redis(): import redis url = os.getenv("VALKEY_URL", os.getenv("REDIS_URL", "redis://valkey:6379/0")) return redis.Redis.from_url(url, decode_responses=False) def _worker_upload_dir() -> str: return layout_path("imports", "temp") def _ensure_worker_has_file_from_redis(job_id: str) -> Optional[str]: r = _get_redis() data = r.get(f"{CLS_IMPORT_FILE_PREFIX}{job_id}") if not data: return None try: raw = base64.b64decode(data) except Exception as e: logger.warning(f"Classes import: failed to decode file from Redis: {e}") return None upload_dir = _worker_upload_dir() os.makedirs(upload_dir, exist_ok=True) file_path = os.path.join(upload_dir, f"cls_{job_id}.csv") with open(file_path, "wb") as f: f.write(raw) return file_path def _ensure_worker_has_meta_from_redis(job_id: str, file_path: str) -> bool: r = _get_redis() data = r.get(f"{CLS_IMPORT_META_PREFIX}{job_id}") if not data: return False try: meta = json.loads(data.decode("utf-8")) except Exception as e: logger.warning(f"Classes import: failed to decode meta from Redis: {e}") return False meta_path = file_path.replace(".csv", ".meta.json") with open(meta_path, "w", encoding="utf-8") as f: json.dump(meta, f) return True def _delete_import_from_redis(job_id: str) -> None: try: r = _get_redis() r.delete( f"{CLS_IMPORT_FILE_PREFIX}{job_id}", f"{CLS_IMPORT_META_PREFIX}{job_id}", f"{CLS_IMPORT_ERROR_LINES_PREFIX}{job_id}", ) except Exception as e: logger.warning(f"Classes import: failed to delete Redis keys: {e}") def normalize_header(name: Optional[str]) -> str: if not name: return "" name = unicodedata.normalize("NFKD", str(name)).upper() name = "".join(ch for ch in name if not unicodedata.combining(ch)) name = re.sub(r"[^A-Z0-9]+", " ", name) return re.sub(r"\s+", " ", name).strip() def _validate_row_class( row: Dict[str, Any], line_num: int, valid_material_keys: Optional[Set[str]] = None, valid_uom_codes: Optional[Set[str]] = None, ) -> Optional[Dict[str, Any]]: class_code = (row.get("CLASE") or "").strip() if not class_code: return {"line": line_num, "col": "CLASE", "msg": "Requerido"} if len(class_code) > 8: return {"line": line_num, "col": "CLASE", "msg": "Máximo 8 caracteres"} desc_es = (row.get("DESCRIPCIONE") or "").strip() if desc_es and len(desc_es) > 500: return {"line": line_num, "col": "DESCRIPCIONE", "msg": "Máximo 500 caracteres"} desc_en = (row.get("DESCRIPCIONI") or "").strip() if desc_en and len(desc_en) > 500: return {"line": line_num, "col": "DESCRIPCIONI", "msg": "Máximo 500 caracteres"} material_key = (row.get("CLAVEMAT") or "").strip() if material_key: if len(material_key) > 10: return {"line": line_num, "col": "CLAVEMAT", "msg": "Máximo 10 caracteres"} if valid_material_keys is not None and material_key not in valid_material_keys: return {"line": line_num, "col": "CLAVEMAT", "msg": "Tipo de material no existe"} uom = (row.get("UNIMED") or "").strip() if uom: if len(uom) > 5: return {"line": line_num, "col": "UNIMED", "msg": "Máximo 5 caracteres"} if valid_uom_codes is not None and uom not in valid_uom_codes: return {"line": line_num, "col": "UNIMED", "msg": "Unidad de medida no existe"} fraction = (row.get("FRACCION") or "").strip() if fraction and len(fraction) > 20: return {"line": line_num, "col": "FRACCION", "msg": "Máximo 20 caracteres"} us_fraction = (row.get("FRACCIONAME") or "").strip() if us_fraction and len(us_fraction) > 16: return {"line": line_num, "col": "FRACCIONAME", "msg": "Máximo 16 caracteres"} sub_key = (row.get("CLAVESUB") or "").strip() if sub_key and len(sub_key) > 5: return {"line": line_num, "col": "CLAVESUB", "msg": "Máximo 5 caracteres"} iva_exempt = (row.get("FRACCIONEXENTAIVA") or "").strip() if iva_exempt and len(iva_exempt) > 4: return {"line": line_num, "col": "FRACCIONEXENTAIVA", "msg": "Máximo 4 caracteres"} rev_fisica = row.get("REVFISICA") if rev_fisica is not None and rev_fisica != "": try: v = int(rev_fisica) if v < -32768 or v > 32767: return {"line": line_num, "col": "REVFISICA", "msg": "Valor fuera de rango"} except (ValueError, TypeError): return {"line": line_num, "col": "REVFISICA", "msg": "Debe ser número entero"} return None @celery_app.task(bind=True) def scan_file(self, job_id: str, config: str = None): logger.info(f"Classes import: starting scan for job {job_id}") file_path = _ensure_worker_has_file_from_redis(job_id) if not file_path: return {"status": "failed", "error": "Archivo no encontrado (expirado o no subido). Sube de nuevo."} _ensure_worker_has_meta_from_redis(job_id, file_path) error_dir = layout_path("imports", "errors") os.makedirs(error_dir, exist_ok=True) error_path = os.path.join(error_dir, f"cls_{job_id}.jsonl") total_rows = 0 try: with open(file_path, "r", encoding="utf-8-sig") as f: total_rows = sum(1 for _ in f) - 1 except Exception as e: return {"status": "failed", "error": str(e)} meta_path = file_path.replace(".csv", ".meta.json") meta = {} if os.path.exists(meta_path): try: with open(meta_path, "r", encoding="utf-8") as f: meta = json.load(f) or {} except Exception as e: logger.warning(f"Classes import: failed to read meta: {e}") tenant_id = meta.get("tenant_id") company_id = meta.get("company_id") if not tenant_id or not company_id: return {"status": "failed", "error": "Falta contexto (tenant/company)"} valid_material_keys: Set[str] = set() valid_uom_codes: Set[str] = set() try: with CoreSessionLocal() as session: from api.v1.modules.public.reference_data.material_types.models import MaterialType from api.v1.modules.a76.general_catalogs.units_of_measure.models import UnitOfMeasure for m in session.query(MaterialType.key).all(): valid_material_keys.add(m[0]) for u in ( session.query(UnitOfMeasure.code) .filter( UnitOfMeasure.tenant_id == tenant_id, UnitOfMeasure.company_id == company_id, ) .all() ): valid_uom_codes.add(u[0]) except Exception as e: logger.warning(f"Classes import: could not load FK sets: {e}") error_count = 0 processed_rows = 0 errors_detail: List[Dict[str, Any]] = [] try: with open(file_path, "r", encoding="utf-8-sig") as f_in, open( error_path, "w", encoding="utf-8" ) as f_err: sample = f_in.read(2048) f_in.seek(0) try: dialect = csv.Sniffer().sniff(sample, delimiters=",;\t") except Exception: dialect = "excel" reader = csv.DictReader(f_in, dialect=dialect) for i, row in enumerate(reader, start=1): if i % 500 == 0: self.update_state( state="PROGRESS", meta={"current": i, "total": total_rows, "errors": error_count}, ) row_norm = row_from_template(row, normalize_header) err = _validate_row_class( row_norm, i, valid_material_keys=valid_material_keys, valid_uom_codes=valid_uom_codes, ) if err: error_count += 1 f_err.write(json.dumps(err) + "\n") if len(errors_detail) < 500: errors_detail.append( {"line": err["line"], "col": err.get("col", ""), "msg": err.get("msg", "")} ) processed_rows += 1 except Exception as e: logger.error(f"Classes import scan failed: {e}") return {"status": "failed", "error": str(e)} error_lines_list = [] try: if os.path.exists(error_path): with open(error_path, "r", encoding="utf-8") as f: for line in f: try: err = json.loads(line) if "line" in err: error_lines_list.append(err["line"]) except Exception: pass if error_lines_list: r = _get_redis() r.set( f"{CLS_IMPORT_ERROR_LINES_PREFIX}{job_id}", json.dumps(error_lines_list).encode("utf-8"), ex=CLS_IMPORT_REDIS_TTL, ) except Exception as e: logger.warning(f"Classes import: failed to store error lines in Redis: {e}") return { "status": "waiting_confirmation", "job_id": job_id, "total_rows": processed_rows, "error_count": error_count, "valid_rows": processed_rows - error_count, "errors": errors_detail, } def _str_or_none(val: Any, max_len: Optional[int] = None) -> Optional[str]: if val is None: return None s = str(val).strip() if not s: return None if max_len and len(s) > max_len: return s[:max_len] return s def _int_or_none(val: Any) -> Optional[int]: if val is None or val == "": return None try: return int(val) except (ValueError, TypeError): return None @celery_app.task(bind=True) def insert_valid_rows(self, job_id: str): logger.info(f"Classes import: starting commit for job {job_id}") file_path = _ensure_worker_has_file_from_redis(job_id) if not file_path: alt_path = os.path.join(_worker_upload_dir(), f"cls_{job_id}.csv") if not os.path.exists(alt_path): return { "status": "failed", "error": "Archivo no encontrado (expirado). Sube y confirma de nuevo.", } file_path = alt_path else: _ensure_worker_has_meta_from_redis(job_id, file_path) error_dir = layout_path("imports", "errors") error_path = os.path.join(error_dir, f"cls_{job_id}.jsonl") error_lines = set() try: r = _get_redis() raw = r.get(f"{CLS_IMPORT_ERROR_LINES_PREFIX}{job_id}") if raw: error_lines = set(json.loads(raw.decode("utf-8"))) except Exception as e: logger.debug(f"Classes import: could not load error lines from Redis: {e}") if not error_lines and os.path.exists(error_path): with open(error_path, "r", encoding="utf-8") as f: for line in f: try: err = json.loads(line) error_lines.add(err["line"]) except Exception: pass meta_path = file_path.replace(".csv", ".meta.json") tenant_id = None company_id = None meta = {} if os.path.exists(meta_path): try: with open(meta_path, "r", encoding="utf-8") as f: meta = json.load(f) or {} tenant_id = meta.get("tenant_id") company_id = meta.get("company_id") except Exception: pass if not tenant_id or not company_id: return {"status": "failed", "error": "Falta contexto (tenant/company)"} from api.v1.modules.a76.classes.models import Class valid_material_keys: Set[str] = set() valid_uom_codes: Set[str] = set() try: with CoreSessionLocal() as session: from api.v1.modules.public.reference_data.material_types.models import MaterialType from api.v1.modules.a76.general_catalogs.units_of_measure.models import UnitOfMeasure for m in session.query(MaterialType.key).all(): valid_material_keys.add(m[0]) for u in ( session.query(UnitOfMeasure.code) .filter( UnitOfMeasure.tenant_id == tenant_id, UnitOfMeasure.company_id == company_id, ) .all() ): valid_uom_codes.add(u[0]) except Exception as e: logger.warning(f"Classes import: could not load FK sets: {e}") inserted_count = 0 skipped_invalid = 0 skipped_details: List[Dict[str, Any]] = [] response = None try: with CoreSessionLocal() as session: existing_by_code: Dict[str, Class] = {} for c in ( session.query(Class) .filter( Class.tenant_id == tenant_id, Class.company_id == company_id, ) .all() ): existing_by_code[c.class_code] = c with open(file_path, "r", encoding="utf-8-sig") as f: sample = f.read(2048) f.seek(0) try: dialect = csv.Sniffer().sniff(sample, delimiters=",;\t") except Exception: dialect = "excel" reader = csv.DictReader(f, dialect=dialect) for i, row in enumerate(reader, start=1): if i in error_lines: continue row_norm = row_from_template(row, normalize_header) err = _validate_row_class( row_norm, i, valid_material_keys=valid_material_keys, valid_uom_codes=valid_uom_codes, ) if err: skipped_invalid += 1 skipped_details.append( {"line": i, "reason": f"{err.get('col', '')}: {err.get('msg', '')}"} ) continue class_code = _str_or_none(row_norm.get("CLASE"), 8) if not class_code: skipped_invalid += 1 continue existing = existing_by_code.get(class_code) desc_es = _str_or_none(row_norm.get("DESCRIPCIONE"), 500) desc_en = _str_or_none(row_norm.get("DESCRIPCIONI"), 500) material_key = _str_or_none(row_norm.get("CLAVEMAT"), 10) if material_key and material_key not in valid_material_keys: material_key = None unit_of_measure = _str_or_none(row_norm.get("UNIMED"), 5) if unit_of_measure and unit_of_measure not in valid_uom_codes: unit_of_measure = None fraction = _str_or_none(row_norm.get("FRACCION"), 20) us_fraction = _str_or_none(row_norm.get("FRACCIONAME"), 16) sub_key = _str_or_none(row_norm.get("CLAVESUB"), 5) physical_review = _int_or_none(row_norm.get("REVFISICA")) iva_exempt_fraction = _str_or_none(row_norm.get("FRACCIONEXENTAIVA"), 4) if existing: existing.description_es = desc_es existing.description_en = desc_en existing.material_key = material_key existing.unit_of_measure = unit_of_measure existing.fraction = fraction existing.us_fraction = us_fraction existing.sub_key = sub_key existing.physical_review = physical_review existing.iva_exempt_fraction = iva_exempt_fraction session.add(existing) inserted_count += 1 else: new_class = Class( tenant_id=tenant_id, company_id=company_id, class_code=class_code, description_es=desc_es, description_en=desc_en, material_key=material_key, unit_of_measure=unit_of_measure, fraction=fraction, us_fraction=us_fraction, sub_key=sub_key, physical_review=physical_review, iva_exempt_fraction=iva_exempt_fraction, ) session.add(new_class) existing_by_code[class_code] = new_class inserted_count += 1 try: session.commit() except Exception as db_err: session.rollback() logger.error(f"Classes import DB error: {db_err}") return {"status": "failed", "error": str(db_err)} total_skipped = skipped_invalid if inserted_count == 0 and total_skipped > 0: response = { "status": "warning", "inserted": 0, "skipped_invalid": skipped_invalid, "skipped_missing_fk": 0, "skipped_details": skipped_details, "message": f"No se insertaron registros. {total_skipped} rechazados.", } elif inserted_count == 0: response = { "status": "failed", "error": "No hay registros válidos en el archivo CSV", "inserted": 0, "skipped_invalid": skipped_invalid, "skipped_missing_fk": 0, "skipped_details": skipped_details, } else: response = { "status": "finished", "inserted": inserted_count, "skipped_invalid": skipped_invalid, "skipped_missing_fk": 0, "skipped_details": skipped_details, } except Exception as e: logger.error(f"Classes import task failed: {e}") import traceback logger.error(traceback.format_exc()) return {"status": "failed", "error": str(e)} try: if file_path and os.path.exists(file_path): os.remove(file_path) if os.path.exists(error_path): os.remove(error_path) if os.path.exists(meta_path): os.remove(meta_path) _delete_import_from_redis(job_id) except Exception as cleanup_err: logger.warning(f"Classes import cleanup failed: {cleanup_err}") if response is None: response = { "status": "failed", "error": "Error inesperado", "inserted": 0, "skipped_invalid": skipped_invalid, "skipped_missing_fk": 0, "skipped_details": skipped_details, } return response