Files
plantillas-proyectos/backend/api/v1/modules/a76/layouts_csv/parts/tasks.py
2026-03-03 16:45:12 -07:00

640 lines
25 KiB
Python

"""
Tareas Celery para importación CSV de Números de Parte.
Flujo: scan_file (validación) → insert_valid_rows (commit).
"""
import os
import base64
import csv
import json
import logging
import re
import unicodedata
from decimal import Decimal, InvalidOperation
from typing import Dict, Any, Optional, List, Set
from core.celery_app import celery_app
from core.database import CoreSessionLocal
from core.paths import layout_path
from .template_config import row_from_template
logger = logging.getLogger(__name__)
PART_IMPORT_FILE_PREFIX = "part_import_file:"
PART_IMPORT_META_PREFIX = "part_import_meta:"
PART_IMPORT_ERROR_LINES_PREFIX = "part_import_error_lines:"
PART_IMPORT_REDIS_TTL = 3600
def _get_redis():
import redis
url = os.getenv("VALKEY_URL", os.getenv("REDIS_URL", "redis://valkey:6379/0"))
return redis.Redis.from_url(url, decode_responses=False)
def _worker_upload_dir() -> str:
return layout_path("imports", "temp")
def _ensure_worker_has_file_from_redis(job_id: str) -> Optional[str]:
r = _get_redis()
data = r.get(f"{PART_IMPORT_FILE_PREFIX}{job_id}")
if not data:
return None
try:
raw = base64.b64decode(data)
except Exception as e:
logger.warning(f"Parts import: failed to decode file from Redis: {e}")
return None
upload_dir = _worker_upload_dir()
os.makedirs(upload_dir, exist_ok=True)
file_path = os.path.join(upload_dir, f"part_{job_id}.csv")
with open(file_path, "wb") as f:
f.write(raw)
return file_path
def _ensure_worker_has_meta_from_redis(job_id: str, file_path: str) -> bool:
r = _get_redis()
data = r.get(f"{PART_IMPORT_META_PREFIX}{job_id}")
if not data:
return False
try:
meta = json.loads(data.decode("utf-8"))
except Exception as e:
logger.warning(f"Parts import: failed to decode meta from Redis: {e}")
return False
meta_path = file_path.replace(".csv", ".meta.json")
with open(meta_path, "w", encoding="utf-8") as f:
json.dump(meta, f)
return True
def _delete_import_from_redis(job_id: str) -> None:
try:
r = _get_redis()
r.delete(
f"{PART_IMPORT_FILE_PREFIX}{job_id}",
f"{PART_IMPORT_META_PREFIX}{job_id}",
f"{PART_IMPORT_ERROR_LINES_PREFIX}{job_id}",
)
except Exception as e:
logger.warning(f"Parts import: failed to delete Redis keys: {e}")
def normalize_header(name: Optional[str]) -> str:
if not name:
return ""
name = unicodedata.normalize("NFKD", str(name)).upper()
name = "".join(ch for ch in name if not unicodedata.combining(ch))
name = re.sub(r"[^A-Z0-9]+", " ", name)
return re.sub(r"\s+", " ", name).strip()
def _validate_row_part(
row: Dict[str, Any],
line_num: int,
valid_class_codes: Optional[Set[str]] = None,
valid_uom_codes: Optional[Set[str]] = None,
valid_currency_codes: Optional[Set[str]] = None,
) -> Optional[Dict[str, Any]]:
part_number = (row.get("NUMPARTE") or "").strip()
if not part_number:
return {"line": line_num, "col": "NUMPARTE", "msg": "Requerido"}
if len(part_number) > 70:
return {"line": line_num, "col": "NUMPARTE", "msg": "Máximo 70 caracteres"}
commercial = (row.get("NUMPARTECOM") or "").strip()
if commercial and len(commercial) > 70:
return {"line": line_num, "col": "NUMPARTECOM", "msg": "Máximo 70 caracteres"}
desc_es = (row.get("DESCRIPCIONE") or "").strip()
if desc_es and len(desc_es) > 500:
return {"line": line_num, "col": "DESCRIPCIONE", "msg": "Máximo 500 caracteres"}
desc_en = (row.get("DESCRIPCIONI") or "").strip()
if desc_en and len(desc_en) > 500:
return {"line": line_num, "col": "DESCRIPCIONI", "msg": "Máximo 500 caracteres"}
part_class = (row.get("CLASE") or "").strip()
if part_class and len(part_class) > 8:
return {"line": line_num, "col": "CLASE", "msg": "Máximo 8 caracteres"}
# Si CLASE no existe en catálogo se guardará null (no se rechaza la fila)
uom = (row.get("UNIMED") or "").strip()
if uom and len(uom) > 5:
return {"line": line_num, "col": "UNIMED", "msg": "Máximo 5 caracteres"}
# Si UNIMED no existe en catálogo se guardará null (no se rechaza la fila)
currency_key = (row.get("MONEDA") or "").strip()
if currency_key and len(currency_key) > 3:
return {"line": line_num, "col": "MONEDA", "msg": "Máximo 3 caracteres"}
# Si MONEDA no existe en catálogo se guardará null (no se rechaza la fila)
unit_cost = row.get("COSTOUNIT")
if unit_cost is not None and unit_cost != "":
try:
Decimal(str(unit_cost))
except (InvalidOperation, ValueError, TypeError):
return {"line": line_num, "col": "COSTOUNIT", "msg": "Debe ser número"}
unit_weight = row.get("PESOUNIT")
if unit_weight is not None and unit_weight != "":
try:
Decimal(str(unit_weight))
except (InvalidOperation, ValueError, TypeError):
return {"line": line_num, "col": "PESOUNIT", "msg": "Debe ser número"}
fraction = (row.get("FRACCION") or "").strip()
if fraction and len(fraction) > 10:
return {"line": line_num, "col": "FRACCION", "msg": "Máximo 10 caracteres"}
us_fraction = (row.get("FRACCIONAME") or "").strip()
if us_fraction and len(us_fraction) > 16:
return {"line": line_num, "col": "FRACCIONAME", "msg": "Máximo 16 caracteres"}
fda_key = (row.get("FDAKEY") or "").strip()
if fda_key and len(fda_key) > 20:
return {"line": line_num, "col": "FDAKEY", "msg": "Máximo 20 caracteres"}
fcc_key = (row.get("FCCKEY") or "").strip()
if fcc_key and len(fcc_key) > 30:
return {"line": line_num, "col": "FCCKEY", "msg": "Máximo 30 caracteres"}
license_code = (row.get("LICENCIA") or "").strip()
if license_code and len(license_code) > 3:
return {"line": line_num, "col": "LICENCIA", "msg": "Máximo 3 caracteres"}
eccn = (row.get("ECCN") or "").strip()
if eccn and len(eccn) > 20:
return {"line": line_num, "col": "ECCN", "msg": "Máximo 20 caracteres"}
export_code = (row.get("EXPORTCODE") or "").strip()
if export_code and len(export_code) > 2:
return {"line": line_num, "col": "EXPORTCODE", "msg": "Máximo 2 caracteres"}
exclusion = (row.get("EXCLUSION") or "").strip()
if exclusion and len(exclusion) > 19:
return {"line": line_num, "col": "EXCLUSION", "msg": "Máximo 19 caracteres"}
weight_type = (row.get("TIPOPESO") or "").strip()
if weight_type and len(weight_type) > 6:
return {"line": line_num, "col": "TIPOPESO", "msg": "Máximo 6 caracteres"}
return None
@celery_app.task(bind=True)
def scan_file(self, job_id: str, config: str = None):
logger.info(f"Parts import: starting scan for job {job_id}")
file_path = _ensure_worker_has_file_from_redis(job_id)
if not file_path:
return {"status": "failed", "error": "Archivo no encontrado (expirado o no subido). Sube de nuevo."}
_ensure_worker_has_meta_from_redis(job_id, file_path)
error_dir = layout_path("imports", "errors")
os.makedirs(error_dir, exist_ok=True)
error_path = os.path.join(error_dir, f"part_{job_id}.jsonl")
total_rows = 0
try:
with open(file_path, "r", encoding="utf-8-sig") as f:
total_rows = sum(1 for _ in f) - 1
except Exception as e:
return {"status": "failed", "error": str(e)}
meta_path = file_path.replace(".csv", ".meta.json")
meta = {}
if os.path.exists(meta_path):
try:
with open(meta_path, "r", encoding="utf-8") as f:
meta = json.load(f) or {}
except Exception as e:
logger.warning(f"Parts import: failed to read meta: {e}")
tenant_id = meta.get("tenant_id")
company_id = meta.get("company_id")
if not tenant_id or not company_id:
return {"status": "failed", "error": "Falta contexto (tenant/company)"}
valid_class_codes: Set[str] = set()
valid_uom_codes: Set[str] = set()
valid_currency_codes: Set[str] = set()
try:
with CoreSessionLocal() as session:
from api.v1.modules.a76.classes.models import Class
from api.v1.modules.a76.general_catalogs.units_of_measure.models import UnitOfMeasure
from api.v1.modules.public.reference_data.currency_types.models import CurrencyType
for c in (
session.query(Class.class_code)
.filter(
Class.tenant_id == tenant_id,
Class.company_id == company_id,
)
.all()
):
valid_class_codes.add(c[0])
for u in (
session.query(UnitOfMeasure.code)
.filter(
UnitOfMeasure.tenant_id == tenant_id,
UnitOfMeasure.company_id == company_id,
)
.all()
):
valid_uom_codes.add(u[0])
for cur in session.query(CurrencyType.code).all():
valid_currency_codes.add(cur[0])
except Exception as e:
logger.warning(f"Parts import: could not load FK sets: {e}")
error_count = 0
processed_rows = 0
errors_detail: List[Dict[str, Any]] = []
try:
with open(file_path, "r", encoding="utf-8-sig") as f_in, open(
error_path, "w", encoding="utf-8"
) as f_err:
sample = f_in.read(2048)
f_in.seek(0)
try:
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t")
except Exception:
dialect = "excel"
reader = csv.DictReader(f_in, dialect=dialect)
for i, row in enumerate(reader, start=1):
if i % 500 == 0:
self.update_state(
state="PROGRESS",
meta={"current": i, "total": total_rows, "errors": error_count},
)
row_norm = row_from_template(row, normalize_header)
err = _validate_row_part(
row_norm,
i,
valid_class_codes=valid_class_codes,
valid_uom_codes=valid_uom_codes,
valid_currency_codes=valid_currency_codes,
)
if err:
error_count += 1
f_err.write(json.dumps(err) + "\n")
if len(errors_detail) < 500:
errors_detail.append(
{"line": err["line"], "col": err.get("col", ""), "msg": err.get("msg", "")}
)
processed_rows += 1
except Exception as e:
logger.error(f"Parts import scan failed: {e}")
return {"status": "failed", "error": str(e)}
error_lines_list = []
try:
if os.path.exists(error_path):
with open(error_path, "r", encoding="utf-8") as f:
for line in f:
try:
err = json.loads(line)
if "line" in err:
error_lines_list.append(err["line"])
except Exception:
pass
if error_lines_list:
r = _get_redis()
r.set(
f"{PART_IMPORT_ERROR_LINES_PREFIX}{job_id}",
json.dumps(error_lines_list).encode("utf-8"),
ex=PART_IMPORT_REDIS_TTL,
)
except Exception as e:
logger.warning(f"Parts import: failed to store error lines in Redis: {e}")
return {
"status": "waiting_confirmation",
"job_id": job_id,
"total_rows": processed_rows,
"error_count": error_count,
"valid_rows": processed_rows - error_count,
"errors": errors_detail,
}
def _str_or_none(val: Any, max_len: Optional[int] = None) -> Optional[str]:
if val is None:
return None
s = str(val).strip()
if not s:
return None
if max_len and len(s) > max_len:
return s[:max_len]
return s
def _int_or_none(val: Any) -> Optional[int]:
if val is None or val == "":
return None
try:
return int(val)
except (ValueError, TypeError):
return None
def _decimal_or_none(val: Any) -> Optional[Decimal]:
if val is None or val == "":
return None
try:
return Decimal(str(val))
except (InvalidOperation, ValueError, TypeError):
return None
def _bool_from_row(val: Any) -> bool:
if val is None or val == "":
return True
s = str(val).strip().upper()
if s in ("0", "F", "FALSE", "NO", "N"):
return False
return True
@celery_app.task(bind=True)
def insert_valid_rows(self, job_id: str):
logger.info(f"Parts import: starting commit for job {job_id}")
file_path = _ensure_worker_has_file_from_redis(job_id)
if not file_path:
alt_path = os.path.join(_worker_upload_dir(), f"part_{job_id}.csv")
if not os.path.exists(alt_path):
return {
"status": "failed",
"error": "Archivo no encontrado (expirado). Sube y confirma de nuevo.",
}
file_path = alt_path
else:
_ensure_worker_has_meta_from_redis(job_id, file_path)
error_dir = layout_path("imports", "errors")
error_path = os.path.join(error_dir, f"part_{job_id}.jsonl")
error_lines = set()
try:
r = _get_redis()
raw = r.get(f"{PART_IMPORT_ERROR_LINES_PREFIX}{job_id}")
if raw:
error_lines = set(json.loads(raw.decode("utf-8")))
except Exception as e:
logger.debug(f"Parts import: could not load error lines from Redis: {e}")
if not error_lines and os.path.exists(error_path):
with open(error_path, "r", encoding="utf-8") as f:
for line in f:
try:
err = json.loads(line)
error_lines.add(err["line"])
except Exception:
pass
meta_path = file_path.replace(".csv", ".meta.json")
tenant_id = None
company_id = None
meta = {}
if os.path.exists(meta_path):
try:
with open(meta_path, "r", encoding="utf-8") as f:
meta = json.load(f) or {}
tenant_id = meta.get("tenant_id")
company_id = meta.get("company_id")
except Exception:
pass
if not tenant_id or not company_id:
return {"status": "failed", "error": "Falta contexto (tenant/company)"}
from api.v1.modules.a76.parts.models import Part
valid_class_codes: Set[str] = set()
valid_uom_codes: Set[str] = set()
valid_currency_codes: Set[str] = set()
try:
with CoreSessionLocal() as session:
from api.v1.modules.a76.classes.models import Class
from api.v1.modules.a76.general_catalogs.units_of_measure.models import UnitOfMeasure
from api.v1.modules.public.reference_data.currency_types.models import CurrencyType
for c in (
session.query(Class.class_code)
.filter(
Class.tenant_id == tenant_id,
Class.company_id == company_id,
)
.all()
):
valid_class_codes.add(c[0])
for u in (
session.query(UnitOfMeasure.code)
.filter(
UnitOfMeasure.tenant_id == tenant_id,
UnitOfMeasure.company_id == company_id,
)
.all()
):
valid_uom_codes.add(u[0])
for cur in session.query(CurrencyType.code).all():
valid_currency_codes.add(cur[0])
except Exception as e:
logger.warning(f"Parts import: could not load FK sets: {e}")
inserted_count = 0
skipped_invalid = 0
skipped_missing_fk = 0
skipped_duplicate = 0
skipped_details: List[Dict[str, Any]] = []
response = None
try:
with CoreSessionLocal() as session:
existing_by_part_number: Dict[str, Part] = {}
for p in (
session.query(Part)
.filter(
Part.tenant_id == tenant_id,
Part.company_id == company_id,
)
.all()
):
existing_by_part_number[p.part_number] = p
with open(file_path, "r", encoding="utf-8-sig") as f:
sample = f.read(2048)
f.seek(0)
try:
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t")
except Exception:
dialect = "excel"
reader = csv.DictReader(f, dialect=dialect)
for i, row in enumerate(reader, start=1):
if i in error_lines:
continue
row_norm = row_from_template(row, normalize_header)
err = _validate_row_part(
row_norm,
i,
valid_class_codes=valid_class_codes,
valid_uom_codes=valid_uom_codes,
valid_currency_codes=valid_currency_codes,
)
if err:
skipped_invalid += 1
skipped_details.append(
{"line": i, "reason": f"{err.get('col', '')}: {err.get('msg', '')}"}
)
continue
part_number = _str_or_none(row_norm.get("NUMPARTE"), 70)
if not part_number:
skipped_invalid += 1
continue
commercial = _str_or_none(row_norm.get("NUMPARTECOM"), 70)
desc_es = _str_or_none(row_norm.get("DESCRIPCIONE"), 500)
desc_en = _str_or_none(row_norm.get("DESCRIPCIONI"), 500)
part_class = _str_or_none(row_norm.get("CLASE"), 8)
if part_class and part_class not in valid_class_codes:
part_class = None
unit_of_measure = _str_or_none(row_norm.get("UNIMED"), 5)
if unit_of_measure and unit_of_measure not in valid_uom_codes:
unit_of_measure = None
currency_key = _str_or_none(row_norm.get("MONEDA"), 3)
if currency_key and currency_key not in valid_currency_codes:
currency_key = None
unit_cost = _decimal_or_none(row_norm.get("COSTOUNIT"))
currency_type = _str_or_none(row_norm.get("MONEDA"), 2) if currency_key else None
unit_weight = _decimal_or_none(row_norm.get("PESOUNIT"))
weight_type = _str_or_none(row_norm.get("TIPOPESO"), 6)
fraction = _str_or_none(row_norm.get("FRACCION"), 10)
us_fraction = _str_or_none(row_norm.get("FRACCIONAME"), 16)
fda_key = _str_or_none(row_norm.get("FDAKEY"), 20)
fcc_key = _str_or_none(row_norm.get("FCCKEY"), 30)
license_code = _str_or_none(row_norm.get("LICENCIA"), 3)
eccn = _str_or_none(row_norm.get("ECCN"), 20)
export_code = _str_or_none(row_norm.get("EXPORTCODE"), 2)
exclusion_symbol = _str_or_none(row_norm.get("EXCLUSION"), 19)
is_active = _bool_from_row(row_norm.get("ACTIVO"))
existing = existing_by_part_number.get(part_number)
if existing:
existing.commercial_part_number = commercial
existing.description_spanish = desc_es
existing.description_english = desc_en
existing.part_class = part_class
existing.unit_of_measure = unit_of_measure
existing.unit_cost = unit_cost
existing.currency_type = currency_type
existing.currency_key = currency_key
existing.unit_weight = unit_weight
existing.weight_type = weight_type
existing.fraction = fraction
existing.us_fraction = us_fraction
existing.fda_key = fda_key
existing.fcc_key = fcc_key
existing.license_code = license_code
existing.eccn = eccn
existing.export_code = export_code
existing.exclusion_symbol = exclusion_symbol
existing.is_active = is_active
session.add(existing)
inserted_count += 1
else:
new_part = Part(
tenant_id=tenant_id,
company_id=company_id,
client_id=company_id,
part_number=part_number,
commercial_part_number=commercial,
description_spanish=desc_es,
description_english=desc_en,
part_class=part_class,
unit_of_measure=unit_of_measure,
unit_cost=unit_cost,
currency_type=currency_type,
currency_key=currency_key,
unit_weight=unit_weight,
weight_type=weight_type,
fraction=fraction,
us_fraction=us_fraction,
fda_key=fda_key,
fcc_key=fcc_key,
license_code=license_code,
eccn=eccn,
export_code=export_code,
exclusion_symbol=exclusion_symbol,
is_active=is_active,
)
session.add(new_part)
existing_by_part_number[part_number] = new_part
inserted_count += 1
try:
session.commit()
except Exception as db_err:
session.rollback()
logger.error(f"Parts import DB error: {db_err}")
return {"status": "failed", "error": str(db_err)}
total_skipped = skipped_invalid + skipped_missing_fk + skipped_duplicate
if inserted_count == 0 and total_skipped > 0:
response = {
"status": "warning",
"inserted": 0,
"skipped_invalid": skipped_invalid,
"skipped_missing_fk": skipped_missing_fk,
"skipped_duplicate": skipped_duplicate,
"skipped_details": skipped_details,
"message": f"No se insertaron registros. {total_skipped} rechazados.",
}
elif inserted_count == 0:
response = {
"status": "failed",
"error": "No hay registros válidos en el archivo CSV",
"inserted": 0,
"skipped_invalid": skipped_invalid,
"skipped_missing_fk": skipped_missing_fk,
"skipped_duplicate": skipped_duplicate,
"skipped_details": skipped_details,
}
else:
response = {
"status": "finished",
"inserted": inserted_count,
"skipped_invalid": skipped_invalid,
"skipped_missing_fk": skipped_missing_fk,
"skipped_duplicate": skipped_duplicate,
"skipped_details": skipped_details,
}
except Exception as e:
logger.error(f"Parts import task failed: {e}")
import traceback
logger.error(traceback.format_exc())
return {"status": "failed", "error": str(e)}
try:
if file_path and os.path.exists(file_path):
os.remove(file_path)
if os.path.exists(error_path):
os.remove(error_path)
if os.path.exists(meta_path):
os.remove(meta_path)
_delete_import_from_redis(job_id)
except Exception as cleanup_err:
logger.warning(f"Parts import cleanup failed: {cleanup_err}")
if response is None:
response = {
"status": "failed",
"error": "Error inesperado",
"inserted": 0,
"skipped_invalid": skipped_invalid,
"skipped_missing_fk": skipped_missing_fk,
"skipped_duplicate": skipped_duplicate,
"skipped_details": skipped_details,
}
return response