Files
plantillas-proyectos/backend/tests/unit/test_csv_reader_encoding.py
2026-04-14 09:34:51 -06:00

151 lines
5.7 KiB
Python

from pathlib import Path
from api.v1.modules.a76.layouts_csv.common.csv_reader import (
CsvReadPlan,
detect_text_encoding,
inspect_csv,
iter_csv_rows,
iter_csv_rows_with_plan,
)
from api.v1.modules.a76.layouts_csv.classes.template_config import (
detect_headers_or_data as detect_classes_headers,
row_from_template as row_from_classes_template,
)
from api.v1.modules.a76.layouts_csv.parts.template_config import detect_headers_or_data as detect_parts_headers
from api.v1.modules.a76.layouts_csv.pedmientos.template_config import (
detect_headers_or_data as detect_pedimentos_headers,
parse_pedimento_col_a,
)
def _write_bytes(tmp_path: Path, name: str, payload: bytes) -> Path:
file_path = tmp_path / name
file_path.write_bytes(payload)
return file_path
def test_detect_text_encoding_handles_truncated_utf8_sample(tmp_path: Path):
# Regression: "ó" in "Descripción" empieza en byte 21 (0xC3 0xB3).
# sample_bytes=22 lee bytes 0-21, terminando en 0xC3 (primer byte de ó, secuencia incompleta).
# El código viejo: raw.decode("utf-8-sig") fallaba → caía a cp1252 → mojibake.
# El código nuevo: decoder incremental tolera el corte → retorna utf-8-sig.
payload = "DESCRIPCION\nDescripción español\n".encode("utf-8")
file_path = _write_bytes(tmp_path, "truncated_utf8.csv", payload)
enc = detect_text_encoding(str(file_path), sample_bytes=22)
assert enc in ("utf-8", "utf-8-sig"), (
f"Got {enc!r} — el archivo UTF-8 con corte de muestra a mitad de multibyte "
"fue detectado como cp1252, produciendo mojibake (español / Descripción)"
)
def test_utf8_enie_at_sample_boundary_not_detected_as_cp1252(tmp_path: Path):
# Regresión directa del bug mojibake reportado en producción.
# Construye un payload donde 'ñ' (0xC3 0xB1 en UTF-8) cae exactamente en el byte 19,
# y sample_bytes=20 lee sólo 0xC3 (primer byte) — secuencia incompleta.
# Resultado esperado: utf-8 / utf-8-sig (no cp1252).
header = b"CLASE,DESC\n" # 11 bytes
row = "C01,español\n".encode("utf-8") # ñ en bytes 19-20 del payload total
payload = header + row
file_path = _write_bytes(tmp_path, "regression_mojibake.csv", payload)
enc = detect_text_encoding(str(file_path), sample_bytes=20)
assert enc in ("utf-8", "utf-8-sig"), (
f"Got {enc!r} en lugar de utf-8 — leer como cp1252 produciría "
"'español' en lugar de 'español'"
)
def test_iter_csv_rows_preserves_utf8_values(tmp_path: Path):
payload = "CLASE,DESCRIPCION ESPAÑOL\nCLASE01,Clase prueba español\n".encode("utf-8")
file_path = _write_bytes(tmp_path, "utf8_values.csv", payload)
rows = list(iter_csv_rows(str(file_path)))
assert len(rows) == 1
_, row = rows[0]
assert row["DESCRIPCION ESPAÑOL"] == "Clase prueba español"
def test_iter_csv_rows_keeps_cp1252_compatibility(tmp_path: Path):
payload = "CLASE,DESCRIPCION ESPAÑOL\nCLASE01,Descripción\n".encode("cp1252")
file_path = _write_bytes(tmp_path, "cp1252_values.csv", payload)
rows = list(iter_csv_rows(str(file_path)))
assert len(rows) == 1
_, row = rows[0]
assert row["DESCRIPCION ESPAÑOL"] == "Descripción"
def test_parts_detect_headers_or_data_handles_cp1252(tmp_path: Path):
payload = "NUMERO DE PARTE,DESCRIPCION EN ESPAÑOL\nP-01,Descripción\n".encode("cp1252")
file_path = _write_bytes(tmp_path, "parts_cp1252.csv", payload)
fieldnames, has_header = detect_parts_headers(str(file_path), lambda s: (s or "").strip().upper())
assert has_header is True
assert fieldnames is None
def test_pedimentos_detect_headers_or_data_handles_cp1252_data_first_row(tmp_path: Path):
payload = "24,1234,1234567,I,A1\n".encode("cp1252")
file_path = _write_bytes(tmp_path, "pedimentos_cp1252_data.csv", payload)
fieldnames, has_header = detect_pedimentos_headers(
str(file_path),
lambda s: (s or "").strip().upper(),
parse_pedimento_col_a,
)
assert has_header is False
assert fieldnames is not None
def test_classes_detect_headers_or_data_handles_cp1252(tmp_path: Path):
payload = "CLAVE CLASE;DESCRIPCION ESPAÑOL\nC01;Descripción\n".encode("cp1252")
file_path = _write_bytes(tmp_path, "classes_cp1252_semicolon.csv", payload)
fieldnames, has_header = detect_classes_headers(str(file_path), lambda s: (s or "").strip().upper())
assert has_header is True
assert fieldnames is None
def test_classes_row_from_template_recovers_collapsed_header_with_semicolon():
row = {"CLAVE CLASE,DESCRIPCION ESPAÑOL": "C01;Descripción;Description"}
mapped = row_from_classes_template(row, lambda s: (s or "").strip().upper())
assert mapped["CLASE"] == "C01"
def test_iter_csv_rows_with_plan_headerless_and_semicolon(tmp_path: Path):
payload = "C01;Descripcion 1\nC02;Descripcion 2\n".encode("utf-8")
file_path = _write_bytes(tmp_path, "headerless_semicolon.csv", payload)
plan = CsvReadPlan(
header_mode="headerless",
fieldnames=["CLASE", "DESCRIPCIONE"],
)
rows = list(iter_csv_rows_with_plan(str(file_path), plan))
assert len(rows) == 2
assert rows[0][1]["CLASE"] == "C01"
assert rows[1][1]["DESCRIPCIONE"] == "Descripcion 2"
def test_inspect_csv_auto_mode_switches_to_headerless(tmp_path: Path):
payload = "C01,Descripcion\n".encode("utf-8")
file_path = _write_bytes(tmp_path, "auto_mode.csv", payload)
plan = CsvReadPlan(
header_mode="auto",
fieldnames=["CLASE", "DESCRIPCIONE"],
headerless_first_cell_values={"C01"},
)
metadata = inspect_csv(str(file_path), plan)
assert metadata.has_header is False
assert metadata.fieldnames == ["CLASE", "DESCRIPCIONE"]