Files
stack/tests/aco/test_load_cclf.py
kert 1baf98eb40 implement data ingestion: CCLF loader, BCDA loader, seed loader, unified staging
- CCLF: fixed-width parser using IP-derived field positions (cclf_layout.py),
  ZIP extraction, file discovery by CMS naming convention, DuckDB loading,
  optional pipeline execution to produce input_layer tables
- BCDA: flatten ndjson → Parquet, load into bcda schema in DuckDB
- Seeds: auto-discover CSV/Excel/Parquet files, load into reference_data schema
- Unified staging: orchestrate all three loaders with optional Iceberg promotion
- CLI: stack load cclf/bcda/seed with full options (--path, --database, etc.)
- 57 tests covering parsers, loaders, CLI commands, and staging pipeline

fixes #5 fixes #6 fixes #7 fixes #8
2026-03-12 15:52:08 -04:00

170 lines
5.8 KiB
Python

"""Tests for CCLF file loading."""
from __future__ import annotations
from datetime import date
from pathlib import Path
import pytest
from aco.load.cclf import (
_parse_value,
discover_cclf_files,
load_cclf_directory,
parse_cclf_file,
)
from aco.table.cclf_layout import LAYOUTS
class TestParseValue:
def test_empty_string(self) -> None:
assert _parse_value("", "X(11)") is None
def test_whitespace_only(self) -> None:
assert _parse_value(" ", "X(11)") is None
def test_string_field(self) -> None:
assert _parse_value("1234567890", "X(10)") == "1234567890"
def test_date_field(self) -> None:
assert _parse_value("2025-07-16", "YYYY-MM-DD") == date(2025, 7, 16)
def test_date_invalid(self) -> None:
assert _parse_value("0000-00-00", "YYYY-MM-DD") is None
def test_numeric_field(self) -> None:
assert _parse_value("123.45", "-9(13).99") == 123.45
def test_numeric_invalid(self) -> None:
assert _parse_value("N/A", "-9(13).99") is None
def test_numeric_9_format(self) -> None:
assert _parse_value("42", "9(13)") == "42"
class TestParseCclfFile:
def test_parse_cclf9(self) -> None:
"""CCLF9 is the simplest file: 6 fields, 55 bytes per line."""
# Build a realistic CCLF9 line: positions 1-55
line = (
"N" # hicn_mbi_xref_ind (1-1)
"1AN0Y00AA04" # crnt_num (2-12)
"2AN0Y00AA05" # prvs_num (13-23)
"2024-01-15" # prvs_id_efctv_dt (24-33)
"2025-06-30" # prvs_id_obslt_dt (34-43)
"RRB123456789" # bene_rrb_num (44-55)
)
df = parse_cclf_file([line], "cclf9")
assert len(df) == 1
assert df["hicn_mbi_xref_ind"][0] == "N"
assert df["crnt_num"][0] == "1AN0Y00AA04"
assert df["prvs_num"][0] == "2AN0Y00AA05"
assert df["prvs_id_efctv_dt"][0] == date(2024, 1, 15)
assert df["prvs_id_obslt_dt"][0] == date(2025, 6, 30)
assert df["bene_rrb_num"][0] == "RRB123456789"
def test_parse_empty_lines(self) -> None:
df = parse_cclf_file(["", " ", ""], "cclf9")
assert df.is_empty()
def test_parse_cclf8_date_and_string(self) -> None:
"""Check a CCLF8 line with date, numeric, and string fields."""
# Build CCLF8 line (549 bytes)
line = " " * 549
parts = list(line)
# bene_mbi_id (1-11)
for i, c in enumerate("1AN0Y00AA04"):
parts[i] = c
# bene_dob (33-42)
for i, c in enumerate("1945-03-22"):
parts[32 + i] = c
# bene_sex_cd (43)
parts[42] = "1"
# bene_race_cd (44)
parts[43] = "2"
line = "".join(parts)
df = parse_cclf_file([line], "cclf8")
assert len(df) == 1
assert df["bene_mbi_id"][0] == "1AN0Y00AA04"
assert df["bene_dob"][0] == date(1945, 3, 22)
assert df["bene_sex_cd"][0] == "1"
assert df["bene_race_cd"][0] == "2"
def test_empty_file_returns_schema(self) -> None:
df = parse_cclf_file([], "cclf9")
assert df.is_empty()
assert "crnt_num" in df.columns
class TestDiscoverCclfFiles:
def test_discover_plain_files(self, tmp_path: Path) -> None:
# Create a couple of fake CCLF files
(tmp_path / "P.A1234.ACO.ZC1Y25.D250716.T1234567").write_text("data\n")
(tmp_path / "P.A1234.ACO.ZC8Y25.D250716.T1234567").write_text("data\n")
(tmp_path / "random.txt").write_text("noise\n")
found = discover_cclf_files(tmp_path)
assert "cclf1" in found
assert "cclf8" in found
assert len(found) == 2
def test_discover_skips_hidden(self, tmp_path: Path) -> None:
(tmp_path / ".extracted").mkdir()
(tmp_path / "P.A1234.ACO.ZC1Y25.D250716.T1234567").write_text("data\n")
found = discover_cclf_files(tmp_path)
assert "cclf1" in found
class TestLayouts:
def test_all_pipeline_inputs_have_layouts(self) -> None:
"""Every CCLF table referenced by the pipeline must have a layout."""
needed = {
"cclf1",
"cclf2",
"cclf3",
"cclf4",
"cclf5",
"cclf6",
"cclf7",
"cclf8",
"cclf9",
}
assert needed.issubset(LAYOUTS.keys())
def test_positions_are_sequential(self) -> None:
"""Field positions should be non-overlapping and ordered."""
for table, fields in LAYOUTS.items():
if table == "cclf0":
continue # CCLF0 has two record types sharing positions
for name, start, end, fmt in fields:
if name in ("blank", "delimiter", "filler"):
continue
assert start <= end, f"{table}.{name}: start {start} > end {end}"
assert start >= 1, f"{table}.{name}: start must be >= 1"
class TestLoadCclfDirectory:
def test_no_files_raises(self, tmp_path: Path) -> None:
with pytest.raises(FileNotFoundError):
load_cclf_directory(tmp_path, run_pipeline=False)
def test_load_raw_no_pipeline(self, tmp_path: Path) -> None:
"""Load a single CCLF9 file into a temp DuckDB (no pipeline)."""
import duckdb
# Create a fake CCLF9 file
line = "N1AN0Y00AA042AN0Y00AA052024-01-152025-06-30RRB123456789"
cclf_file = tmp_path / "P.A1234.ACO.ZC9Y25.D250716.T1234567"
cclf_file.write_text(line + "\n")
db_path = str(tmp_path / "test.duckdb")
stats = load_cclf_directory(tmp_path, database=db_path, run_pipeline=False)
assert stats["cclf9"] == 1
# Verify it's in DuckDB
con = duckdb.connect(db_path, read_only=True)
rows = con.execute("SELECT * FROM cclf.cclf9").fetchall()
assert len(rows) == 1
assert rows[0][1] == "1AN0Y00AA04" # crnt_num
con.close()