Files
stack/tests/conftest.py
kert f0cc1b3d29
Some checks failed
CI / lint (push) Failing after 29s
CI / test (push) Has been skipped
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
Deploy / api (push) Has been skipped
Deploy / mc (push) Has been skipped
Deploy / restart (push) Successful in 32s
ci: modular actions + test speedup
CI overhaul:
- Split lint (fast, every push) from test (needs lint first)
- Drop 13-job skinny-install matrix — redundant with full test suite
- Add session-scoped zotero_db fixture: create_db() once per session
  instead of ~200 times, saves ~30min cumulative

Deploy overhaul:
- Split monolithic build-scan-report into 5 independent per-image jobs
- Each job has path filters: only builds when relevant files change
  - notebooks: notebooks/, infra/images/notebooks.Dockerfile
  - zotero: infra/images/zotero.Dockerfile, data/zotero/
  - docs: infra/images/docs.Dockerfile, src/, docs/
  - api: infra/images/api.Dockerfile, src/api/
  - mc: infra/images/mc.Dockerfile, infra/rustfs/
- Python-only change no longer rebuilds all 5 Docker images
- Each image: build → trivy scan → push (parallel, independent)

Net effect: Python commit goes from ~50min (test + 5 image builds +
13 skinny-installs) to ~20min (lint + test only). Dockerfile commit
triggers only the affected image (~5min).
2026-04-18 20:10:05 -04:00

818 lines
26 KiB
Python

"""Shared pytest fixtures for the stack test suite.
Fixtures are organized by domain:
- polars DataFrames for every major input_layer schema shape
- PFS computation DataFrames (RVU, GPCI, labor, supply, equipment)
- rex primitives (Sieve, FieldMap, sample raw text)
- bib.tag helpers
- temporary file paths
"""
from __future__ import annotations
import re
from pathlib import Path
import polars as pl
import pytest
# ── Perf hooks — auto-file Gitea issues on test failure/skip in CI ───────────
# Only active when STACK_FILE_TEST_ISSUES=true (set in CI workflows).
try:
from perf.hooks import pytest_runtest_logreport # noqa: F401
except ImportError:
pass
# ── Zotero DB fixture (session-scoped for speed) ─────────────────────────────
# create_db() takes ~10s. By creating once and copying per-test, we cut
# cumulative DB creation from ~30min to ~10s across 200+ Zotero tests.
import shutil
import tempfile
@pytest.fixture(scope="session")
def _zotero_template_db():
"""Session-scoped: create one Zotero schema DB, reuse everywhere."""
from zot.schema import create_db
with tempfile.TemporaryDirectory() as td:
path = str(Path(td) / "template.sqlite")
con = create_db(path)
con.close()
yield path
@pytest.fixture
def zotero_db(_zotero_template_db, tmp_path):
"""Per-test Zotero DB: fast copy of the session template."""
dest = str(tmp_path / "z.sqlite")
shutil.copy2(_zotero_template_db, dest)
return dest
# ── input_layer fixtures ──────────────────────────────────────────────────────
@pytest.fixture
def medical_claim_df() -> pl.DataFrame:
"""One-row medical claim matching input_layer.medical_claim schema."""
return pl.DataFrame(
{
"claim_id": ["CLM001"],
"claim_line_number": [1],
"claim_type": ["professional"],
"person_id": ["P001"],
"member_id": ["M001"],
"patient_id": ["PT001"],
"payer": ["Medicare"],
"plan": [None],
"encounter_id": [None],
"claim_start_date": [None],
"claim_end_date": [None],
"service_unit": [1],
"hcpcs_code": ["99213"],
"hcpcs_modifier_1": [None],
"hcpcs_modifier_2": [None],
"hcpcs_modifier_3": [None],
"hcpcs_modifier_4": [None],
"rendering_npi": ["1234567890"],
"billing_npi": [None],
"facility_npi": [None],
"paid_date": [None],
"paid_amount": [75.00],
"allowed_amount": [95.00],
"charge_amount": [150.00],
"diagnosis_code_type": ["icd-10-cm"],
"diagnosis_code_1": ["E11.9"],
"diagnosis_code_2": [None],
"diagnosis_poa_1": [None],
"procedure_code_type": [None],
"procedure_code_1": [None],
"procedure_date_1": [None],
"data_source": ["test"],
"file_name": ["test.csv"],
"ingest_datetime": [None],
}
)
@pytest.fixture
def eligibility_df() -> pl.DataFrame:
"""One-row eligibility matching input_layer.eligibility schema."""
return pl.DataFrame(
{
"person_id": ["P001"],
"member_id": ["M001"],
"subscriber_id": [None],
"gender": ["M"],
"race": [None],
"birth_date": [None],
"death_date": [None],
"death_flag": [0],
"enrollment_start_date": [None],
"enrollment_end_date": [None],
"payer": ["Medicare"],
"payer_type": ["government"],
"plan": [None],
"dual_status_code": [None],
"medicare_status_code": [None],
"first_name": [None],
"last_name": [None],
"state": ["TX"],
"zip_code": ["78701"],
"data_source": ["test"],
"file_name": ["test.csv"],
"ingest_datetime": [None],
}
)
@pytest.fixture
def pharmacy_claim_df() -> pl.DataFrame:
"""One-row pharmacy claim matching input_layer.pharmacy_claim schema."""
return pl.DataFrame(
{
"claim_id": ["RX001"],
"claim_line_number": [1],
"person_id": ["P001"],
"member_id": ["M001"],
"patient_id": ["PT001"],
"payer": ["Medicare Part D"],
"plan": [None],
"dispensing_provider_id": [None],
"dispensing_date": [None],
"ndc_code": ["00071015523"],
"quantity": [30.0],
"days_supply": [30],
"refills": [0],
"paid_date": [None],
"paid_amount": [12.50],
"allowed_amount": [15.00],
"charge_amount": [25.00],
"data_source": ["test"],
"file_name": ["test.csv"],
"ingest_datetime": [None],
}
)
@pytest.fixture
def encounter_df() -> pl.DataFrame:
"""One-row encounter matching input_layer.encounter schema."""
return pl.DataFrame(
{
"encounter_id": ["ENC001"],
"person_id": ["P001"],
"patient_id": ["PT001"],
"encounter_type": ["acute inpatient"],
"encounter_start_date": [None],
"encounter_end_date": [None],
"length_of_stay": [3],
"admit_source_code": [None],
"admit_source_description": [None],
"admit_type_code": [None],
"admit_type_description": [None],
"discharge_disposition_code": ["01"],
"discharge_disposition_description": ["home"],
"attending_provider_id": [None],
"attending_provider_name": [None],
"facility_id": [None],
"facility_name": [None],
"primary_diagnosis_code_type": ["icd-10-cm"],
"primary_diagnosis_code": ["E11.9"],
"primary_diagnosis_description": [None],
"drg_code_type": [None],
"drg_code": [None],
"drg_description": [None],
"paid_amount": [5000.00],
"allowed_amount": [6000.00],
"charge_amount": [12000.00],
"data_source": ["test"],
"file_name": ["test.csv"],
"ingest_datetime": [None],
}
)
@pytest.fixture
def condition_df() -> pl.DataFrame:
"""One-row condition matching input_layer.condition schema."""
return pl.DataFrame(
{
"condition_id": ["COND001"],
"person_id": ["P001"],
"patient_id": ["PT001"],
"encounter_id": ["ENC001"],
"claim_id": [None],
"recorded_date": [None],
"onset_date": [None],
"resolved_date": [None],
"status": ["active"],
"condition_type": ["problem"],
"source_code_type": ["icd-10-cm"],
"source_code": ["E119"],
"source_description": ["Type 2 diabetes mellitus without complications"],
"normalized_code_type": ["icd-10-cm"],
"normalized_code": ["E11.9"],
"normalized_description": [
"Type 2 diabetes mellitus without complications"
],
"condition_rank": [1],
"present_on_admit_code": [None],
"present_on_admit_description": [None],
"data_source": ["test"],
"file_name": ["test.csv"],
"ingest_datetime": [None],
"payer": ["Medicare"],
}
)
@pytest.fixture
def procedure_df() -> pl.DataFrame:
"""One-row procedure matching input_layer.procedure schema."""
return pl.DataFrame(
{
"procedure_id": ["PROC001"],
"person_id": ["P001"],
"member_id": [None],
"patient_id": ["PT001"],
"encounter_id": ["ENC001"],
"claim_id": ["CLM001"],
"procedure_date": [None],
"source_code_type": ["hcpcs"],
"source_code": ["99213"],
"source_description": ["Office visit, established patient"],
"normalized_code_type": ["hcpcs"],
"normalized_code": ["99213"],
"normalized_description": ["Office visit, established patient"],
"modifier_1": [None],
"modifier_2": [None],
"modifier_3": [None],
"modifier_4": [None],
"modifier_5": [None],
"practitioner_id": [None],
"data_source": ["test"],
"file_name": ["test.csv"],
"ingest_datetime": [None],
}
)
@pytest.fixture
def patient_df() -> pl.DataFrame:
"""One-row patient matching input_layer.patient schema."""
return pl.DataFrame(
{
"person_id": ["P001"],
"patient_id": ["PT001"],
"name_suffix": [None],
"first_name": ["Jane"],
"middle_name": [None],
"last_name": ["Smith"],
"sex": ["F"],
"race": [None],
"birth_date": [None],
"death_date": [None],
"death_flag": [0],
"social_security_number": [None],
"address": ["123 Main St"],
"city": ["Austin"],
"state": ["TX"],
"zip_code": ["78701"],
"county": [None],
"latitude": [None],
"longitude": [None],
"phone": [None],
"email": [None],
"ethnicity": [None],
"data_source": ["test"],
"file_name": ["test.csv"],
"ingest_datetime": [None],
}
)
@pytest.fixture
def practitioner_df() -> pl.DataFrame:
"""One-row practitioner matching input_layer.practitioner schema."""
return pl.DataFrame(
{
"practitioner_id": ["DR001"],
"npi": ["1234567890"],
"first_name": ["Alice"],
"last_name": ["Johnson"],
"practice_affiliation": [None],
"specialty": ["Internal Medicine"],
"sub_specialty": [None],
"data_source": ["test"],
}
)
@pytest.fixture
def location_df() -> pl.DataFrame:
"""One-row location matching input_layer.location schema."""
return pl.DataFrame(
{
"location_id": ["LOC001"],
"npi": ["9876543210"],
"name": ["General Hospital"],
"facility_type": ["hospital"],
"parent_organization": [None],
"address": ["456 Oak Ave"],
"city": ["Austin"],
"state": ["TX"],
"zip_code": ["78702"],
"latitude": [30.2672],
"longitude": [-97.7431],
"data_source": ["test"],
}
)
@pytest.fixture
def medication_df() -> pl.DataFrame:
"""One-row medication matching input_layer.medication schema."""
return pl.DataFrame(
{
"medication_id": ["MED001"],
"person_id": ["P001"],
"patient_id": ["PT001"],
"encounter_id": [None],
"dispensing_date": [None],
"prescribing_date": [None],
"source_code_type": ["ndc"],
"source_code": ["00071015523"],
"source_description": ["Lipitor 10mg"],
"ndc_code": ["00071015523"],
"ndc_description": ["Atorvastatin 10mg"],
"rxnorm_code": ["617312"],
"rxnorm_description": [None],
"atc_code": [None],
"atc_description": [None],
"route": ["oral"],
"strength": ["10mg"],
"quantity": [30.0],
"quantity_unit": ["tablets"],
"days_supply": [30],
"practitioner_id": [None],
"data_source": ["test"],
"file_name": ["test.csv"],
"ingest_datetime": [None],
}
)
@pytest.fixture
def observation_df() -> pl.DataFrame:
"""One-row observation matching input_layer.observation schema."""
return pl.DataFrame(
{
"observation_id": ["OBS001"],
"person_id": ["P001"],
"patient_id": ["PT001"],
"encounter_id": ["ENC001"],
"panel_id": [None],
"observation_date": [None],
"observation_type": ["vital-sign"],
"source_code_type": ["loinc"],
"source_code": ["8480-6"],
"source_description": ["Systolic blood pressure"],
"normalized_code_type": ["loinc"],
"normalized_code": ["8480-6"],
"normalized_description": ["Systolic blood pressure"],
"result": ["120"],
"source_units": ["mm[Hg]"],
"normalized_units": ["mmHg"],
"source_reference_range_low": ["90"],
"source_reference_range_high": ["140"],
"normalized_reference_range_low": ["90"],
"normalized_reference_range_high": ["140"],
"data_source": ["test"],
}
)
@pytest.fixture
def appointment_df() -> pl.DataFrame:
"""One-row appointment matching input_layer.appointment schema."""
return pl.DataFrame(
{
"appointment_id": ["APT001"],
"person_id": ["P001"],
"patient_id": ["PT001"],
"encounter_id": ["ENC001"],
"source_appointment_type_code": ["office"],
"source_appointment_type_description": ["Office Visit"],
"normalized_appointment_type_code": ["office"],
"normalized_appointment_type_description": ["Office Visit"],
"start_datetime": [None],
"end_datetime": [None],
"duration": [30],
"location_id": ["LOC001"],
"practitioner_id": ["DR001"],
"source_status": ["completed"],
"normalized_status": ["completed"],
"appointment_specialty": ["Internal Medicine"],
"reason": ["Follow-up"],
"source_reason_code_type": [None],
"source_reason_code": [None],
"source_reason_description": [None],
"normalized_reason_code_type": [None],
"normalized_reason_code": [None],
"normalized_reason_description": [None],
"cancellation_reason": [None],
"source_cancellation_reason_code_type": [None],
"source_cancellation_reason_code": [None],
"source_cancellation_reason_description": [None],
"normalized_cancellation_reason_code_type": [None],
"normalized_cancellation_reason_code": [None],
"normalized_cancellation_reason_description": [None],
"data_source": ["test"],
}
)
@pytest.fixture
def immunization_df() -> pl.DataFrame:
"""One-row immunization matching input_layer.immunization schema."""
return pl.DataFrame(
{
"immunization_id": ["IMM001"],
"person_id": ["P001"],
"patient_id": ["PT001"],
"encounter_id": ["ENC001"],
"source_code_type": ["cvx"],
"source_code": ["208"],
"source_description": ["COVID-19 Pfizer"],
"normalized_code_type": ["cvx"],
"normalized_code": ["208"],
"normalized_description": ["COVID-19 Pfizer-BioNTech"],
"status": ["completed"],
"status_reason": [None],
"occurrence_date": [None],
"source_dose": ["0.3 mL"],
"normalized_dose": ["0.3 mL"],
"lot_number": ["EL9269"],
"body_site": ["left arm"],
"route": ["intramuscular"],
"location_id": ["LOC001"],
"practitioner_id": ["DR001"],
"data_source": ["test"],
}
)
@pytest.fixture
def lab_result_df() -> pl.DataFrame:
"""One-row lab result matching input_layer.lab_result schema."""
return pl.DataFrame(
{
"lab_result_id": ["LAB001"],
"person_id": ["P001"],
"patient_id": ["PT001"],
"encounter_id": ["ENC001"],
"accession_number": ["ACC001"],
"source_order_type": ["loinc"],
"source_order_code": ["2345-7"],
"source_order_description": ["Glucose"],
"source_component_type": ["loinc"],
"source_component_code": ["2345-7"],
"source_component_description": ["Glucose [Mass/volume] in Serum"],
"normalized_order_type": ["loinc"],
"normalized_order_code": ["2345-7"],
"normalized_order_description": ["Glucose"],
"normalized_component_type": ["loinc"],
"normalized_component_code": ["2345-7"],
"normalized_component_description": ["Glucose [Mass/volume] in Serum"],
"status": ["final"],
"result": ["105"],
"result_datetime": [None],
"collection_datetime": [None],
"source_units": ["mg/dL"],
"normalized_units": ["mg/dL"],
"source_reference_range_low": ["70"],
"source_reference_range_high": ["100"],
"normalized_reference_range_low": ["70"],
"normalized_reference_range_high": ["100"],
"source_abnormal_flag": ["H"],
"normalized_abnormal_flag": ["high"],
"specimen": ["serum"],
"ordering_practitioner_id": ["DR001"],
"data_source": ["test"],
}
)
@pytest.fixture
def provider_attribution_df() -> pl.DataFrame:
"""One-row provider attribution matching input_layer.provider_attribution."""
return pl.DataFrame(
{
"person_id": ["P001"],
"member_id": ["M001"],
"practitioner_id": ["DR001"],
"practitioner_npi": ["1234567890"],
"attribution_type": ["primary"],
"attribution_start_date": [None],
"attribution_end_date": [None],
"data_source": ["test"],
}
)
@pytest.fixture
def member_months_df() -> pl.DataFrame:
"""Simple member-months DataFrame used by claims_preprocessing pipeline."""
return pl.DataFrame(
{
"person_id": ["P001"],
"member_id": ["M001"],
"year": [2024],
"month": [1],
"payer": ["Medicare"],
"plan": [None],
"data_source": ["test"],
}
)
# ── PFS computation fixtures ──────────────────────────────────────────────────
TEST_CF: float = 32.7442
"""Conversion factor used by PFS payment() tests (historical CY2023)."""
@pytest.fixture
def rvu_df() -> pl.DataFrame:
"""RVU data for PFS payment calculation tests.
Note: ``conv_factor`` is no longer a column on pfs.rvu — it's
passed as the ``cf=`` keyword to ``pfs.calcs.payment.payment()``.
Use ``tests.conftest.TEST_CF`` when calling.
"""
return pl.DataFrame(
{
"hcpcs": ["99213", "99214", "99215"],
"work_rvu": [0.97, 1.50, 2.11],
"non_fac_pe_rvu": [1.04, 1.56, 2.22],
"fac_pe_rvu": [0.41, 0.63, 0.95],
"mp_rvu": [0.07, 0.11, 0.16],
"mac": ["10212", "10212", "10212"],
"locality": ["0201", "0201", "0201"],
}
)
@pytest.fixture
def gpci_df() -> pl.DataFrame:
"""GPCI data for PFS payment calculation tests."""
return pl.DataFrame(
{
"mac": ["10212"],
"locality": ["0201"],
"work_gpci": [1.0],
"pe_gpci": [0.998],
"mp_gpci": [0.633],
}
)
@pytest.fixture
def labor_df() -> pl.DataFrame:
"""Clinical labor inputs for direct PE cost tests."""
return pl.DataFrame(
{
"hcpcs": ["99213", "99213", "99214"],
"nf_minutes": [14.0, 5.0, 18.0],
"f_minutes": [9.0, 3.0, 12.0],
"rate_per_minute": [0.63, 0.49, 0.63],
}
)
@pytest.fixture
def supply_df() -> pl.DataFrame:
"""Medical supply inputs for direct PE cost tests."""
return pl.DataFrame(
{
"hcpcs": ["99213", "99214"],
"nf_quantity": [2.0, 3.0],
"f_quantity": [1.0, 2.0],
"unit_price": [0.12, 0.12],
}
)
@pytest.fixture
def equipment_df() -> pl.DataFrame:
"""Medical equipment inputs for direct PE cost tests."""
return pl.DataFrame(
{
"hcpcs": ["99213", "99214"],
"nf_minutes": [10.0, 15.0],
"f_minutes": [6.0, 10.0],
"unit_price": [1200.0, 1200.0],
"useful_life": [7.0, 7.0],
"minutes_per_year": [525600.0, 525600.0],
}
)
@pytest.fixture
def equipment_with_maintenance_df() -> pl.DataFrame:
"""Equipment DataFrame that includes a maintenance_factor column."""
return pl.DataFrame(
{
"hcpcs": ["99213"],
"nf_minutes": [10.0],
"f_minutes": [6.0],
"unit_price": [1200.0],
"useful_life": [7.0],
"minutes_per_year": [525600.0],
"maintenance_factor": [0.05],
}
)
@pytest.fixture
def work_time_df() -> pl.DataFrame:
"""Physician work time for time-intensity ratio tests."""
return pl.DataFrame(
{
"hcpcs": ["99213", "99214", "99215"],
"work_rvu": [0.97, 1.50, 2.11],
"total_time": [21.0, 31.0, 46.0],
"intra_service_time": [15.0, 22.0, 33.0],
}
)
@pytest.fixture
def mppr_claims_df() -> pl.DataFrame:
"""Two procedures in the same session for MPPR reduction tests."""
return pl.DataFrame(
{
"claim_id": ["CLM001", "CLM001"],
"hcpcs": ["99214", "99213"],
"session_id": ["SES001", "SES001"],
"non_fac_pe_rvu": [1.56, 1.04],
}
)
# ── rex fixtures ──────────────────────────────────────────────────────────────
@pytest.fixture
def sas_listing_text() -> str:
"""Multiline SAS listing file content for sieve classification tests."""
return (
" The SAS System 10:30 AM\n"
"\n"
" Obs claim_id paid_amount diag_code\n"
" 1 CLM001 75.00 E119\n"
" 2 CLM002 150.00 I10\n"
" 3 CLM003 225.00 J449\n"
" Total: 450.00\n"
"\n"
"NOTE: The data set WORK.CLAIMS has 3 observations.\n"
)
@pytest.fixture
def fixed_width_text() -> str:
"""Fixed-width records for FieldMap positional extraction tests."""
return (
"CLM001 P001 20240115000007500\n"
"CLM002 P002 20240116000015000\n"
"CLM003 P003 20240117000022500\n"
)
@pytest.fixture
def pipe_delimited_text() -> str:
"""Pipe-delimited file for FieldMap delimiter-mode tests."""
return (
"claim_id|person_id|service_date|paid_amount\n"
"CLM001|P001|20240115|75.00\n"
"CLM002|P002|20240116|150.00\n"
)
@pytest.fixture
def copybook_text() -> str:
"""Simple COBOL copybook definition for copybook parser tests."""
return """
01 CLAIM-RECORD.
05 CLAIM-ID PIC X(10).
05 PERSON-ID PIC X(10).
05 SERVICE-DATE PIC 9(8).
05 PAID-AMOUNT PIC S9(7)V99 COMP-3.
05 FILLER PIC X(5).
05 DX-CODE PIC X(7).
"""
@pytest.fixture
def basic_sieve():
"""Minimal Sieve with only a data pattern for simple unit tests."""
from rex.sieve import Sieve
return Sieve(
name="test_basic",
data=re.compile(r"^\s+\d+\s+\w"),
)
@pytest.fixture
def sas_sieve():
"""Full four-pattern SAS listing Sieve matching the module docstring."""
from rex.sieve import Sieve
return Sieve(
name="sas_listing",
head=re.compile(r"^\s*(The SAS System|\x0c)|^\s+\d{2}:\d{2}"),
rule=re.compile(r"^\s*(Obs\s+\w|---+[\s-]*---+)"),
data=re.compile(r"^\s+\d+\s+\w"),
skip=re.compile(r"^\s*(Total|NOTE:|WARNING:|$)"),
)
@pytest.fixture
def fixed_width_field_map():
"""FieldMap in fixed-width positional mode for claim record extraction."""
from rex.sieve import FieldMap
return FieldMap(
positions={
"claim_id": (0, 10),
"person_id": (10, 20),
"service_date": (20, 28),
"paid_amount": (28, 37),
}
)
@pytest.fixture
def delimited_field_map():
"""FieldMap in pipe-delimited mode with index mapping."""
from rex.sieve import FieldMap
return FieldMap(
delimiter="|",
indices={
"claim_id": 0,
"person_id": 1,
"service_date": 2,
"paid_amount": 3,
},
)
# ── bib.tag fixtures ──────────────────────────────────────────────────────────
@pytest.fixture
def tag_labels() -> list[str]:
"""Mixed tag label strings for filter_tags tests."""
return [
"module:pfs",
"module:aco",
"table:pfs.rvu",
"source:cms-website",
"year:2026",
"rule:cms-1807-f",
"file:rvu",
]
# ── temp file fixtures ────────────────────────────────────────────────────────
@pytest.fixture
def tmp_text_file(tmp_path: Path) -> Path:
"""Write the SAS listing sample to a temporary .lst file."""
content = (
" The SAS System 10:30 AM\n"
"\n"
" Obs claim_id paid_amount\n"
" 1 CLM001 75.00\n"
" 2 CLM002 150.00\n"
)
f = tmp_path / "report.lst"
f.write_text(content, encoding="utf-8")
return f
@pytest.fixture
def tmp_csv_file(tmp_path: Path) -> Path:
"""Write a simple CSV to a temporary file."""
content = "claim_id,person_id,paid_amount\nCLM001,P001,75.00\nCLM002,P002,150.00\n"
f = tmp_path / "claims.csv"
f.write_text(content, encoding="utf-8")
return f