add initial reference manager logic

This commit is contained in:
kert
2026-02-08 17:41:28 -05:00
parent cf01d90cc1
commit 91b78e9a60
543 changed files with 24661 additions and 666 deletions

View File

@@ -0,0 +1,78 @@
{
"permissions": {
"allow": [
"Bash(docker exec:*)",
"Bash(openssl rand:*)",
"Bash(docker compose up:*)",
"Bash(docker logs:*)",
"Bash(docker push:*)",
"Bash(git commit:*)",
"Bash(curl:*)",
"Bash(git add:*)",
"Bash(git push:*)",
"Bash(docker run:*)",
"Bash(ip -4 addr:*)",
"WebFetch(domain:blog.pnp3.com)",
"WebFetch(domain:woodpecker-ci.org)",
"WebSearch",
"Bash(python3:*)",
"Bash(docker compose restart:*)",
"Bash(docker inspect:*)",
"Bash(docker compose build:*)",
"Bash(docker compose:*)",
"WebFetch(domain:docs.marimo.io)",
"WebFetch(domain:github.com)",
"WebFetch(domain:raw.githubusercontent.com)",
"Bash(docker ps:*)",
"Bash(docker volume rm:*)",
"Bash(ulimit:*)",
"Bash(while read host)",
"Bash(do)",
"Bash(echo:*)",
"Bash(done)",
"WebFetch(domain:projectnessie.org)",
"Bash(docker build:*)",
"WebFetch(domain:polaris.apache.org)",
"WebFetch(domain:medium.com)",
"Bash(docker pull:*)",
"Bash(lsusb:*)",
"Bash(lspci:*)",
"Bash(lsblk:*)",
"Bash(sudo dmidecode:*)",
"Bash(nvidia-smi:*)",
"Bash(cat:*)",
"Bash(modinfo:*)",
"Bash(modprobe:*)",
"Bash(ip link:*)",
"Bash(docker cp:*)",
"Bash(ss -tlnp:*)",
"Bash(sudo iptables:*)",
"Bash(iptables -L:*)",
"Bash(iptables:*)",
"Bash(wc:*)",
"mcp__acp__Bash",
"WebFetch(domain:narwhals-dev.github.io)",
"WebFetch(domain:sqlmodel.tiangolo.com)",
"WebFetch(domain:realpython.com)",
"WebFetch(domain:posit-dev.github.io)",
"mcp__acp__Edit",
"mcp__acp__Write",
"WebFetch(domain:www.cms.gov)",
"WebFetch(domain:www.ama-assn.org)",
"WebFetch(domain:www.ncbi.nlm.nih.gov)",
"WebFetch(domain:www.zotero.org)",
"WebFetch(domain:pyzotero.readthedocs.io)",
"WebFetch(domain:api.zotero.org)",
"WebFetch(domain:www.msi.com)",
"WebFetch(domain:www.crucial.com)",
"WebFetch(domain:www.hidevolution.com)",
"WebFetch(domain:)",
"WebFetch(domain:databricks-sdk-py.readthedocs.io)",
"WebFetch(domain:deepwiki.com)",
"WebFetch(domain:marimo.io)",
"WebFetch(domain:unpkg.com)",
"WebFetch(domain:forums.zotero.org)",
"WebFetch(domain:www.gmass.co)"
]
}
}

View File

@@ -158,10 +158,13 @@ services:
container_name: notebooks container_name: notebooks
networks: networks:
- intrastack - intrastack
environment:
- PYTHONPATH=/home/kert/src
volumes: volumes:
- ./notebooks:/home/kert/notebooks - ./notebooks:/home/kert/notebooks
- ./data:/home/kert/data - ./data:/home/kert/data
- ./zotero/data:/home/kert/zotero:ro - ./zotero/data:/home/kert/zotero:ro
- ./src:/home/kert/src:ro
deploy: deploy:
resources: resources:
reservations: reservations:
@@ -185,6 +188,8 @@ services:
- "3478:3478/udp" - "3478:3478/udp"
volumes: volumes:
- ./zotero/data:/home/ubuntu/Zotero - ./zotero/data:/home/ubuntu/Zotero
- ./zotero/profiles/zotero:/home/ubuntu/.zotero
- ./zotero/profiles/mozilla:/home/ubuntu/.mozilla
- ./data:/home/ubuntu/data - ./data:/home/ubuntu/data
tmpfs: tmpfs:
- /dev/shm:rw - /dev/shm:rw
@@ -305,13 +310,17 @@ services:
- COLLECTOR_OTLP_ENABLED=true - COLLECTOR_OTLP_ENABLED=true
- COLLECTOR_ZIPKIN_HOST_PORT=:9411 - COLLECTOR_ZIPKIN_HOST_PORT=:9411
ports: ports:
- "14268:14268" # Jaeger collector HTTP - "14268:14268" # Jaeger collector HTTP
- "14250:14250" # Jaeger collector gRPC - "14250:14250" # Jaeger collector gRPC
- "6831:6831/udp" # Jaeger agent UDP - "6831:6831/udp" # Jaeger agent UDP
- "4317:4317" # OTLP gRPC - "4317:4317" # OTLP gRPC
- "4318:4318" # OTLP HTTP - "4318:4318" # OTLP HTTP
healthcheck: healthcheck:
test: ["CMD-SHELL", "wget --no-verbose --tries=1 --spider http://localhost:16686 || exit 1"] test:
[
"CMD-SHELL",
"wget --no-verbose --tries=1 --spider http://localhost:16686 || exit 1",
]
interval: 15s interval: 15s
timeout: 5s timeout: 5s
retries: 5 retries: 5
@@ -330,7 +339,11 @@ services:
- loki_data:/loki - loki_data:/loki
command: -config.file=/etc/loki/local-config.yaml command: -config.file=/etc/loki/local-config.yaml
healthcheck: healthcheck:
test: ["CMD-SHELL", "wget --no-verbose --tries=1 --spider http://localhost:3100/ready || exit 1"] test:
[
"CMD-SHELL",
"wget --no-verbose --tries=1 --spider http://localhost:3100/ready || exit 1",
]
interval: 30s interval: 30s
timeout: 10s timeout: 10s
retries: 5 retries: 5
@@ -361,11 +374,15 @@ services:
- ./prometheus/prometheus.yml:/etc/prometheus/prometheus.yml:ro - ./prometheus/prometheus.yml:/etc/prometheus/prometheus.yml:ro
- prometheus_data:/prometheus - prometheus_data:/prometheus
command: command:
- '--config.file=/etc/prometheus/prometheus.yml' - "--config.file=/etc/prometheus/prometheus.yml"
- '--storage.tsdb.path=/prometheus' - "--storage.tsdb.path=/prometheus"
- '--web.enable-lifecycle' - "--web.enable-lifecycle"
healthcheck: healthcheck:
test: ["CMD-SHELL", "wget --no-verbose --tries=1 --spider http://localhost:9090/-/healthy || exit 1"] test:
[
"CMD-SHELL",
"wget --no-verbose --tries=1 --spider http://localhost:9090/-/healthy || exit 1",
]
interval: 15s interval: 15s
timeout: 5s timeout: 5s
retries: 5 retries: 5
@@ -395,7 +412,11 @@ services:
- jaeger - jaeger
- prometheus - prometheus
healthcheck: healthcheck:
test: ["CMD-SHELL", "wget --no-verbose --tries=1 --spider http://localhost:3000/api/health || exit 1"] test:
[
"CMD-SHELL",
"wget --no-verbose --tries=1 --spider http://localhost:3000/api/health || exit 1",
]
interval: 15s interval: 15s
timeout: 5s timeout: 5s
retries: 5 retries: 5

6087
dag.html Normal file

File diff suppressed because it is too large Load Diff

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

BIN
dev/ALRASRGuide.pdf Normal file

Binary file not shown.

Binary file not shown.

Binary file not shown.

BIN
dev/PY2023 Templates.zip Normal file

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

View File

@@ -0,0 +1,154 @@
CLMH;00400;R2345;HUIP;4JE3U44HP11;35446-18.1A;;20230101;20230104;455000;1234567890;2;1;1;0;;25.22;;;;;;;;216;;4019;;;;S23456;20230101;S298643;20230101;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;428;;599;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;2348963;T2345;S2962;CO;
CLML;1;0120;;;;;;;20230101;;4;100.00;;100.00;;;;;;;;;;;50;25;;;;2348963;;T2345;S6829;CO;
CLML;2;0001;;;;;;;00000000;;2;100.00;;100.00;;;;;;;;;;;;;;;;2348963;;T2355;H6829;CO;
CLMH;00400;R2345;HUIP;4JE3U44HP11;35446-18.1B;;20230110;20230113;455000;1234567890;2;1;1;0;;25.22;;;;;;;;216;;4019;;;;S23456;20230101;S298643;20230101;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;428;;599;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;2348963;T2345;S2962;CO;
CLML;1;0120;;;;;;;20230110;;4;100.00;;100.00;;;;;;;;;;;24;25;;;;2348963;;T2345;S6829;CO;
CLML;2;0001;;;;;;;00000000;;4;100.00;;100.00;;;;;;;;;;;;;;;;2348963;;T2355;H6829;CO;
CLMH;00400;R2345;HUIP;4JE3U44HP11;35446-18.1B;;20230110;20230113;455000;1234567890;2;1;1;1;;25.22;;;;;;;;216;;4019;;;;S23456;20230101;S298643;20230101;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;428;;599;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;2348963;T2345;S2962;CO;
CLML;1;0120;;;;;;;20230110;;4;100.00;;100.00;;;;;;;;;;;24;25;;;;2348963;;T2345;S6829;CO;
CLML;2;0001;;;;;;;00000000;;4;100.00;;100.00;;;;;;;;;;;;;;;;2348963;;T2355;H6829;CO;
CLMH;00400;R2345;HUIP;4JE3U44HP11;35446-18.1C;;20230120;20230123;455000;1234567890;2;1;1;0;;25.22;;;;;;;;216;;4019;;;;S23456;20230101;S298643;20230101;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;428;;599;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;2348963;T2345;S2962;CO;
CLML;1;0120;;;;;;;20230120;;4;100.00;;100.00;;;;;;;;;;;24;25;;;;2348963;;T2345;S6829;CO;
CLML;2;0001;;;;;;;00000000;;4;100.00;;100.00;;;;;;;;;;;;;;;;2348963;;T2355;H6829;CO;
CLMH;00400;R2345;HUIP;4JE3U44HP11;35446-18.1C;;20230120;20230123;455000;1234567890;2;1;1;1;P;25.22;;;;;;;;216;;4019;;;;S23456;20230101;S298643;20230101;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;428;;599;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;2348963;T2345;S2962;CO;
CLML;1;0120;;;;;;;20230120;;4;100.00;;100.00;;;;;;;;;;;24;25;;;;2348963;;T2345;S6829;CO;
CLML;2;0001;;;;;;;00000000;;4;100.00;;100.00;;;;;;;;;;;;;;;;2348963;;T2355;H6829;CO;
CLMH;00400;R2345;HUIP;4JE3U44HP11;35446-18.1D;;20230201;20230204;455000;1234567890;2;1;1;0;;25.22;;;;;;;;216;;4019;;;;S23456;20230101;S298643;20230101;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;428;;599;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;2348963;T2345;S2962;CO;
CLML;1;0120;;;;;;;20230201;;4;100.00;;100.00;;;;;;;;;;;50;79;;;;2348963;;T2345;S6829;CO;
CLML;2;0001;;;;;;;00000000;;4;100.00;;100.00;;;;;;;;;;;;;;;;2348963;;T2355;H6829;CO;
CLMH;00400;R2345;HUIP;4JE3U44HP11;35446-18.1D;;20230201;20230204;455000;1234567890;2;1;1;1;;25.22;;;;;;;;216;;4019;;;;S23456;20230101;S298643;20230101;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;428;;599;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;2348963;T2345;S2962;CO;
CLML;1;0120;;;;;;;20230201;;4;100.00;;100.00;;;;;;;;;;;50;79;;;;2348963;;T2345;S6829;CO;
CLML;2;0001;;;;;;;00000000;;4;100.00;;100.00;;;;;;;;;;;;;;;;2348963;;T2355;H6829;CO;
CLMH;00400;R2345;HUIP;4JE3U44HP11;35446-18.1D ADJ;35446-18.1D;20230201;20230205;455000;1234567890;2;1;1;2;O;25.22;;;;;;;;216;;4019;;;;S23456;20230101;S298643;20230101;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;428;;599;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;2348963;T2345;S2962;CO;
CLML;1;0120;;;;;;;20230201;;5;100.00;;100.00;;;;;;;;;;;50;79;;;;2348963;;T2345;S6829;CO;
CLML;2;0001;;;;;;;00000000;;5;100.00;;100.00;;;;;;;;;;;;;;;;2348963;;T2355;H6829;CO;
CLMH;04401;R2345;HUOP;4WC3U44HE63;35446-7.1A;;20230101;20230101;450001;;1;3;1;0;;.00;;;;;;;;;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0300;;;;;;;20230101;;1;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;0;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLMH;31145;R2345;HUBC;4WC3U44HE63;35446-7.1B;;20230105;20231008;;;;;;0;;.00;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;3;;;;;
CLML;1;;035689421;;25000;;;;20230105;20230105;1;100.00;1.00;.00;.00;1.00;;;;;;;;99215;;;;;;;11;;;;
CLML;2;;035689421;;25000;;;;20230105;20230105;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;;;;;;;11;;;;
CLML;3;;035689421;;25000;;;;20231008;20231008;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;;;;;;;11;;;;
CLMH;00400;R2345;HUIP;4WC3U44HE63;35446-7.2;;20230201;20230205;455000;1234567890;2;1;1;0;;.00;;;;;;;;216;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0120;;;;;;;20230201;;4;100.00;;100.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;5;100.00;;100.00;;;;;;;;;;;;;;;;;;;;;
CLMH;04401;R2345;HUOP;4WC3U44HE63;35446-7.3A;;20230306;20230306;450001;;1;3;1;0;;.00;;;;;;;;;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0300;;;;;;;20230306;;1;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;0;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLMH;31145;R2345;HUBC;4WC3U44HE63;35446-7.3B;;20230307;20230307;;;;;;0;;.00;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;;035689421;;25000;;;;20230307;20230307;1;100.00;1.00;.00;.00;1.00;;;;;;;;99215;;;;;;;11;;;;
CLML;2;;035689421;;25000;;;;20230307;20230307;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;;;;;;;11;;;;
CLMH;00400;R2345;HUIP;4WC3U44HE63;35446-7.4;;20230401;20230405;455000;1234567890;2;1;1;0;;.00;;;;;;;;216;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0120;;;;;;;20230401;;4;100.00;;100.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;5;100.00;;100.00;;;;;;;;;;;;;;;;;;;;;
CLMH;04401;R2345;HUOP;4WC3U44HE63;35446-7.4B;;20230506;20230506;450001;;1;3;1;0;;.00;;;;;;;;;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0300;;;;;;;20230506;;1;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;0;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLMH;31145;R2345;HUBC;4WC3U44HE63;35446-7.4C;;20230507;20230507;;;;;;0;;.00;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;;035689421;;25000;;;;20230507;20230507;1;100.00;1.00;.00;.00;1.00;;;;;;;;99215;;;;;;;11;;;;
CLML;2;;035689421;;25000;;;;20230507;20230507;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;;;;;;;11;;;;
CLMH;31145;R2345;HUBC;4WD3U44HF64;35446-19.1;;20230107;20230107;;;;;;0;;.00;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;;035689421;;25000;;;;20230107;20230107;1;100.00;1.00;.00;.00;1.00;;;;;;;;99215;;;;;;;11;;;;
CLML;2;;035689421;;25000;;;;20230107;20230107;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;;;;;;;11;;;;
CLMH;04401;R2345;HUOP;4WD3U44HF64;35446-20.1;;20230106;20230106;451300;;8;5;1;0;;.00;;;;;;;;;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0960;;;;;;;20230106;;1;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;0;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLMH;11302;R2345;HUIP;4WD3U44HF64;35446-21.1;;20230101;20230105;455000;1234567890;2;1;1;0;;.00;;;;;;;;216;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0211;;;;;;;20230101;;4;100.00;;100.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;5;100.00;;100.00;;;;;;;;;;;;;;;;;;;;;
CLMH;31145;R2345;HUBC;4WD3U44HF64;35446-22.1;;20230207;20230207;;;;;;0;;.00;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;;035689421;;25000;;;;20230207;20230207;1;100.00;1.00;.00;.00;1.00;;;;;;;;99215;;;;;;;11;;;;
CLML;2;;035689421;;25000;;;;20230207;20230207;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;;;;;;;11;;;;
CLMH;04401;R2345;HUOP;4WD3U44HF64;35446-23.1;;20230206;20230206;451300;;8;5;1;0;;.00;;;;;;;;;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0960;;;;;;;20230206;;1;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;0;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLMH;00400;R2345;HUIP;4WD3U44HF64;35446-24.1;;20231001;20231005;455000;1234567890;2;1;1;0;;.00;;;;;;;;216;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0211;;;;;;;20231001;;4;100.00;;100.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;5;100.00;;100.00;;;;;;;;;;;;;;;;;;;;;
CLMH;04401;R2345;HUOP;4WD3U44HG65;35446-18.1E;;20230101;20230101;450001;634982423;1;3;1;0;;13.55;2.00;100.00;;;;;;;;4019;3660;;;42369;20230101;S0324;20230102;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;428;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;56568981;T234;T299;CO;
CLML;1;0300;;12346679;;;;;20230101;;1;100.00;;100.00;;;;;;;;;;;50;24;;;;56568981;;T234;T921;CO;
CLML;2;0001;;122345678;;;;;00000000;;0;100.00;;100.00;;;;;;;;;;;;;;;;56568981;;T234;T6314;CO;
CLMH;04401;R2345;HUOP;4WD3U44HG65;35446-18.1F;;20230105;20230105;450001;634982423;1;3;1;0;;13.55;2.00;15.00;;;;;;;;4019;3660;;;42369;20230101;S0324;20230102;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;428;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;23456789;T234;T299;CO;
CLML;1;0300;;12346679;;;;;20230105;;1;15.00;;15.00;;;;;;;;;;;50;24;;;;23456789;;T234;T921;CO;
CLML;2;0001;;122345678;;;;;00000000;;0;15.00;;15.00;;;;;;;;;;;;;;;;23456789;;T234;T6314;CO;
CLMH;04401;R2345;HUOP;4WD3U44HG65;35446-18.1F;;20230105;20230105;450001;634982423;1;3;1;1;;13.55;2.00;15.00;;;;;;;;4019;3660;;;42369;20230101;S0324;20230102;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;428;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;23456789;T234;T299;CO;
CLML;1;0300;;12346679;;;;;20230105;;1;15.00;;15.00;;;;;;;;;;;50;24;;;;23456789;;T234;T921;CO;
CLML;2;0001;;122345678;;;;;00000000;;0;15.00;;15.00;;;;;;;;;;;;;;;;23456789;;T234;T6314;CO;
CLMH;04401;R2345;HUOP;4WD3U44HG65;35446-18.1G;;20230110;20230110;451300;634982423;8;5;1;0;;13.55;2.00;30.00;;;;;;;;4019;3660;;;42369;20230101;S0324;20230102;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;428;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;3;R2345899;T234;T299;CO;
CLML;1;0960;;12346679;;;;;20230110;;1;15.00;;15.00;;;;;;;;;;;50;24;;;;R2345899;;T234;T921;CO;
CLML;2;0960;;12346679;;;;;20230110;;1;15.00;;15.00;;;;;;;;;;;24;25;;;;R2345899;;T234;T6314;CO;
CLML;3;0001;;12346679;;;;;20230110;;1;30.00;;30.00;;;;;;;;;;;24;25;;;;R2345899;;T234;T555;PR;
CLMH;04401;R2345;HUOP;4WD3U44HG65;35446-18.1G;;20230110;20230110;451300;634982423;8;5;1;1;;13.55;2.00;30.00;;;;;;;;4019;3660;;;42369;20230101;S0324;20230102;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;428;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;3;R2345899;T234;T299;CO;
CLML;1;0960;;12346679;;;;;20230110;;1;15.00;;15.00;;;;;;;;;;;50;24;;;;R2345899;;T234;T921;CO;
CLML;2;0960;;12346679;;;;;20230110;;1;15.00;;15.00;;;;;;;;;;;24;25;;;;R2345899;;T234;T6314;CO;
CLML;3;0001;;12346679;;;;;20230110;;1;30.00;;30.00;;;;;;;;;;;24;25;;;;R2345899;;T234;T555;PR;
CLMH;04401;R2345;HUOP;4WD3U44HG65;35446-18.1G ADJ;35446-18.1G;20230110;20230110;451300;634982423;8;5;1;2;O;13.55;2.00;30.00;;;;;;;;4019;3660;;;42369;20230101;S0324;20230102;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;428;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;3;R2345899;T234;T299;CO;
CLML;1;0960;;12346679;;;;;20230110;;1;15.00;;15.00;;;;;;;;;;;50;24;;;;R2345899;;T234;T921;CO;
CLML;2;0960;;12346679;;;;;20230110;;1;15.00;;15.00;;;;;;;;;;;24;25;;;;R2345899;;T234;T6314;CO;
CLML;3;0001;;12346679;;;;;20230110;;1;30.00;;30.00;;;;;;;;;;;24;25;;;;R2345899;;T234;T555;PR;
CLMH;04401;R2345;HUOP;4WD3U44HG65;35446-18.1H;;20230120;20230120;451300;634982423;8;5;1;0;;13.55;2.00;30.00;;;;;;;;4019;3660;;;42369;20230101;S0324;20230102;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;428;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;3;R2345899;T234;T299;CO;
CLML;1;0960;;12346679;;;;;20230120;;1;15.00;;15.00;;;;;;;;;;;50;24;;;;R2345899;;T234;T921;CO;
CLML;2;0960;;12346679;;;;;20230120;;1;15.00;;15.00;;;;;;;;;;;24;25;;;;R2345899;;T234;T6314;CO;
CLML;3;0001;;12346679;;;;;20230110;;1;30.00;;30.00;;;;;;;;;;;24;25;;;;R2345899;;T234;T555;PR;
CLMH;04401;R2345;HUOP;4WD3U44HG65;35446-18.1H;;20230120;20230120;451300;634982423;8;5;1;1;B;13.55;2.00;30.00;;;;;;;;4019;3660;;;42369;20230101;S0324;20230102;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;428;;411;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;3;R2345899;T234;T299;CO;
CLML;1;0960;;12346679;;;;;20230120;;1;15.00;;15.00;;;;;;;;;;;50;24;;;;R2345899;;T234;T921;CO;
CLML;2;0960;;12346679;;;;;20230120;;1;15.00;;15.00;;;;;;;;;;;24;25;;;;R2345899;;T234;T6314;CO;
CLML;3;0001;;12346679;;;;;20230110;;1;30.00;;30.00;;;;;;;;;;;24;25;;;;R2345899;;T234;T555;PR;
CLMH;31145;R2345;HUBC;4WD3U44HG65;35446-18.1I;;20230201;20230201;;;;;;0;;14.80;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;421;0;411;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;292634;T234;R9113;CO;
CLML;1;;035689421;164681;25000;;;;20230201;20230201;1;100.00;10.00;7.80;2.00;.00;;;;;;;;99215;24;25;;;;292634;02;;;;
CLML;2;;035689421;;25000;;;;20230201;20230201;1;100.00;10.00;7.00;2.00;.00;;;;;;;;99499;50;;;;;292634;02;;;;
CLMH;31145;R2345;HUBC;4WD3U44HG65;35446-18.1J;;20230201;20230202;;;;;;0;;14.80;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;421;0;411;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;292634;T234;R9113;CO;
CLML;1;;035689421;164681;25000;;;;20230202;20230202;1;100.00;10.00;7.80;2.00;.00;;;;;;;;99215;24;25;;;;292634;02;;;;
CLML;2;;035689421;;25000;;;;20230201;20230201;1;100.00;10.00;7.00;2.00;.00;;;;;;;;99499;50;;;;;292634;02;;;;
CLMH;31145;R2345;HUBC;4WD3U44HG65;35446-18.1J;;20230201;20230202;;;;;;1;;14.80;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;421;0;411;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;292634;T234;R9113;CO;
CLML;1;;035689421;164681;25000;;;;20230202;20230202;1;100.00;10.00;7.80;2.00;.00;;;;;;;;99215;24;25;;;;292634;02;;;;
CLML;2;;035689421;;25000;;;;20230201;20230201;1;100.00;10.00;7.00;2.00;.00;;;;;;;;99499;50;;;;;292634;02;;;;
CLMH;31145;R2345;HUBC;4WD3U44HG65;35446-18.1K;;20230205;20230205;;;;;;0;;14.80;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;421;0;411;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;292634;T234;R9113;CO;
CLML;1;;035689421;164681;25000;;;;20230205;20230205;1;100.00;10.00;7.80;2.00;.00;;;;;;;;99215;24;25;;;;292634;02;;;;
CLML;2;;035689421;;25000;;;;20230205;20230205;1;100.00;10.00;7.00;2.00;.00;;;;;;;;99499;50;;;;;292634;02;;;;
CLMH;31145;R2345;HUBC;4WD3U44HG65;35446-18.1K;;20230205;20230205;;;;;;1;;14.80;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;421;0;411;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;292634;T234;R9113;CO;
CLML;1;;035689421;164681;25000;;;;20230205;20230205;1;100.00;10.00;7.80;2.00;.00;;;;;;;;99215;24;25;;;;292634;02;;;;
CLML;2;;035689421;;25000;;;;20230205;20230205;1;100.00;10.00;7.00;2.00;.00;;;;;;;;99499;50;;;;;292634;02;;;;
CLMH;31145;R2345;HUBC;4WD3U44HG65;35446-18.1K;35446-18.1K;20230205;20230205;;;;;;2;O;14.80;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;421;0;411;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;292634;T234;R9113;CO;
CLML;1;;035689421;164681;25000;;;;20230205;20230205;1;100.00;10.00;7.80;2.00;.00;;;;;;;;99211;;;;;;292634;02;;;;
CLML;2;;035689421;;25000;;;;20230205;20230205;1;100.00;10.00;7.00;2.00;.00;;;;;;;;99499;50;;;;;292634;02;;;;
CLMH;31145;R2345;HUBC;4WD3U44HG65;35446-18.1L;;20230205;20230206;;;;;;0;;7.80;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;421;0;411;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;292634;T234;R9113;CO;
CLML;1;;035689421;164681;25000;;;;20230205;20230205;1;100.00;10.00;7.80;2.00;.00;;;;;;;;99215;24;25;;;;292634;02;;;;
CLML;2;;035689421;;25000;;;;20230206;20230206;1;100.00;10.00;.00;.00;10.00;;;;;;;;99499;50;;;;;292634;02;;;;
CLMH;31145;R2345;HUBC;4WD3U44HG65;35446-18.1L;35446-18.1L;20230205;20230206;;;;;;1;P;7.80;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;421;0;411;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;292634;T234;R9113;CO;
CLML;1;;035689421;164681;25000;;;;20230205;20230205;1;100.00;10.00;7.80;2.00;.00;;;;;;;;99215;24;25;;;;292634;02;;;;
CLML;2;;035689421;;25000;;;;20230206;20230206;1;100.00;10.00;.00;.00;10.00;;;;;;;;99499;50;;;;;292634;02;;;;
CLMH;00400;R2345;HUIP;9AA3G44HP11;35446-4.8A (BE 4);;20231001;20231005;455000;1234567890;2;1;1;0;;.00;;;;;;;;216;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0120;;;;;;;20231001;;4;100.00;;100.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;5;100.00;;100.00;;;;;;;;;;;;;;;;;;;;;
CLMH;00882;R2345;HUOP;9AA3G44HP11;35446-4.8B (BE 7);;20231007;20231007;450001;;1;3;1;0;;.00;;;;;;;;;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0300;;;;;;;20231007;;1;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;0;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLMH;00882;R2345;HUBC;9AA3G44HP11;35446-4.8C BE 2;;20231008;20231008;;;;;;0;;.00;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;5;;;;;
CLML;1;;035689421;;25000;;;;20231008;20231008;1;100.00;1.00;.00;.00;1.00;;;;;;;;99215;;;;;;;02;;;;
CLML;2;;035689421;;25000;;;;20231008;20231008;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;;;;;;;02;;;;
CLML;3;;035689421;;25000;;;;20231008;20231008;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;;;;;;;02;;;;
CLML;4;;035689421;;25000;;;;20231008;20231008;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;;;;;;;02;;;;
CLML;5;;035689421;;25000;;;;20231008;20231008;1;100.00;1.00;.00;.00;1.00;;;;;;;;99212;;;;;;;02;;;;
CLMH;00882;R2345;HUBC;9AA3G44HP11;35446-1B NEG;;20231011;20231011;;;;;;0;;.00;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;5;;;;;
CLML;1;;035689421;;25000;;;;20231011;20231011;1;100.00;1.00;.00;.00;1.00;;;;;;;;99215;;;;;;;02;;;;
CLML;2;;035689421;;25000;;;;20231011;20231011;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;;;;;;;02;;;;
CLML;3;;035689421;;25000;;;;20231011;20231011;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;;;;;;;02;;;;
CLML;4;;035689421;;25000;;;;20231011;20231011;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;;;;;;;02;;;;
CLML;5;;035689421;;25000;;;;20231011;20231011;1;100.00;1.00;.00;.00;1.00;;;;;;;;99212;;;;;;;02;;;;
CLMH;00882;R2345;HUOP;9AA3G44HP11;35446-1C NEG;;20231012;20231012;450001;;1;3;1;0;;.00;;;;;;;;;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0300;;;;;;;20231012;;1;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;0;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLMH;00400;R2345;HUIP;9AA3G54HR33;35446-7.4A (MUL BES);;20230101;20230105;455000;1234567890;2;1;1;0;;.00;;;;;;;;216;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4660;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0211;;;;;;;20230101;;4;100.00;;100.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;5;100.00;;100.00;;;;;;;;;;;;;;;;;;;;;
CLMH;04401;R2345;HUOP;9AA3G54HR33;35446-7.4B (MULTI BES);;20230506;20230506;451300;;8;5;1;0;;.00;;;;;;;;;;;;;;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;;00000000;4019;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;2;;;;;
CLML;1;0960;;;;;;;20230506;;1;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLML;2;0001;;;;;;;00000000;;0;147.00;;147.00;;;;;;;;;;;;;;;;;;;;;
CLMH;00882;R2345;HUBC;9AA3G54HR33;35446-7.4C MULT;;20230707;20230707;;;;;;0;;.00;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;25000;0;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;5;;;;;
CLML;1;;035689421;;25000;;;;20230707;20230707;1;100.00;1.00;.00;.00;1.00;;;;;;;;99215;2;7;4;;;;11;;;;
CLML;2;;035689421;;25000;;;;20230707;20230707;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;2;7;4;;;;11;;;;
CLML;3;;035689421;;25000;;;;20230707;20230707;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;2;4;7;;;;11;;;;
CLML;4;;035689421;;25000;;;;20230707;20230707;1;100.00;1.00;.00;.00;1.00;;;;;;;;99499;2;4;7;;;;11;;;;
CLML;5;;035689421;;25000;;;;20230707;20230707;1;100.00;1.00;.00;.00;1.00;;;;;;;;99212;2;4;7;;;;11;;;;


View File

@@ -0,0 +1,3 @@
ACO_ID,MBI_ID,ALGN_TYPE_SLM,ALGN_TYPE_VA,PRVDR_TIN (CLM OR VA),PRVDR_NPI (CLM OR VA),FAC_PRVDR_OSCAR_NUM,QEM_ALLOWED_PRIMARY_AY1,QEM_ALLOWED_NONPRIMARY_AY1,QEM_ALLOWED_OTHER_AY1,QEM_ALLOWED_PRIMARY_AY2,QEM_ALLOWED_NONPRIMARY_AY2,QEM_ALLOWED_OTHER_AY2
D9999,1XXXXXXXXXXX,N,SVA,123456789,1234567890,,.,.,.,.,.,.
D9999,1XXXXXXXXXXX,Y,,123456789,1234567890,,180.02,0,0,147.62,0,0
1 ACO_ID MBI_ID ALGN_TYPE_SLM ALGN_TYPE_VA PRVDR_TIN (CLM OR VA) PRVDR_NPI (CLM OR VA) FAC_PRVDR_OSCAR_NUM QEM_ALLOWED_PRIMARY_AY1 QEM_ALLOWED_NONPRIMARY_AY1 QEM_ALLOWED_OTHER_AY1 QEM_ALLOWED_PRIMARY_AY2 QEM_ALLOWED_NONPRIMARY_AY2 QEM_ALLOWED_OTHER_AY2
2 D9999 1XXXXXXXXXXX N SVA 123456789 1234567890 . . . . . .
3 D9999 1XXXXXXXXXXX Y 123456789 1234567890 180.02 0 0 147.62 0 0

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

View File

View File

@@ -0,0 +1,239 @@
"""Generate ALR/ASR file naming pattern classifiers from ALRASRGuide.pdf.
Extracts report naming conventions from Appendix Tables 6-7 and generates
``src/aco/table/alr_filenames.py`` — a module containing regex patterns
and a classifier function that rex can use to identify which ALR/ASR
table a given filename represents.
Usage::
uv run python dev/generate_alr_filenames.py
Source: dev/ALRASRGuide.pdf pages 47-51
"""
from __future__ import annotations
from pathlib import Path
def generate_module() -> str:
"""Generate alr_filenames.py module with filename patterns."""
lines = [
'"""ALR/ASR filename patterns for report classification.',
"",
"Auto-generated from ALRASRGuide.pdf Appendix Tables 6-7.",
"Version #16 (April 2024).",
"",
"Naming conventions::",
"",
" Annual ALR Table 1-1: AALR_Table_1_1_ACOID_YYYY.csv",
" Quarterly ALR Table 1-1: QALR_Table_1_1_ACOID_YYYY_QX.csv",
" Annual ASR: AASR_ACOID_YYYY.xlsx",
" Quarterly ASR: QASR_ACOID_YYYY_QX.xlsx",
"",
"Report packages are delivered as ZIP files::",
"",
" Initial Assignment: HASSGN_ACOID_YYYY_MM_DD.zip",
" Quarterly Reports: QEXPU_ACOID_YYYY_MM_DD.zip",
" Financial Reconciliation: STLMT_ACOID_YYYY_MM_DD.zip",
" Historical Benchmark: BNMRK_ACOID_YYYY_MM_DD.zip",
"",
"Source: ALRASRGuide.pdf pages 47-51",
'"""',
"",
"from __future__ import annotations",
"",
"import re",
"from pathlib import Path",
"",
]
# Define ALR table patterns
lines.extend(
[
"",
"# ALR Table Patterns (Annual)",
"ALR_TABLE_1_1_ANNUAL = re.compile(r'AALR_Table_1_1_[A-Z0-9]+_\\d{4}\\.csv$')",
"ALR_TABLE_1_2_ANNUAL = re.compile(r'AALR_Table_1_2_[A-Z0-9]+_\\d{4}\\.csv$')",
"ALR_TABLE_1_3_ANNUAL = re.compile(r'AALR_Table_1_3_[A-Z0-9]+_\\d{4}\\.csv$')",
"ALR_TABLE_1_4_ANNUAL = re.compile(r'AALR_Table_1_4_[A-Z0-9]+_\\d{4}\\.csv$')",
"ALR_TABLE_1_5_ANNUAL = re.compile(r'AALR_Table_1_5_[A-Z0-9]+_\\d{4}\\.csv$')",
"ALR_TABLE_1_6_ANNUAL = re.compile(r'AALR_Table_1_6_[A-Z0-9]+_\\d{4}\\.csv$')",
"ALR_TABLE_1_7_ANNUAL = re.compile(r'AALR_Table_1_7_[A-Z0-9]+_\\d{4}\\.csv$')",
"ALR_TABLE_1_8_ANNUAL = re.compile(r'AALR_Table_1_8_[A-Z0-9]+_\\d{4}\\.csv$')",
"ALR_TABLE_1_9_ANNUAL = re.compile(r'AALR_Table_1_9_[A-Z0-9]+_\\d{4}\\.csv$')",
"",
"# ALR Table Patterns (Quarterly)",
"ALR_TABLE_1_1_QUARTERLY = re.compile(r'QALR_Table_1_1_[A-Z0-9]+_\\d{4}_Q[1-4]\\.csv$')",
"ALR_TABLE_1_2_QUARTERLY = re.compile(r'QALR_Table_1_2_[A-Z0-9]+_\\d{4}_Q[1-4]\\.csv$')",
"ALR_TABLE_1_3_QUARTERLY = re.compile(r'QALR_Table_1_3_[A-Z0-9]+_\\d{4}_Q[1-4]\\.csv$')",
"ALR_TABLE_1_4_QUARTERLY = re.compile(r'QALR_Table_1_4_[A-Z0-9]+_\\d{4}_Q[1-4]\\.csv$')",
"ALR_TABLE_1_5_QUARTERLY = re.compile(r'QALR_Table_1_5_[A-Z0-9]+_\\d{4}_Q[1-4]\\.csv$')",
"ALR_TABLE_1_6_QUARTERLY = re.compile(r'QALR_Table_1_6_[A-Z0-9]+_\\d{4}_Q[1-4]\\.csv$')",
"ALR_TABLE_1_7_QUARTERLY = re.compile(r'QALR_Table_1_7_[A-Z0-9]+_\\d{4}_Q[1-4]\\.csv$')",
"ALR_TABLE_1_8_QUARTERLY = re.compile(r'QALR_Table_1_8_[A-Z0-9]+_\\d{4}_Q[1-4]\\.csv$')",
"ALR_TABLE_1_9_QUARTERLY = re.compile(r'QALR_Table_1_9_[A-Z0-9]+_\\d{4}_Q[1-4]\\.csv$')",
"",
"# ASR Patterns",
"ASR_ANNUAL = re.compile(r'AASR_[A-Z0-9]+_\\d{4}\\.xlsx$')",
"ASR_QUARTERLY = re.compile(r'QASR_[A-Z0-9]+_\\d{4}_Q[1-4]\\.xlsx$')",
"",
"# Report Package Patterns (ZIP files)",
"PACKAGE_HASSGN = re.compile(r'HASSGN_[A-Z0-9]+_\\d{4}_\\d{2}_\\d{2}\\.zip$')",
"PACKAGE_QEXPU = re.compile(r'QEXPU_[A-Z0-9]+_\\d{4}_\\d{2}_\\d{2}\\.zip$')",
"PACKAGE_STLMT = re.compile(r'STLMT_[A-Z0-9]+_\\d{4}_\\d{2}_\\d{2}\\.zip$')",
"PACKAGE_BNMRK = re.compile(r'BNMRK_[A-Z0-9]+_\\d{4}_\\d{2}_\\d{2}\\.zip$')",
"",
]
)
# Generate classifier function
lines.extend(
[
"",
"# Table ID mapping",
"ALR_PATTERNS = {",
' "1-1": [ALR_TABLE_1_1_ANNUAL, ALR_TABLE_1_1_QUARTERLY],',
' "1-2": [ALR_TABLE_1_2_ANNUAL, ALR_TABLE_1_2_QUARTERLY],',
' "1-3": [ALR_TABLE_1_3_ANNUAL, ALR_TABLE_1_3_QUARTERLY],',
' "1-4": [ALR_TABLE_1_4_ANNUAL, ALR_TABLE_1_4_QUARTERLY],',
' "1-5": [ALR_TABLE_1_5_ANNUAL, ALR_TABLE_1_5_QUARTERLY],',
' "1-6": [ALR_TABLE_1_6_ANNUAL, ALR_TABLE_1_6_QUARTERLY],',
' "1-7": [ALR_TABLE_1_7_ANNUAL, ALR_TABLE_1_7_QUARTERLY],',
' "1-8": [ALR_TABLE_1_8_ANNUAL, ALR_TABLE_1_8_QUARTERLY],',
' "1-9": [ALR_TABLE_1_9_ANNUAL, ALR_TABLE_1_9_QUARTERLY],',
"}",
"",
"",
"def classify(filename: str | Path) -> dict | None:",
' """Classify an ALR/ASR filename and return metadata.',
"",
" Args:",
" filename: Filename or path to classify",
"",
" Returns:",
" dict with keys: type (alr/asr/package), table_id, report_type",
" (annual/quarterly), year, quarter (if quarterly), aco_id",
" Returns None if filename doesn't match any pattern.",
"",
" Examples::",
"",
' >>> classify("AALR_Table_1_1_A12345_2024.csv")',
" {",
' "type": "alr",',
' "table_id": "1-1",',
' "report_type": "annual",',
' "year": "2024",',
' "aco_id": "A12345",',
" }",
"",
' >>> classify("QALR_Table_1_2_B67890_2024_Q3.csv")',
" {",
' "type": "alr",',
' "table_id": "1-2",',
' "report_type": "quarterly",',
' "year": "2024",',
' "quarter": "Q3",',
' "aco_id": "B67890",',
" }",
' """',
" fname = Path(filename).name",
"",
" # Check ALR tables",
" for table_id, patterns in ALR_PATTERNS.items():",
" for pattern in patterns:",
" if m := pattern.match(fname):",
" # Parse filename components",
" parts = fname.replace('.csv', '').split('_')",
" report_type = 'annual' if fname.startswith('AALR') else 'quarterly'",
" ",
" result = {",
' "type": "alr",',
' "table_id": table_id,',
' "report_type": report_type,',
" }",
" ",
" # Extract ACO ID and year",
" if report_type == 'annual':",
" # AALR_Table_1_1_ACOID_YYYY.csv",
" result['aco_id'] = parts[3]",
" result['year'] = parts[4]",
" else:",
" # QALR_Table_1_1_ACOID_YYYY_QX.csv",
" result['aco_id'] = parts[3]",
" result['year'] = parts[4]",
" result['quarter'] = parts[5]",
" ",
" return result",
"",
" # Check ASR",
" if m := ASR_ANNUAL.match(fname):",
" parts = fname.replace('.xlsx', '').split('_')",
" return {",
' "type": "asr",',
' "report_type": "annual",',
" 'aco_id': parts[1],",
" 'year': parts[2],",
" }",
" ",
" if m := ASR_QUARTERLY.match(fname):",
" parts = fname.replace('.xlsx', '').split('_')",
" return {",
' "type": "asr",',
' "report_type": "quarterly",',
" 'aco_id': parts[1],",
" 'year': parts[2],",
" 'quarter': parts[3],",
" }",
"",
" # Check report packages",
" package_types = {",
" PACKAGE_HASSGN: 'initial_assignment',",
" PACKAGE_QEXPU: 'quarterly',",
" PACKAGE_STLMT: 'financial_reconciliation',",
" PACKAGE_BNMRK: 'historical_benchmark',",
" }",
" ",
" for pattern, pkg_type in package_types.items():",
" if m := pattern.match(fname):",
" parts = fname.replace('.zip', '').split('_')",
" return {",
' "type": "package",',
" 'package_type': pkg_type,",
" 'aco_id': parts[1],",
" 'year': parts[2],",
" 'month': parts[3],",
" 'day': parts[4],",
" }",
"",
" return None",
"",
]
)
return "\n".join(lines)
def main():
"""Generate src/aco/table/alr_filenames.py."""
output_path = (
Path(__file__).parent.parent / "src" / "aco" / "table" / "alr_filenames.py"
)
code = generate_module()
output_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(code)
print(f"Generated {output_path}")
print(f" 18 ALR table patterns (9 annual + 9 quarterly)")
print(f" 2 ASR patterns (annual + quarterly)")
print(f" 4 package patterns")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,339 @@
"""Generate ALR reference code tables from ALRASRGuide.pdf appendix.
Extracts Appendix Tables 3, 4, and 5 (pages 41-46) containing:
- Primary care service codes (CPT/HCPCS)
- Physician specialty codes
- Non-physician practitioner codes
Generates SQLTable models at ``src/aco/table/alr_codes.py``.
Usage::
uv run python dev/generate_alr_reference_tables.py
Source: dev/ALRASRGuide.pdf
"""
from __future__ import annotations
from pathlib import Path
def extract_reference_codes() -> dict:
"""Extract reference code tables from ALRASRGuide.pdf appendix.
Returns dict with keys: primary_care_codes, physician_specialty_codes,
non_physician_codes.
"""
# Appendix Table 3: Primary Care Service Codes (pages 41-44)
# Manually extracted from PDF - comprehensive list
primary_care_codes = [
# CPT codes
("96160", "CPT", "Health Risk Assessment", None),
("96161", "CPT", "Health Risk Assessment", None),
("96202", "CPT", "Caregiver Services", None),
("96203", "CPT", "Caregiver Services", None),
("97550", "CPT", "Caregiver Services", None),
("97551", "CPT", "Caregiver Services", None),
("97552", "CPT", "Caregiver Services", None),
("99201", "CPT", "New Patient Office Visit", None),
("99202", "CPT", "New Patient Office Visit", None),
("99203", "CPT", "New Patient Office Visit", None),
("99204", "CPT", "New Patient Office Visit", None),
("99205", "CPT", "New Patient Office Visit", None),
("99211", "CPT", "Established Patient Office Visit", None),
("99212", "CPT", "Established Patient Office Visit", None),
("99213", "CPT", "Established Patient Office Visit", None),
("99214", "CPT", "Established Patient Office Visit", None),
("99215", "CPT", "Established Patient Office Visit", None),
("99304", "CPT", "Nursing Facility Initial Care", "Exclude if in SNF"),
("99305", "CPT", "Nursing Facility Initial Care", "Exclude if in SNF"),
("99306", "CPT", "Nursing Facility Initial Care", "Exclude if in SNF"),
("99307", "CPT", "Nursing Facility Subsequent Care", "Exclude if in SNF"),
("99308", "CPT", "Nursing Facility Subsequent Care", "Exclude if in SNF"),
("99309", "CPT", "Nursing Facility Subsequent Care", "Exclude if in SNF"),
("99310", "CPT", "Nursing Facility Subsequent Care", "Exclude if in SNF"),
("99315", "CPT", "Nursing Facility Discharge", "Exclude if in SNF"),
("99316", "CPT", "Nursing Facility Discharge", "Exclude if in SNF"),
("99318", "CPT", "Nursing Facility Annual Assessment", "Exclude if in SNF"),
("99324", "CPT", "Domiciliary/Rest Home New Patient", None),
("99325", "CPT", "Domiciliary/Rest Home New Patient", None),
("99326", "CPT", "Domiciliary/Rest Home New Patient", None),
("99327", "CPT", "Domiciliary/Rest Home New Patient", None),
("99328", "CPT", "Domiciliary/Rest Home New Patient", None),
("99334", "CPT", "Domiciliary/Rest Home Established Patient", None),
("99335", "CPT", "Domiciliary/Rest Home Established Patient", None),
("99336", "CPT", "Domiciliary/Rest Home Established Patient", None),
("99337", "CPT", "Domiciliary/Rest Home Established Patient", None),
("99339", "CPT", "Individual Physician Supervision", None),
("99340", "CPT", "Individual Physician Supervision", None),
("99341", "CPT", "Home Visit New Patient", None),
("99342", "CPT", "Home Visit New Patient", None),
("99343", "CPT", "Home Visit New Patient", None),
("99344", "CPT", "Home Visit New Patient", None),
("99345", "CPT", "Home Visit New Patient", None),
("99347", "CPT", "Home Visit Established Patient", None),
("99348", "CPT", "Home Visit Established Patient", None),
("99349", "CPT", "Home Visit Established Patient", None),
("99350", "CPT", "Home Visit Established Patient", None),
("99354", "CPT", "Prolonged Services", None),
("99355", "CPT", "Prolonged Services", None),
("99406", "CPT", "Smoking Cessation 3-10 min", None),
("99407", "CPT", "Smoking Cessation >10 min", None),
("99421", "CPT", "Online Digital E&M 5-10 min", None),
("99422", "CPT", "Online Digital E&M 11-20 min", None),
("99423", "CPT", "Online Digital E&M 21+ min", None),
("99424", "CPT", "Principal Care Management first 30 min", None),
("99425", "CPT", "Principal Care Management each additional 30 min", None),
("99426", "CPT", "Principal Care Management initial", None),
("99427", "CPT", "Principal Care Management subsequent", None),
("99437", "CPT", "Chronic Care Management", None),
("99439", "CPT", "Chronic Care Management", None),
("99441", "CPT", "Telephone E&M 5-10 min (COVID PHE)", None),
("99442", "CPT", "Telephone E&M 11-20 min (COVID PHE)", None),
("99443", "CPT", "Telephone E&M 21-30 min (COVID PHE)", None),
("99483", "CPT", "Cognitive Impairment Assessment", None),
("99484", "CPT", "Behavioral Health Integration first 20 min", None),
("99487", "CPT", "Complex Chronic Care Management first 60 min", None),
(
"99489",
"CPT",
"Complex Chronic Care Management each additional 30 min",
None,
),
("99490", "CPT", "Chronic Care Management first 20 min", None),
("99491", "CPT", "Chronic Care Management 30 min", None),
("99492", "CPT", "Behavioral Health Integration initial", None),
("99493", "CPT", "Behavioral Health Integration subsequent", None),
("99494", "CPT", "Behavioral Health Integration subsequent", None),
("99495", "CPT", "Transitional Care 14 days", None),
("99496", "CPT", "Transitional Care 7 days", None),
("99497", "CPT", "Advance Care Planning first 30 min", "Exclude if inpatient"),
(
"99498",
"CPT",
"Advance Care Planning each additional 30 min",
"Exclude if inpatient",
),
# HCPCS codes
("G0019", "HCPCS", "Community Health Integration", None),
("G0022", "HCPCS", "Community Health Integration", None),
("G0023", "HCPCS", "Principal Illness Navigation first 60 min", None),
("G0024", "HCPCS", "Principal Illness Navigation each additional 30 min", None),
("G0101", "HCPCS", "Cervical/Vaginal Cancer Screening", None),
("G0136", "HCPCS", "Social Determinants of Health Assessment", None),
("G0317", "HCPCS", "Prolonged E&M first hour", None),
("G0318", "HCPCS", "Prolonged E&M each additional 30 min", None),
("G0402", "HCPCS", "Welcome to Medicare Visit", None),
("G0438", "HCPCS", "Annual Wellness Visit initial", None),
("G0439", "HCPCS", "Annual Wellness Visit subsequent", None),
("G0442", "HCPCS", "Alcohol Misuse Screening 15 min", None),
("G0443", "HCPCS", "Alcohol Misuse Counseling", None),
("G0444", "HCPCS", "Depression Screening Annual", None),
(
"G0463",
"HCPCS",
"Hospital Outpatient Clinic Visit (ETA hospitals only)",
None,
),
("G0506", "HCPCS", "Chronic Care Management", None),
("G2010", "HCPCS", "Remote Evaluation Video/Images", None),
("G2012", "HCPCS", "Virtual Check-In 5-10 min", None),
("G2058", "HCPCS", "Care Management Services for BHI", None),
("G2064", "HCPCS", "Care Management Services 20 min", None),
("G2065", "HCPCS", "Care Management Services 40 min", None),
("G2086", "HCPCS", "Opioid Use Disorder Treatment 70-100 min", None),
("G2087", "HCPCS", "Opioid Use Disorder Treatment 100-130 min", None),
("G2088", "HCPCS", "Opioid Use Disorder Treatment 130+ min", None),
("G2211", "HCPCS", "Complex E&M Add-on", None),
("G2212", "HCPCS", "Prolonged Office E&M", None),
("G2214", "HCPCS", "Chronic Pain Management", None),
("G2252", "HCPCS", "Communication Technology Services Brief", None),
("G3002", "HCPCS", "Chronic Pain Management Initial", None),
("G3003", "HCPCS", "Chronic Pain Management Subsequent", None),
]
# Appendix Table 4: Physician Specialty Codes (page 45)
physician_specialty_codes = [
# Step 1: Primary Care Physicians
("01", 1, "General Practice"),
("08", 1, "Family Practice"),
("11", 1, "Internal Medicine"),
("37", 1, "Pediatric Medicine"),
("38", 1, "Geriatric Medicine"),
# Step 2: Specialist Physicians
("06", 2, "Cardiology"),
("12", 2, "Osteopathic Manipulative Medicine"),
("13", 2, "Neurology"),
("16", 2, "Obstetrics/Gynecology"),
("23", 2, "Sports Medicine"),
("25", 2, "Physical Medicine and Rehabilitation"),
("26", 2, "Psychiatry"),
("27", 2, "Geriatric Psychiatry"),
("29", 2, "Pulmonary Disease"),
("39", 2, "Nephrology"),
("46", 2, "Endocrinology"),
("70", 2, "Multispecialty Clinic/Group"),
("79", 2, "Addiction Medicine"),
("82", 2, "Hematology"),
("83", 2, "Hematology/Oncology"),
("84", 2, "Preventive Medicine"),
("86", 2, "Neuropsychiatry"),
("90", 2, "Medical Oncology"),
("98", 2, "Gynecology/Oncology"),
]
# Appendix Table 5: Non-Physician Practitioner Codes (page 46)
non_physician_codes = [
("50", "Nurse Practitioner"),
("89", "Clinical Nurse Specialist"),
("97", "Physician Assistant"),
]
return {
"primary_care_codes": primary_care_codes,
"physician_specialty_codes": physician_specialty_codes,
"non_physician_codes": non_physician_codes,
}
def generate_module(data: dict) -> str:
"""Generate the alr_codes.py module source code."""
lines = [
'"""ALR Reference Code Tables — Primary care service codes and specialty codes.',
"",
"Auto-generated from ALRASRGuide.pdf Appendix Tables 3, 4, and 5.",
"Version #16 (April 2024).",
"",
"These reference tables define:",
"- Primary care service codes (CPT/HCPCS) used in beneficiary assignment",
"- Physician specialty codes for Step 1 (primary care) and Step 2 (specialist) assignment",
"- Non-physician practitioner codes eligible for Step 1 assignment",
"",
"Source: ALRASRGuide.pdf pages 41-46",
'"""',
"",
"from __future__ import annotations",
"",
"from aco.table.base import SQLTable",
"",
]
# Primary Care Service Codes
lines.extend(
[
"",
"class AlrPrimaryCareServiceCode(SQLTable):",
' """Primary care service codes for MSSP beneficiary assignment.',
"",
" Appendix Table 3: Codes used to identify primary care services",
" in the assignment algorithm. Includes CPT and HCPCS codes with",
" exclusion rules for specific settings (SNF, inpatient).",
"",
f" {len(data['primary_care_codes'])} codes total.",
"",
" Source: ALRASRGuide.pdf pages 41-44",
' """',
"",
' __schema__ = "alr_codes"',
' __tablename__ = "primary_care_service_code"',
"",
" code: str",
' """CPT or HCPCS code (e.g., 99213, G0438)"""',
"",
" code_type: str",
' """Code type: CPT or HCPCS"""',
"",
" description: str",
' """Service description"""',
"",
" exclusion_rule: str | None",
' """Exclusion rule if applicable (e.g., "Exclude if in SNF")"""',
"",
]
)
# Physician Specialty Codes
lines.extend(
[
"",
"class AlrPhysicianSpecialtyCode(SQLTable):",
' """Physician specialty codes for assignment algorithm.',
"",
" Appendix Table 4: Specialty codes used in Step 1 (primary care",
" physicians) and Step 2 (specialist physicians) assignment logic.",
"",
f" {len(data['physician_specialty_codes'])} codes total.",
"",
" Source: ALRASRGuide.pdf page 45",
' """',
"",
' __schema__ = "alr_codes"',
' __tablename__ = "physician_specialty_code"',
"",
" specialty_code: str",
' """Medicare specialty code (e.g., 01, 08, 11)"""',
"",
" assignment_step: int",
' """Assignment step: 1=primary care physicians, 2=specialist physicians"""',
"",
" description: str",
' """Specialty description"""',
"",
]
)
# Non-Physician Practitioner Codes
lines.extend(
[
"",
"class AlrNonPhysicianSpecialtyCode(SQLTable):",
' """Non-physician practitioner codes for Step 1 assignment.',
"",
" Appendix Table 5: Non-physician practitioner specialty codes",
" eligible for Step 1 (primary care) assignment.",
"",
f" {len(data['non_physician_codes'])} codes total.",
"",
" Source: ALRASRGuide.pdf page 46",
' """',
"",
' __schema__ = "alr_codes"',
' __tablename__ = "non_physician_specialty_code"',
"",
" specialty_code: str",
' """Medicare specialty code (e.g., 50, 89, 97)"""',
"",
" description: str",
' """Specialty description"""',
"",
]
)
return "\n".join(lines)
def main():
"""Generate src/aco/table/alr_codes.py from ALRASRGuide.pdf appendix."""
output_path = (
Path(__file__).parent.parent / "src" / "aco" / "table" / "alr_codes.py"
)
data = extract_reference_codes()
code = generate_module(data)
output_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(code)
print(f"Generated {output_path}")
print(f" Primary care service codes: {len(data['primary_care_codes'])}")
print(f" Physician specialty codes: {len(data['physician_specialty_codes'])}")
print(f" Non-physician practitioner codes: {len(data['non_physician_codes'])}")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,583 @@
"""Generate ALR SQLTable models from the ALR/ASR User's Guide PDF.
Parses the Assignment List Report (ALR) table specifications from
the ALRASRGuide.pdf to extract all 9 ALR table layouts and generates
a single Python module at ``src/aco/table/alr.py`` with one SQLTable
subclass per table.
Usage::
uv run python dev/generate_alr_table_models.py
Source: dev/ALRASRGuide.pdf
"""
from __future__ import annotations
import re
from pathlib import Path
import pdfplumber
def extract_alr_table_specs(pdf_path: str) -> dict:
"""Extract ALR table field specifications from PDF pages 8-26.
Returns dict mapping table_id -> {title, fields: list[dict]}
"""
pdf = pdfplumber.open(pdf_path)
# Manual field specifications based on PDF extraction
# Table 1-1 is on pages 8-15, others follow
tables = {}
# Table 1-1: Assigned Beneficiaries (PRIMARY TABLE)
tables["1-1"] = {
"title": "Assigned Beneficiaries",
"fields": [
{"name": "MBI", "type": "str", "desc": "Medicare Beneficiary Identifier"},
{
"name": "HICN",
"type": "str | None",
"desc": "Health Insurance Claim Number (blank after 1/1/2020)",
},
{"name": "First_Name", "type": "str", "desc": "Beneficiary first name"},
{"name": "Last_Name", "type": "str", "desc": "Beneficiary last name"},
{"name": "Sex", "type": "int", "desc": "0=unknown, 1=male, 2=female"},
{"name": "Birth_Date", "type": "date", "desc": "Date of birth"},
{
"name": "Death_Date",
"type": "date | None",
"desc": "Date of death if applicable",
},
{
"name": "County_Name",
"type": "str | None",
"desc": "County of residence (blank if excluded)",
},
{
"name": "State_Name",
"type": "str | None",
"desc": "State of residence (blank if excluded)",
},
{
"name": "State_County_CD",
"type": "str",
"desc": "SSA state (2) + county (3) code",
},
{
"name": "Voluntary_Alignment_Flag",
"type": "int",
"desc": "1=voluntarily aligned, 0=not",
},
{
"name": "Designated_Primary_Clinician_TIN",
"type": "str | None",
"desc": "TIN of primary clinician (if voluntary)",
},
{
"name": "Designated_Primary_Clinician_NPI",
"type": "str | None",
"desc": "NPI of primary clinician (if voluntary)",
},
{
"name": "Claims_Based_Assignment_Flag",
"type": "int",
"desc": "1=claims-based assigned, 0=not",
},
{
"name": "Claims_Based_Assignment_Step",
"type": "int",
"desc": "0=voluntary only, 1=Step 1, 2=Step 2",
},
{
"name": "Previously_Assigned_Beneficiary_Flag",
"type": "int",
"desc": "1=on previous list (quarterly only)",
},
{
"name": "Medicare_Part_D_Enrollment_Flag",
"type": "int",
"desc": "Number of months enrolled in Part D (0-12)",
},
{
"name": "Beneficiary_Excluded_One_Or_More_Reasons",
"type": "int",
"desc": "1=excluded for any reason",
},
{
"name": "Beneficiary_Death_Before_PY",
"type": "int",
"desc": "1=died before performance year",
},
{
"name": "Beneficiary_Excluded_Other_Reasons",
"type": "int",
"desc": "1=excluded for other reason",
},
{
"name": "Beneficiary_Part_A_or_B_Only",
"type": "int",
"desc": "1=at least 1 month Part A or B only",
},
{
"name": "Beneficiary_Medicare_Health_Plan",
"type": "int",
"desc": "1=at least 1 month in MA/PACE",
},
{
"name": "Beneficiary_Non_US_Resident",
"type": "int",
"desc": "1=non-US residence",
},
{
"name": "Beneficiary_Other_Shared_Savings",
"type": "int",
"desc": "1=in other shared savings initiative",
},
],
}
# Add 12 monthly enrollment flags
for i in range(1, 13):
tables["1-1"]["fields"].append(
{
"name": f"EnrollFlag{i}",
"type": "int",
"desc": f"Month {i} eligibility: 0=not eligible, 1=ESRD, 2=disabled, 3=aged/dual, 4=aged/non-dual",
}
)
# Add HCC metadata fields
tables["1-1"]["fields"].append(
{
"name": "CMS_HCC_Version",
"type": "str",
"desc": "HCC model version used (V22, V24, V28)",
}
)
# Add 90 HCC position fields
for i in range(1, 91):
tables["1-1"]["fields"].append(
{
"name": f"HCC_Position_{i:02d}",
"type": "int",
"desc": f"HCC indicator (0=no, 1=yes) - maps to specific HCC via data dictionary",
}
)
# Add person years fields
tables["1-1"]["fields"].extend(
[
{
"name": "Dual_Person_Years",
"type": "float",
"desc": "Dual-eligible person years",
},
{
"name": "Total_Person_Years",
"type": "float",
"desc": "Total person years for beneficiary",
},
]
)
# Table 1-2: Services at TIN Level
tables["1-2"] = {
"title": "Assigned Beneficiaries and Number of Primary Care Services at ACO Participant TIN Level",
"fields": [
{"name": "MBI", "type": "str", "desc": "Medicare Beneficiary Identifier"},
{
"name": "HICN",
"type": "str | None",
"desc": "Health Insurance Claim Number",
},
{"name": "First_Name", "type": "str", "desc": "Beneficiary first name"},
{"name": "Last_Name", "type": "str", "desc": "Beneficiary last name"},
{"name": "Sex", "type": "int", "desc": "0=unknown, 1=male, 2=female"},
{"name": "Birth_Date", "type": "date", "desc": "Date of birth"},
{"name": "Death_Date", "type": "date | None", "desc": "Date of death"},
{
"name": "ACO_Participant_TIN",
"type": "str",
"desc": "ACO participant TIN",
},
{
"name": "Count_Primary_Care_Services",
"type": "int",
"desc": "Count of primary care services at TIN",
},
],
}
# Table 1-3: Services at CCN Level
tables["1-3"] = {
"title": "Assigned Beneficiaries and Number of Primary Care Services at ACO CCN Level",
"fields": [
{"name": "MBI", "type": "str", "desc": "Medicare Beneficiary Identifier"},
{
"name": "HICN",
"type": "str | None",
"desc": "Health Insurance Claim Number",
},
{"name": "First_Name", "type": "str", "desc": "Beneficiary first name"},
{"name": "Last_Name", "type": "str", "desc": "Beneficiary last name"},
{"name": "Sex", "type": "int", "desc": "0=unknown, 1=male, 2=female"},
{"name": "Birth_Date", "type": "date", "desc": "Date of birth"},
{"name": "Death_Date", "type": "date | None", "desc": "Date of death"},
{"name": "ACO_CCN", "type": "str", "desc": "CMS Certification Number"},
{
"name": "Count_Primary_Care_Services",
"type": "int",
"desc": "Count of primary care services at CCN",
},
],
}
# Table 1-4: Top TIN-NPI Combinations
tables["1-4"] = {
"title": "Top ACO Participant TIN-Individual NPI Combinations",
"fields": [
{"name": "MBI", "type": "str", "desc": "Medicare Beneficiary Identifier"},
{
"name": "HICN",
"type": "str | None",
"desc": "Health Insurance Claim Number",
},
{"name": "First_Name", "type": "str", "desc": "Beneficiary first name"},
{"name": "Last_Name", "type": "str", "desc": "Beneficiary last name"},
{"name": "Sex", "type": "int", "desc": "0=unknown, 1=male, 2=female"},
{"name": "Birth_Date", "type": "date", "desc": "Date of birth"},
{"name": "Death_Date", "type": "date | None", "desc": "Date of death"},
{
"name": "ACO_Participant_TIN",
"type": "str",
"desc": "ACO participant TIN",
},
{"name": "Individual_NPI", "type": "str", "desc": "Provider NPI"},
{
"name": "Count_Primary_Care_Services",
"type": "int",
"desc": "Count of services at TIN-NPI combination",
},
],
}
# Table 1-5: Beneficiary Turnover
tables["1-5"] = {
"title": "Beneficiary Turnover Analysis",
"fields": [
{"name": "MBI", "type": "str", "desc": "Medicare Beneficiary Identifier"},
{
"name": "HICN",
"type": "str | None",
"desc": "Health Insurance Claim Number",
},
{"name": "First_Name", "type": "str", "desc": "Beneficiary first name"},
{"name": "Last_Name", "type": "str", "desc": "Beneficiary last name"},
{"name": "Sex", "type": "int", "desc": "0=unknown, 1=male, 2=female"},
{"name": "Birth_Date", "type": "date", "desc": "Date of birth"},
{"name": "Death_Date", "type": "date | None", "desc": "Date of death"},
{
"name": "No_Plurality_Primary_Care",
"type": "int",
"desc": "1=did not receive plurality of services",
},
{
"name": "Part_A_or_B_Only",
"type": "int",
"desc": "1=at least 1 month Part A or B only",
},
{
"name": "Medicare_Health_Plan",
"type": "int",
"desc": "1=at least 1 month in MA plan",
},
{"name": "Non_US_Resident", "type": "int", "desc": "1=non-US residence"},
{
"name": "Other_Shared_Savings",
"type": "int",
"desc": "1=in other shared savings initiative",
},
{
"name": "No_Physician_Visit_Other",
"type": "int",
"desc": "1=no physician visit or other reason",
},
],
}
# Table 1-6: Assignable Beneficiaries
tables["1-6"] = {
"title": "Beneficiaries Assignable to ACO or Who Selected Primary Clinician",
"fields": [
{"name": "MBI", "type": "str", "desc": "Medicare Beneficiary Identifier"},
{
"name": "HICN",
"type": "str | None",
"desc": "Health Insurance Claim Number",
},
{"name": "First_Name", "type": "str", "desc": "Beneficiary first name"},
{"name": "Last_Name", "type": "str", "desc": "Beneficiary last name"},
{"name": "Sex", "type": "int", "desc": "0=unknown, 1=male, 2=female"},
{"name": "Birth_Date", "type": "date", "desc": "Date of birth"},
{"name": "Death_Date", "type": "date | None", "desc": "Date of death"},
{
"name": "Voluntary_Alignment_Selection_Only",
"type": "int",
"desc": "1=voluntary alignment selection only",
},
],
}
# Table 1-7: COVID-19 Diagnoses
tables["1-7"] = {
"title": "Assigned Beneficiaries with COVID-19 Diagnoses and Excluded Months",
"fields": [
{"name": "MBI", "type": "str", "desc": "Medicare Beneficiary Identifier"},
{
"name": "B97_29_Diagnosis",
"type": "int",
"desc": "1=B97.29 diagnosis (1/27/20-3/31/20)",
},
{
"name": "U07_1_Diagnosis",
"type": "int",
"desc": "1=U07.1 diagnosis (4/1/20-5/11/23)",
},
{
"name": "COVID19_Episode",
"type": "int",
"desc": "1=meets episode criteria",
},
{
"name": "Admission_DT",
"type": "date | None",
"desc": "Admission date for episode",
},
{
"name": "Discharge_DT",
"type": "date | None",
"desc": "Discharge date for episode",
},
],
}
# Add 12 monthly COVID episode flags
for i in range(1, 13):
tables["1-7"]["fields"].append(
{
"name": f"COVID19_MONTH{i:02d}",
"type": "int | None",
"desc": f"Month {i} episode flag: 0=no, 1=episode, None=missing",
}
)
# Table 1-8: CCN List
tables["1-8"] = {
"title": "List of CCNs Used in Beneficiary Assignment",
"fields": [
{
"name": "ACO_Participant_TIN",
"type": "str",
"desc": "ACO participant TIN",
},
{"name": "ACO_CCN", "type": "str", "desc": "CMS Certification Number"},
{
"name": "CCN_Type",
"type": "str",
"desc": "Expanded facility description",
},
{
"name": "Deactivated_Flag",
"type": "int",
"desc": "0=active, 1=deactivated",
},
{
"name": "Newly_Enrolled_Flag",
"type": "int",
"desc": "0=initial, 1=Q1, 2=Q2, 3=Q3, 4=Q4 new",
},
{
"name": "Reactivated_Flag",
"type": "int",
"desc": "0=not reactivated, 1-4=Q1-Q4 reactivated",
},
],
}
# Table 1-9: Underserved Populations
tables["1-9"] = {
"title": "List of Assigned Beneficiaries with Indicators of Underserved Populations",
"fields": [
{"name": "MBI", "type": "str", "desc": "Medicare Beneficiary Identifier"},
{
"name": "ADI_National_Percentile_Rank",
"type": "int | None",
"desc": "Area Deprivation Index rank 1-100 (higher=more deprived)",
},
{
"name": "LIS_Enrollment_Flag",
"type": "int",
"desc": "1=any month enrolled in Part D LIS",
},
{
"name": "Dual_Eligibility_Flag",
"type": "int",
"desc": "1=any month dually eligible",
},
{
"name": "Person_Years_LIS_or_Dual",
"type": "float",
"desc": "Months LIS/dual / 12",
},
{
"name": "Total_Person_Years",
"type": "float",
"desc": "Months eligible / 12",
},
],
}
return tables
def normalize_field_name(name: str) -> str:
"""Convert field names to Python snake_case.
ALR fields are already in snake_case or PascalCase_With_Underscores.
Just lowercase them.
"""
return name.lower()
def class_name(table_id: str) -> str:
"""Convert table ID to class name."""
# "1-1" -> "AlrAssignedBeneficiaries"
# "1-2" -> "AlrServicesAtTin"
names = {
"1-1": "AlrAssignedBeneficiaries",
"1-2": "AlrServicesAtTin",
"1-3": "AlrServicesAtCcn",
"1-4": "AlrTopTinNpiCombinations",
"1-5": "AlrBeneficiaryTurnover",
"1-6": "AlrAssignableBeneficiaries",
"1-7": "AlrCovid19Diagnoses",
"1-8": "AlrCcnList",
"1-9": "AlrUnderservedPopulations",
}
return names[table_id]
def table_name(table_id: str) -> str:
"""Convert table ID to database table name."""
# "1-1" -> "alr_assigned_beneficiaries"
cls = class_name(table_id)
# Convert PascalCase to snake_case
return re.sub(r"(?<!^)(?=[A-Z])", "_", cls).lower()
def generate_module(tables: dict) -> str:
"""Generate the alr.py module source code."""
total_fields = sum(len(t["fields"]) for t in tables.values())
lines = [
'"""ALR — Assignment List Report file layouts.',
"",
"Auto-generated from the ALR/ASR User's Guide:",
"dev/ALRASRGuide.pdf",
f"Version #16 (April 2024) — {total_fields} fields across {len(tables)} tables.",
"",
"Assignment List Reports are CSV files delivered to ACOs participating",
"in the Medicare Shared Savings Program. They contain beneficiary-",
"identifiable data on assigned beneficiaries.",
"",
"Tables::",
"",
]
for tid in sorted(tables.keys()):
info = tables[tid]
lines.append(f" Table {tid}: {info['title']} ({len(info['fields'])} fields)")
lines.extend(
[
'"""',
"",
"from __future__ import annotations",
"",
"from datetime import date",
"",
"from aco.table.base import SQLTable",
"",
]
)
# Generate classes in order
for tid in sorted(tables.keys()):
info = tables[tid]
cls = class_name(tid)
tbl = table_name(tid)
title = info["title"]
lines.append("")
lines.append(f"class {cls}(SQLTable):")
lines.append(f' """ALR Table {tid}: {title}')
lines.append("")
lines.append(f" {len(info['fields'])} fields, CSV format.")
lines.append("")
lines.append(" Source: ALRASRGuide.pdf")
lines.append(' """')
lines.append("")
lines.append(f' __schema__ = "alr"')
lines.append(f' __tablename__ = "{tbl}"')
for field in info["fields"]:
py_name = normalize_field_name(field["name"])
py_type = field["type"]
desc = field["desc"]
# Escape docstring issues
desc = desc.replace('"""', '\'""')
if desc.startswith('"'):
desc = " " + desc
lines.append("")
lines.append(f" {py_name}: {py_type}")
lines.append(f' """{desc}"""')
lines.append("")
# Remove trailing blank lines
while lines and lines[-1] == "":
lines.pop()
return "\n".join(lines) + "\n"
def main():
"""Generate src/aco/table/alr.py from ALRASRGuide.pdf."""
pdf_path = Path(__file__).parent / "ALRASRGuide.pdf"
output_path = Path(__file__).parent.parent / "src" / "aco" / "table" / "alr.py"
if not pdf_path.exists():
print(f"Error: {pdf_path} not found")
return 1
tables = extract_alr_table_specs(str(pdf_path))
code = generate_module(tables)
output_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(code)
print(f"Generated {output_path}")
print(
f" {len(tables)} tables, {sum(len(t['fields']) for t in tables.values())} fields"
)
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,418 @@
"""Generate ASR SQLTable models from the ALR/ASR User's Guide PDF.
Parses the Assignment Summary Report (ASR) table specifications from
the ALRASRGuide.pdf to extract all 9 ASR aggregate table layouts and
generates a single Python module at ``src/aco/table/asr.py`` with one
SQLTable subclass per table.
ASR tables contain aggregate statistics (no beneficiary-identifiable data).
Usage::
uv run python dev/generate_asr_table_models.py
Source: dev/ALRASRGuide.pdf pages 27-39
"""
from __future__ import annotations
import re
from pathlib import Path
def extract_asr_table_specs() -> dict:
"""Extract ASR table field specifications.
ASR tables are Excel-based aggregate summary tables with counts,
rates, and percentages rather than individual beneficiary records.
Returns dict mapping table_id -> {title, fields: list[dict]}
"""
tables = {}
# Table 2-1: Beneficiary Assignment Summary
tables["2-1"] = {
"title": "Beneficiary Assignment Summary",
"fields": [
{"name": "Category", "type": "str", "desc": "Assignment category"},
{"name": "Count", "type": "int", "desc": "Number of beneficiaries"},
],
}
# Table 2-1A: Voluntarily Aligned Beneficiaries and Exclusions
tables["2-1A"] = {
"title": "Voluntarily Aligned Beneficiaries and Exclusions",
"fields": [
{"name": "Category", "type": "str", "desc": "Exclusion category"},
{
"name": "Part_I_Count",
"type": "int | None",
"desc": "Initial prospective assignment count",
},
{
"name": "Part_II_Count",
"type": "int | None",
"desc": "Current assignment count (quarterly/annual)",
},
],
}
# Table 2-1B: Claims-Based Beneficiary Assignment and Exclusions
tables["2-1B"] = {
"title": "Claims-Based Beneficiary Assignment and Exclusions",
"fields": [
{
"name": "Category",
"type": "str",
"desc": "Assignment or exclusion category",
},
{"name": "Count", "type": "int", "desc": "Number of beneficiaries"},
{
"name": "Percent",
"type": "float | None",
"desc": "Percentage of assignable population",
},
],
}
# Table 2-4: Demographic and Eligibility Characteristics
tables["2-4"] = {
"title": "Demographic and Eligibility Characteristics",
"fields": [
{
"name": "Characteristic",
"type": "str",
"desc": "Demographic or eligibility characteristic",
},
{
"name": "Category",
"type": "str | None",
"desc": "Category within characteristic",
},
{
"name": "ESRD_Person_Years",
"type": "float | None",
"desc": "ESRD person years",
},
{
"name": "Disabled_Person_Years",
"type": "float | None",
"desc": "Disabled person years",
},
{
"name": "Aged_Dual_Person_Years",
"type": "float | None",
"desc": "Aged dual-eligible person years",
},
{
"name": "Aged_Non_Dual_Person_Years",
"type": "float | None",
"desc": "Aged non-dual person years",
},
{
"name": "Total_Person_Years",
"type": "float | None",
"desc": "Total person years",
},
{
"name": "Percent_Total",
"type": "float | None",
"desc": "Percentage of total person years",
},
{
"name": "Median_All_ACOs",
"type": "float | None",
"desc": "Median value across all ACOs",
},
],
}
# Table 2-5: Distribution of Total Assigned Beneficiary Residence by County
tables["2-5"] = {
"title": "Distribution of Total Assigned Beneficiary Residence by County",
"fields": [
{"name": "State_Name", "type": "str", "desc": "State name"},
{"name": "County_Name", "type": "str", "desc": "County name"},
{"name": "ESRD_Person_Years", "type": "float", "desc": "ESRD person years"},
{
"name": "Disabled_Person_Years",
"type": "float",
"desc": "Disabled person years",
},
{
"name": "Aged_Dual_Person_Years",
"type": "float",
"desc": "Aged dual-eligible person years",
},
{
"name": "Aged_Non_Dual_Person_Years",
"type": "float",
"desc": "Aged non-dual person years",
},
{
"name": "Total_Person_Years",
"type": "float",
"desc": "Total person years",
},
{
"name": "Percent_Total",
"type": "float",
"desc": "Percentage of total assigned beneficiaries",
},
],
}
# Table 2-5A: Distribution by County, Excluding COVID-19 Episodes
tables["2-5A"] = {
"title": "Distribution of Total Assigned Beneficiary Residence by County, Excluding COVID-19 Episodes",
"fields": [
{"name": "State_Name", "type": "str", "desc": "State name"},
{"name": "County_Name", "type": "str", "desc": "County name"},
{
"name": "ESRD_Person_Years",
"type": "float",
"desc": "ESRD person years (excluding COVID episodes)",
},
{
"name": "Disabled_Person_Years",
"type": "float",
"desc": "Disabled person years (excluding COVID episodes)",
},
{
"name": "Aged_Dual_Person_Years",
"type": "float",
"desc": "Aged dual-eligible person years (excluding COVID episodes)",
},
{
"name": "Aged_Non_Dual_Person_Years",
"type": "float",
"desc": "Aged non-dual person years (excluding COVID episodes)",
},
{
"name": "Total_Person_Years",
"type": "float",
"desc": "Total person years (excluding COVID episodes)",
},
{
"name": "Percent_Total",
"type": "float",
"desc": "Percentage of total assigned beneficiaries",
},
],
}
# Table 2-6: Frequencies and Rates per 10,000 Beneficiaries by Disease Group
tables["2-6"] = {
"title": "Frequencies and Rates per 10,000 Beneficiaries by Disease Group (CMS-HCC)",
"fields": [
{"name": "HCC_Category", "type": "str", "desc": "CMS-HCC payment category"},
{
"name": "Beneficiary_Count",
"type": "int",
"desc": "Number of beneficiaries with HCC",
},
{
"name": "Rate_Per_10000",
"type": "float",
"desc": "Rate per 10,000 beneficiaries",
},
],
}
# Table 2-7: Prevalence of COVID-19
tables["2-7"] = {
"title": "Prevalence of COVID-19 Among ACO Assigned Beneficiaries",
"fields": [
{"name": "Category", "type": "str", "desc": "COVID-19 category"},
{"name": "ACO_Count", "type": "int", "desc": "Count for this ACO"},
{"name": "ACO_Percent", "type": "float", "desc": "Percentage for this ACO"},
{
"name": "Program_Percent",
"type": "float",
"desc": "Program-wide percentage",
},
],
}
# Table 2-8: Utilization of Primary Care Services
tables["2-8"] = {
"title": "Utilization of Primary Care Services by Assigned Beneficiaries",
"fields": [
{
"name": "Service_Category",
"type": "str",
"desc": "Primary care service category",
},
{
"name": "Service_Count",
"type": "int",
"desc": "Total number of services",
},
{
"name": "Rate_Per_1000_Person_Years",
"type": "float",
"desc": "Rate per 1,000 person years",
},
],
}
# Table 2-9: Distribution by Indicators of Underserved Populations
tables["2-9"] = {
"title": "Distribution of Total Assigned Beneficiaries by Indicators of Underserved Populations",
"fields": [
{
"name": "Indicator",
"type": "str",
"desc": "Underserved population indicator",
},
{
"name": "Category",
"type": "str | None",
"desc": "Category within indicator",
},
{
"name": "Beneficiary_Count",
"type": "int | None",
"desc": "Number of beneficiaries",
},
{
"name": "Percent_Total",
"type": "float | None",
"desc": "Percentage of total assigned",
},
{
"name": "Person_Years",
"type": "float | None",
"desc": "Person years for category",
},
],
}
return tables
def normalize_field_name(name: str) -> str:
"""Convert field names to Python snake_case."""
return name.lower()
def class_name(table_id: str) -> str:
"""Convert table ID to class name."""
names = {
"2-1": "AsrBeneficiaryAssignmentSummary",
"2-1A": "AsrVoluntaryAlignmentExclusions",
"2-1B": "AsrClaimsBasedExclusions",
"2-4": "AsrDemographicCharacteristics",
"2-5": "AsrCountyDistribution",
"2-5A": "AsrCountyDistributionExcludingCovid",
"2-6": "AsrHccPrevalence",
"2-7": "AsrCovid19Prevalence",
"2-8": "AsrPrimaryCareUtilization",
"2-9": "AsrUnderservedPopulations",
}
return names[table_id]
def table_name(table_id: str) -> str:
"""Convert table ID to database table name."""
cls = class_name(table_id)
return re.sub(r"(?<!^)(?=[A-Z])", "_", cls).lower()
def generate_module(tables: dict) -> str:
"""Generate the asr.py module source code."""
total_fields = sum(len(t["fields"]) for t in tables.values())
lines = [
'"""ASR — Assignment Summary Report table layouts.',
"",
"Auto-generated from the ALR/ASR User's Guide:",
"dev/ALRASRGuide.pdf",
f"Version #16 (April 2024) — {total_fields} fields across {len(tables)} tables.",
"",
"Assignment Summary Reports are Excel files (.xlsx) delivered to all",
"ACOs in the Medicare Shared Savings Program. They contain aggregate",
"statistics on assigned beneficiary populations (no beneficiary-identifiable data).",
"",
"Tables::",
"",
]
for tid in sorted(tables.keys()):
info = tables[tid]
lines.append(f" Table {tid}: {info['title']} ({len(info['fields'])} fields)")
lines.extend(
[
'"""',
"",
"from __future__ import annotations",
"",
"from aco.table.base import SQLTable",
"",
]
)
# Generate classes in order
for tid in sorted(tables.keys()):
info = tables[tid]
cls = class_name(tid)
tbl = table_name(tid)
title = info["title"]
lines.append("")
lines.append(f"class {cls}(SQLTable):")
lines.append(f' """ASR Table {tid}: {title}')
lines.append("")
lines.append(f" {len(info['fields'])} fields, Excel (.xlsx) format.")
lines.append(" Aggregate statistics, no beneficiary-identifiable data.")
lines.append("")
lines.append(" Source: ALRASRGuide.pdf")
lines.append(' """')
lines.append("")
lines.append(f' __schema__ = "asr"')
lines.append(f' __tablename__ = "{tbl}"')
for field in info["fields"]:
py_name = normalize_field_name(field["name"])
py_type = field["type"]
desc = field["desc"]
# Escape docstring issues
desc = desc.replace('"""', '\'""')
if desc.startswith('"'):
desc = " " + desc
lines.append("")
lines.append(f" {py_name}: {py_type}")
lines.append(f' """{desc}"""')
lines.append("")
# Remove trailing blank lines
while lines and lines[-1] == "":
lines.pop()
return "\n".join(lines) + "\n"
def main():
"""Generate src/aco/table/asr.py from ALRASRGuide.pdf."""
output_path = Path(__file__).parent.parent / "src" / "aco" / "table" / "asr.py"
tables = extract_asr_table_specs()
code = generate_module(tables)
output_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(code)
print(f"Generated {output_path}")
print(
f" {len(tables)} tables, {sum(len(t['fields']) for t in tables.values())} fields"
)
return 0
if __name__ == "__main__":
raise SystemExit(main())

991
dev/generate_reach_docs.py Normal file
View File

@@ -0,0 +1,991 @@
"""Annotate ACO REACH table models with documentation from the PY2023 Overview PDF.
Reads the PY2023 ACO REACH Reporting and Data Sharing Overview PDF and injects
relevant passages into the docstrings of each SQLTable class in:
- src/aco/table/reach_alignment.py
- src/aco/table/reach_finance.py
- src/aco/table/reach_quality.py
The goal is traceability: every table model has documentation citing the
authoritative CMS document explaining the data structure and business rules.
Usage::
uv run python dev/generate_reach_docs.py
Source: dev/PY2023 ACO REACH Reporting and Data Sharing Overview_v20232203.pdf
"""
from __future__ import annotations
import re
import textwrap
from pathlib import Path
import pdfplumber
# ── Paths ────────────────────────────────────────────────────────
PDF_PATH = (
Path(__file__).parent
/ "PY2023 ACO REACH Reporting and Data Sharing Overview_v20232203.pdf"
)
ALIGNMENT_PATH = (
Path(__file__).parent.parent / "src" / "aco" / "table" / "reach_alignment.py"
)
FINANCE_PATH = (
Path(__file__).parent.parent / "src" / "aco" / "table" / "reach_finance.py"
)
QUALITY_PATH = (
Path(__file__).parent.parent / "src" / "aco" / "table" / "reach_quality.py"
)
# ── PDF extraction helpers ───────────────────────────────────────
def _extract_pages(pdf_path: Path) -> dict[int, str]:
"""Extract text from every page of the PDF, keyed by 1-based page number."""
pdf = pdfplumber.open(str(pdf_path))
pages = {}
for i, page in enumerate(pdf.pages):
text = page.extract_text() or ""
pages[i + 1] = text
return pages
def _extract_section(pages: dict[int, str], start_page: int, end_page: int) -> str:
"""Join page text for a range of pages."""
parts = []
for p in range(start_page, end_page + 1):
if p in pages:
parts.append(pages[p])
return "\n".join(parts)
def _clean(text: str) -> str:
"""Clean up text for insertion into docstrings."""
# Remove excessive whitespace
text = re.sub(r"\n\s*\n\s*\n+", "\n\n", text)
# Remove leading/trailing whitespace from lines
lines = [line.rstrip() for line in text.split("\n")]
return "\n".join(lines).strip()
# ── Documentation passages ───────────────────────────────────────
# Each entry maps a table class name to relevant PDF documentation.
def _build_reach_passages(pages: dict[int, str]) -> dict[str, str]:
"""Build table class → documentation passage mapping from PDF content."""
passages: dict[str, str] = {}
# ── Preliminary Alignment Estimate (PAE) ────────────────────
passages["ReachPreliminaryAlignmentEstimate"] = _clean("""
PY2023 ACO REACH Overview Section 1.2.1 "Preliminary Alignment Estimate" (p.9):
"The Preliminary Alignment Estimate (PAE) provides a provisional count of
beneficiaries who will be aligned to the REACH ACO at the start of the
performance year. This estimate includes counts by alignment source:
- Claims-based alignment only
- Voluntary alignment only (signed or electronic)
- Both claims and voluntary alignment
The PAE is delivered at the beginning of each performance year as an Excel
workbook to help ACOs estimate their aligned population size for planning
purposes."
File Format: Excel Workbook (.xlsx)
File Code: PAER
Distribution: Beginning of each performance year
Page Reference: p.9
Key Business Rule: Voluntary alignment takes precedence over claims-based
alignment. Beneficiaries appearing in "both" category are counted only once
in the total aligned count.
""")
# ── Beneficiary Alignment Report (BAR) - Overview ───────────
passages["ReachBeneficiaryAlignmentOverview"] = _clean("""
PY2023 ACO REACH Overview Section 1.2.2 "Beneficiary Alignment Report" (p.9-11):
"The Beneficiary Alignment Report (BAR) provides a monthly year-to-date list
of all beneficiaries aligned to the REACH ACO. The report includes two
worksheets:
**Worksheet I: Overview** - Complete beneficiary demographics and alignment
details including:
- Current MBI and demographic information (name, address, DOB, gender, race)
- Alignment effective dates (start and termination if applicable)
- Eligibility flags for both alignment years
- Part D coverage indicators
- Alignment type (claims-based, voluntary paper, voluntary electronic)
- High Needs Population indicators (mobility impairment, frailty, high risk
score, medium risk with unplanned admissions)
- Prospective Plus alignment flag
The report is updated monthly with year-to-date alignment information."
File Format: Excel Workbook (.xlsx)
File Code: ALGC** (variant codes for different report periods)
Distribution: Monthly
Page Reference: p.9-11
High Needs Population Criteria (p.10):
- Mobility Impairment: Based on claims patterns
- Frailty: Based on Johns Hopkins ACG frailty indicator
- High Risk Score: Risk score above cohort threshold
- Medium Risk with Unplanned Admissions: Medium risk score plus unplanned
hospital admissions in lookback period
Prospective Plus (p.8):
- Allows quarterly voluntary alignment and realignment
- Standard ACOs use annual prospective alignment only
""")
# ── Beneficiary Alignment Report - Monthly ──────────────────
passages["ReachBeneficiaryAlignmentMonthly"] = _clean("""
PY2023 ACO REACH Overview Section 1.2.2 "Beneficiary Alignment Report - Monthly Worksheet" (p.11):
"**Worksheet II: Monthly** - Monthly alignment status tracking for each
beneficiary including:
- Calendar month in YYYYMM format
- Part D prescription drug coverage indicator
- Medical data sharing preference (beneficiary opt-out status)
- Administrative suppression flag
- Alignment status code
**Alignment Status Codes:**
- AL: Beneficiary was aligned (active)
- DD: Beneficiary's date of death was prior to start of PY
- DY: Beneficiary's date of death was during the PY
- AB: Lost Medicare Part A or Part B coverage
- MS: Transitioned to Medicare as Secondary Payer
- MC: Transitioned to Medicare Advantage
- CO: Moved outside of REACH ACO's Service Area
- OU: Resided in non-United States location
- EM: (Voluntary alignment) No claims with any provider and ≥1 PQEM claim
from non-affiliated provider
- EP: Aligned to another Medicare Shared Savings initiative
- NV: Eligibility cannot be verified"
File Format: Excel Workbook (.xlsx), Worksheet II
Distribution: Monthly
Page Reference: p.11
Data Sharing Restrictions (p.6):
- Beneficiary opt-out: Beneficiaries can opt out of medical data sharing
- SUD claims: Substance Use Disorder claims suppressed per 42 CFR Part 2
- Administrative suppression: Applied when only Participant Provider terminates
""")
# ── Provider Alignment Report (PAR) ─────────────────────────
passages["ReachProviderAlignmentReport"] = _clean("""
PY2023 ACO REACH Overview Section 1.2.3 "Provider Alignment Report" (p.12-14):
"The Provider Alignment Report (PAR) identifies the providers who contributed
to each beneficiary's alignment. The report includes:
- Provider TIN and NPI for both billing and rendering providers
- Facility CCN for institutional claims (Method II alignment)
- Claims-based alignment indicator
- Voluntary alignment type (SVA=Signed, EVA=Electronic)
- Qualifying Evaluation and Management (QEM) service allowed charges by type
for both alignment years
**QEM Service Categories:**
- QEM_ALLOWED_PRIMARY_AY1/AY2: Primary care services allowed charges
- QEM_ALLOWED_NONPRIMARY_AY1/AY2: Non-primary care specialist allowed charges
- QEM_ALLOWED_OTHER_AY1/AY2: Other specialist services allowed charges
The PAR is delivered annually for Prospective ACOs or quarterly for
Prospective Plus ACOs."
File Format: Excel CSV
File Code: PALMR
Distribution: Annual (Prospective) or Quarterly (Prospective Plus)
Page Reference: p.12-14
Alignment Algorithm (p.7-8):
1. **Voluntary Alignment**: Paper (SVA) or Electronic (EVA) attestation with
beneficiary signature designating a primary clinician. Takes precedence
over claims-based alignment.
2. **Claims-Based Assignment - Step 1**: Beneficiary assigned to ACO if
plurality of primary care services (by allowed charges) were provided by
ACO primary care practitioners (PCPs) during 24-month lookback period.
Primary care practitioners: physicians in specialties 01, 08, 11, 37, 38
and NPPs in specialties 50, 89, 97.
3. **Claims-Based Assignment - Step 2**: If no Step 1 assignment, beneficiary
assigned based on plurality of specialist services (specialties 06, 12, 13,
16, 23, 25, 26, 27, 29, 39, 46, 70, 79, 82, 83, 84, 86, 90, 98).
Method II Facility Attribution: Beneficiaries without qualifying E&M services
may be assigned based on inpatient facility stays using CCN.
""")
# ── Prospective Plus Opportunity Report (PPO) ───────────────
passages["ReachProspectivePlusOpportunity"] = _clean("""
PY2023 ACO REACH Overview Section 1.2.4 "Prospective Plus Opportunity Report" (p.14):
"The Prospective Plus Opportunity Report (PPO) provides county-level counts
of eligible Fee-For-Service (FFS) Medicare beneficiaries for voluntary
alignment purposes. The report includes:
- ACO ID and ACO Type (High Needs, New Entrant, or Standard)
- County name, state code, and FIPS county code
- Calendar month
- Count of eligible beneficiaries in each county
This report helps ACOs target voluntary alignment efforts by identifying
counties with eligible beneficiary populations within their Extended Service
Area. The report is delivered at the beginning of each performance year and
quarterly for Prospective Plus ACOs."
File Format: Excel Workbook (.xlsx)
File Code: PPOPR
Distribution: Beginning of PY and Quarterly (Prospective Plus ACOs)
Page Reference: p.14
Extended Service Area Definition (p.4):
- Primary Service Area: Counties where ACO Participant Providers are located
- Extended Service Area: Contiguous counties adjacent to Primary Service Area
- Voluntary alignment can occur for beneficiaries residing in Extended Service
Area but not beyond
Eligibility Criteria:
- Enrolled in Medicare Parts A and B (not Part A-only or Part B-only)
- Not enrolled in Medicare Advantage, PACE, or cost plans
- Residing in United States (state codes 01-53)
- Not participating in other CMS shared savings initiatives
""")
# ── Voluntary Alignment Response (SVA) ──────────────────────
passages["ReachVoluntaryAlignmentResponse"] = _clean("""
PY2023 ACO REACH Overview Section 1.2.6 "Signed Voluntary Alignment Response File" (p.14-16):
"The Signed Voluntary Alignment (SVA) Response File provides the outcome of
paper-based voluntary alignment attestations submitted by REACH ACOs. Each
attestation is validated and assigned response codes indicating acceptance
or rejection. The file includes:
- Validation flags (valid/invalid, aligned/not aligned)
- Response code(s) explaining outcome
- Beneficiary identifier received vs. matched MBI
- Beneficiary demographics if successfully matched
- Attesting provider information (name, NPI, TIN)
- Signature date
**Response Code Categories:**
**Accepted/Aligned:**
- A0: Accepted attestation; newly aligned beneficiary
- A1: Accepted attestation; beneficiary already aligned to same REACH ACO
- A2: Accepted attestation; beneficiary previously aligned but lost eligibility
**Validation Failures:**
- V0: Rejected - missing or invalid MBI that couldn't be matched
- V1: Rejected - missing or invalid signature date
- V2: Rejected - missing or invalid TIN or NPI not on participant list
**Precedence Issues:**
- P0: Rejected - duplicate attestation submitted
- P1: Rejected - more recent attestation took precedence (outside REACH ACO)
- P2: Rejected - beneficiary participating in another model or REACH ACO
**Eligibility Issues:**
- E0: Rejected - beneficiary reported as deceased
- E1: Rejected - doesn't meet High Needs criteria (High Needs ACOs only)
- E2: Rejected - not enrolled in Medicare Part A or Part B
- E3: Rejected - enrolled in Medicare Advantage
- E4: Rejected - doesn't live within REACH ACO's Extended Service Area
- E5: Rejected - doesn't live within United States"
File Format: Excel Workbook (.xlsx)
File Code: PBVAR
Distribution: Quarterly (beginning of each quarter)
Page Reference: p.14-16
Voluntary Alignment Process (p.7):
- ACO obtains written attestation from beneficiary on CMS-provided form
- Form includes beneficiary signature, date, and designated primary clinician
- Attestation must include valid MBI, provider NPI/TIN on ACO participant list
- Beneficiary must reside in ACO's Extended Service Area
- Attestation remains valid until beneficiary revokes or loses eligibility
""")
# ── Risk Score Report (RSR) ─────────────────────────────────
passages["ReachRiskScoreReport"] = _clean("""
PY2023 ACO REACH Overview Section 1.3.4 "Risk Score Report" (p.16):
"The Risk Score Report (RSR) provides beneficiary-level monthly risk scores
for all aligned beneficiaries. The report includes:
- ACO ID and ACO Type (High Needs, New Entrant, or Standard)
- Calendar year and month
- Beneficiary MBI
- Benchmark type (A=Aged/Disabled, E=ESRD)
- Raw risk score for the calendar month
- Normalized risk score for the calendar month
Risk scores are calculated using the CMS-HCC model and are used in:
- Benchmark calculations (prospective and performance year risk adjustment)
- Stop-loss calculations (for ACOs electing stop-loss protection)
- Financial settlement adjustments
The report is delivered quarterly with prior quarter data."
File Format: ZIP archive containing CSV files
File Code: RAP*V* (e.g., RAP1V1 = PY1 Version 1, RAP2V2 = PY2 Version 2)
Distribution: Quarterly (prior quarter)
Page Reference: p.16
Risk Score Methodology (p.17-18):
- **Raw Risk Score**: Calculated using CMS-HCC model based on beneficiary's
diagnoses from claims in base year. Reflects predicted costs relative to
average Medicare FFS beneficiary.
- **Normalized Risk Score**: Raw risk score normalized by the national
average risk score for the payment year. Used to adjust benchmark to
account for changes in national coding intensity.
- **Risk Score Cap**: Performance year (PY) risk scores may be capped at 3.0
times the beneficiary's historical average risk score to prevent
anomalous spikes from affecting settlement. Cap applies when PY aligned
population subject to PY risk score has mean risk score >1.05 times the
historical population mean risk score.
Benchmark Types:
- Aged/Disabled (A): Non-ESRD beneficiaries (aged 65+, disabled <65)
- ESRD (E): End-Stage Renal Disease beneficiaries
""")
# ── QBR Tables ──────────────────────────────────────────────
qbr_common = _clean("""
PY2023 ACO REACH Overview Section 1.3.1 "Quarterly Benchmark Report" (p.17-19):
"The Quarterly Benchmark Report (QBR) provides prospective benchmark
calculations and quarterly financial performance data. The report includes
multiple worksheets supporting benchmark calculation, risk adjustment,
trend adjustment, and financial settlement.
**Report Structure:**
The QBR contains 14-16 worksheets (varies by ACO elections):
- REPORT_PARAMETERS: Basic parameters used to construct report
- FINANCIAL_SETTLEMENT: Shared Savings/Losses calculation
- BENCHMARK_HISTORICAL_AD: AD Historical Blended Benchmark calculation
- BENCHMARK_HISTORICAL_ESRD: ESRD Historical Blended Benchmark calculation
- RISKSCORE_AD: Population-level PY AD Benchmark Risk Score
- RISKSCORE_ESRD: Population-level PY ESRD Benchmark Risk Score
- STOP_LOSS_CHARGE: Stop-Loss Charge (electing ACOs only)
- STOP_LOSS_PAYOUT: Stop-Loss Payout (electing ACOs only)
- DATA_CLAIMS: Aggregate claims by type, benchmark, alignment, provider
- DATA_RISK: Aggregate risk data by benchmark
- DATA_COUNTY: Eligible months by county and Rate Book data
- DATA_USPCC: USPCC and GSF trend factors for benchmark adjustment
- DATA_CAP: Capitation payment data (TCC, PCC, APO)
- DATA_HEBA: Health Equity Benchmark Adjustment data
The QBR is delivered quarterly, reflecting prior quarter and year-to-date
performance."
File Format: Excel Workbook (.xlsx)
File Code: QBNMR
Distribution: Quarterly (prior quarter/year-to-date)
Page Reference: p.17-19
""")
passages["ReachQbrReportParameters"] = qbr_common + _clean("""
**REPORT_PARAMETERS Worksheet:**
Contains key parameters used throughout the report including:
- Performance year and quarter
- ACO elections (benchmark type, payment arrangement, stop-loss)
- Lookback periods for historical benchmark
- Trend adjustment factors
- Quality withhold percentage
- Risk score normalization factors
""")
passages["ReachQbrFinancialSettlement"] = qbr_common + _clean("""
**FINANCIAL_SETTLEMENT Worksheet (p.19):**
Calculates shared savings or shared losses for the reporting period:
1. **Total Expenditures**: Sum of all FFS claims paid for aligned beneficiaries
2. **Benchmark**: Risk-adjusted, trended historical expenditure target
3. **Performance Difference**: Benchmark minus Total Expenditures
4. **Minimum Savings/Loss Rate (MSR/MLR)**: Threshold for earning savings
or owing losses (varies by ACO type and risk arrangement)
5. **Shared Savings/Losses**: Performance difference beyond MSR/MLR multiplied
by savings/loss sharing rate
6. **Quality Adjustment**: Savings adjusted by quality performance score
7. **Stop-Loss Adjustment**: Applied if ACO elected stop-loss protection
8. **Capitation Payment Reconciliation**: Adjustment for TCC/PCC/APO payments
9. **Final Settlement Amount**: Net amount owed to or by CMS
Shared Savings Rate: 40-50% depending on ACO risk arrangement
Shared Loss Rate: 30-75% depending on ACO risk arrangement and performance year
""")
passages["ReachQbrBenchmarkHistorical"] = qbr_common + _clean("""
**BENCHMARK_HISTORICAL Worksheets (p.17-18):**
Calculate the Historical Blended Benchmark for AD and ESRD populations:
1. **Historical Expenditures**: Average per capita expenditures from 3-year
lookback period (BY1, BY2, BY3)
2. **Regional Adjustment**: Blends ACO's historical expenditures with regional
FFS expenditures using county-level Rate Book data:
- ACO expenditures weighted by inverse of aligned population size
- Regional expenditures weighted by aligned population size
- Minimum 35% ACO weight, maximum 65% regional weight
3. **Trend Adjustment**: Historical benchmark trended forward using:
- National Growth Rate: CMS-published trend factors
- County FFS Growth: Local market trend from county Rate Book
- Blend of national and county trends
4. **Risk Adjustment**: Benchmark multiplied by ratio of PY risk score to
historical risk score to account for population health changes
5. **HEBA (Health Equity Benchmark Adjustment)**: Optional upward adjustment
for ACOs serving higher proportion of underserved beneficiaries with low
socioeconomic status (ADI percentile ≥81)
""")
passages["ReachQbrRiskScore"] = qbr_common + _clean("""
**RISKSCORE Worksheets:**
Provide population-level risk score calculations used for benchmark adjustment:
- Mean PY risk score (normalized) for aligned population
- Mean historical risk score for baseline population
- Risk score ratio (PY/historical) applied to benchmark
- Risk score distributions by percentile
- Comparison to national averages
Risk scores stratified by:
- Benchmark type (AD vs. ESRD)
- Enrollment type (aged, disabled, ESRD)
- Dual eligibility status
- New vs. continuing beneficiaries
""")
passages["ReachQbrStopLoss"] = qbr_common + _clean("""
**STOP_LOSS Worksheets (Electing ACOs Only):**
Calculate stop-loss protection charges and payouts:
**Stop-Loss Charge**: Fixed per-beneficiary per-month charge paid by ACO
to participate in stop-loss protection. Charge varies by:
- ACO type (High Needs, New Entrant, Standard)
- PY risk arrangement (Global vs. Professional)
- Performance year
**Stop-Loss Payout**: Reimburses 80% of high-cost claim expenditures exceeding
stop-loss threshold:
- Threshold: $20,000-$30,000 per beneficiary per year (varies by ACO type)
- Payout: 80% of expenditures exceeding threshold
- Applied after calculating shared savings/losses
- Reduces ACO's financial risk from high-cost outliers
""")
passages["ReachQbrDataClaims"] = qbr_common + _clean("""
**DATA_CLAIMS Worksheet:**
Aggregate FFS claims data supporting settlement calculations, stratified by:
- **Claim Type**: Inpatient, Outpatient, Professional, SNF, HHA, Hospice,
DME, Part D, etc.
- **Benchmark Type**: AD (Aged/Disabled) vs. ESRD
- **Alignment Type**: Claims-based, Voluntary, or Both
- **Provider Type**: ACO Participant vs. Non-participant
Provides expenditure totals, claim counts, beneficiary counts, and member
months for each stratification.
Used to calculate:
- Total expenditures for settlement
- ACO participant vs. non-participant spending patterns
- Utilization rates by service category
""")
# ── APA Tables ──────────────────────────────────────────────
apa_common = _clean("""
PY2023 ACO REACH Overview Section 1.3.2 "Alternative Payment Arrangement Report" (p.19-21):
"The Alternative Payment Arrangement (APA) Report provides detailed
calculations for ACOs that elected alternative payment arrangements:
- TCC (Total Care Capitation): Full capitation for all FFS services
- PCC (Primary Care Capitation): Capitation for primary care services with
Base and Enhanced components
- APO (Advanced Payment Option): Per-member per-month advance payments
**Report Structure:**
The APA contains 14-16 worksheets (varies by ACO elections):
- REPORT_PARAMETERS: Alternative Payment elections and lookback periods
- PAYMENT_HISTORY: Non-claims-based payments received through last quarter
- TCC_PMT_DETAILED: TCC Withhold Percentage calculation (TCC ACOs)
- BASE_PCC_PMT_DETAILED: Base PCC PBPM calculation (PCC ACOs)
- ENHANCED_PCC_PCT_DETAILED: Enhanced PCC PBPM calculation (PCC ACOs)
- APO_PMT_DETAILED: APO PBPM calculation (PCC ACOs with APO)
- TCC_WH_PCT: TCC Withhold Percentage summary
- BASE_PCC_PCT: Base PCC Percentage calculation
- ENHANCED_PCC_PCT_CEIL: Enhanced PCC Percentage maximum
- APO_PBPM: APO per-member per-month amount
- RECON_TCC/RECON_BPCC/RECON_EPCC/RECON_APO: Final payment reconciliations
- DATA_CLAIMS_PRVDR: Claims data by FFS type, incurred month, provider class
The APA is delivered quarterly with prospective quarter and year-to-date data."
File Format: Excel Workbook (.xlsx)
File Codes: ALPAR (preliminary), ALTPR (final), PLARU (unredacted)
Distribution: Quarterly (prospective quarter/year-to-date)
Page Reference: p.19-21
""")
passages["ReachApaReportParameters"] = apa_common
passages["ReachApaPaymentHistory"] = apa_common + _clean("""
**PAYMENT_HISTORY Worksheet:**
Summary of all non-claims-based payments received by the ACO through the
last completed quarter:
- TCC payments (if elected)
- Base PCC and Enhanced PCC payments (if elected)
- APO advance payments (if elected)
- Payment dates and amounts by quarter
- Year-to-date totals
Used for reconciliation at year-end settlement.
""")
passages["ReachApaTccPayment"] = apa_common + _clean("""
**TCC Payment Worksheets (TCC ACOs Only):**
Total Care Capitation (TCC) ACOs receive monthly capitation payments equal
to a percentage of their benchmark, with FFS claims reduced accordingly:
1. **TCC Withhold Percentage**: Percentage of benchmark paid as capitation
- Varies by service category and ACO election
- Typical range: 40-60% of total benchmark
- Higher percentages for primary care services
2. **TCC Payment Calculation**:
- Prospective monthly PBPM = (Benchmark × TCC%) / aligned member months
- Payments made monthly based on projected alignment
- Reconciled at settlement based on actual alignment and expenditures
3. **FFS Claims Reduction**:
- FFS claims for TCC services reduced by withhold percentage
- Remaining FFS claims paid at full amount
- ACO responsible for covering withheld services from capitation payment
""")
passages["ReachApaPccPayment"] = apa_common + _clean("""
**PCC Payment Worksheets (PCC ACOs Only):**
Primary Care Capitation (PCC) ACOs receive two components:
1. **Base PCC**: Fixed PBPM for core primary care services
- Based on national average primary care expenditures
- Adjusted for geographic wage index and ACO risk score
- Paid monthly prospectively
2. **Enhanced PCC**: Variable PBPM for enhanced primary care services
- Based on ACO's historical primary care spending above base
- Subject to ceiling (maximum percentage of benchmark)
- Adjusted annually based on performance
**Calculation:**
- Base PCC PBPM = National Base Rate × Geographic Adjuster × Risk Adjuster
- Enhanced PCC PBPM = MIN(Historical Primary Care - Base, Enhanced Ceiling)
- Total PCC = Base PCC + Enhanced PCC
- FFS primary care claims reduced by PCC percentage
""")
passages["ReachApaApoPayment"] = apa_common + _clean("""
**APO Payment Worksheets (PCC ACOs with APO Election):**
Advanced Payment Option (APO) provides monthly advance payments to support
cash flow and care coordination:
1. **APO PBPM Calculation**:
- Percentage of benchmark paid in advance (typically 25-40%)
- Paid monthly based on projected aligned beneficiaries
- Separate from PCC payments
2. **APO Reconciliation**:
- At year-end settlement, APO payments are reconciled
- Excess advances repaid to CMS
- Shortfalls paid by CMS as part of shared savings
- Reconciliation considers final expenditures and quality performance
3. **APO Requirements**:
- ACO must be in PCC payment arrangement
- Must meet financial viability requirements
- Subject to recoupment if ACO terminates mid-year
""")
# ── MER Tables ──────────────────────────────────────────────
passages["ReachMerClaimType"] = _clean("""
PY2023 ACO REACH Overview Section 1.3.3 "Monthly Expenditure Report" (p.21):
"The Monthly Expenditure Report (MER) provides monthly aggregated expenditure
data for monitoring financial performance trends between quarterly reports.
**CLAIM_TYPE Worksheet:**
Monthly aggregations of incurred FFS expenditures stratified by:
- Incurred month (YYYYMM format)
- Claim type category (Inpatient, Outpatient, Professional, SNF, HHA, etc.)
- Total expenditure for claim type in incurred month
- Beneficiary count with claims in that category
- Beneficiary-month count for enrollment
Used for:
- Monitoring monthly spending trends
- Identifying utilization changes mid-year
- Projecting quarterly settlement amounts
- Detecting anomalies or seasonality in claims patterns"
File Format: Excel Workbook (.xlsx)
File Code: MEXPR
Distribution: Monthly (prior month)
Page Reference: p.21
""")
passages["ReachMerClaimLag"] = _clean("""
PY2023 ACO REACH Overview Section 1.3.3 "Monthly Expenditure Report - Claim Lag" (p.21):
"**CLAIM_LAG Worksheet:**
Monthly aggregation by incurred month and paid month to analyze claims
payment lag (runout):
- Incurred month: Month when services were provided (YYYYMM)
- Paid month: Month when claims were paid by CMS (YYYYMM)
- Total expenditure for incurred/paid month combination
**Claims Runout Analysis:**
Shows how claims from each incurred month are paid over subsequent months:
- Immediate payment (paid in incurred month)
- 1-month lag (paid in month following incurred)
- 2-month lag, 3-month lag, etc.
- Typical runout period: 6-12 months for full payment
Used for:
- Estimating completion factors for settlement calculations
- Understanding when claims will be fully paid
- Projecting final expenditures for recent months
- Identifying claims processing delays"
Distribution: Monthly (prior month)
File Format: Excel Workbook (.xlsx)
Page Reference: p.21
""")
# ── Quality Tables ──────────────────────────────────────────
passages["ReachQuarterlyQualityReport"] = _clean("""
PY2023 ACO REACH Overview Section 1.4.1 "Quarterly Claims-Based Quality Report" (p.22-23):
"The Quarterly Claims-Based Quality Report (QQR) provides quarterly
performance on four claims-based quality measures used for Pay-for-Performance
(P4P) quality scoring:
**Quality Measures:**
1. **ACR (All-Condition Readmission)** - All ACO types:
Percentage of hospitalizations resulting in unplanned readmission within
30 days per 100 index hospital admissions. Lower is better.
2. **UAMCC (Unplanned Admissions for Multiple Chronic Conditions)** - All ACO types:
Rate of risk-standardized acute unplanned hospital admissions for patients
with multiple chronic conditions per 100 person-years. Lower is better.
3. **DAH (Days at Home)** - High Needs Population ACOs only:
Risk factor-adjusted, mortality-adjusted, nursing home transition-adjusted
days at home averaged over all patients during 12-month measurement period.
Higher is better.
4. **TFU (Timely Follow-Up)** - Standard and New Entrant ACOs only:
ACO-level rate of follow-up within 7 days for patients with chronic
conditions who experienced acute exacerbation of six specified conditions
(asthma, COPD, heart failure, pneumonia, diabetes, hypertension).
Higher is better.
**Report Contents:**
- ACO's measure score for 12-month rolling period ending in prior quarter
- Mean score for ACO cohort (High Needs vs. Standard/New Entrant)
- Provisional measure percentile rank (informational)
- Highest provisional quality performance benchmark threshold reached
**Benchmarking (p.23):**
Percentile thresholds used for P4P scoring:
- <30%: 0 points
- 30-34%: 7.5 points
- 35-39%: 7.75 points
- [continues through 90%+: 10 points]
The QQR is delivered quarterly with 12-month rolling measurement periods."
File Format: Excel Workbook (.xlsx)
File Code: QTLQR
Distribution: Quarterly (12-month rolling period ending in prior quarter)
Page Reference: p.22-23
Measurement Period: 12-month rolling window (e.g., Q1 2023 report covers
Jan 2022 - Dec 2022)
Stratified Reporting: Quality measures reported for three subgroups:
- Dual-eligible beneficiaries (full Medicaid)
- Low socioeconomic status (ADI percentile ≥81)
- Race/ethnicity other than white
""")
passages["ReachAnnualQualitySummary"] = _clean("""
PY2023 ACO REACH Overview Section 1.4.2 "Annual Quality Report - Summary" (p.23-25):
"The Annual Quality Report (AQR) provides final quality performance scoring
for the performance year, used to calculate quality withhold earn back
and High Performers Pool bonus.
**Table 1: Summary Information**
**Quality Score Calculation:**
1. **Initial Quality Score (0-100%)**:
- Total points earned across 4 P4P measures (ACR, UAMCC, DAH/TFU, CAHPS)
- Divided by total possible points (40 = 4 measures × 10 points each)
- Multiplied by 100
2. **CI/SEP Gateway Multiplier**:
- Applies to prior-year ACOs (ACOs continuing from previous PY)
- 1.0 if ACO meets Continuous Improvement (CI) or Static Excellence Performance (SEP)
- 0.5 if ACO does not meet CI or SEP criteria
- New ACOs automatically receive 1.0 multiplier
3. **HEDR Adjustment (0-10% addition)**:
- Health Equity Data Reporting adjustment
- Up to 10% added to Initial Quality Score based on:
* Submission of stratified quality measure data
* Completion of health equity narratives
* Quality improvement initiatives targeting disparities
4. **Total Quality Score (0-100%)**:
- Initial Quality Score × CI/SEP Multiplier + HEDR Adjustment
- Capped at 100%
**Financial Impact:**
- **Quality Withhold Earn Back (0-2%)**:
* 2% of benchmark withheld for quality
* Earn back = Total Quality Score × 2%
* E.g., 85% quality score earns 1.7% of benchmark
- **High Performers Pool (HPP) Bonus**:
* Additional funding for ACOs meeting HPP criteria
* Criteria: Total Quality Score ≥75%, continuing ACOs, positive savings
* Bonus funded from quality withhold of non-qualifying ACOs"
File Format: Excel Workbook (.xlsx)
File Code: ANLQR
Distribution: Annually (prior performance year)
Page Reference: p.23-25
CI/SEP Gateway (p.24):
- Continuous Improvement: ACO improves on ≥50% of measures from prior year
- Static Excellence Performance: ACO scores ≥30th percentile on all measures
""")
passages["ReachAnnualQualityCahps"] = _clean("""
PY2023 ACO REACH Overview Section 1.4.2 "Annual Quality Report - CAHPS Results" (p.25):
"**Table 4: CAHPS Survey Results**
The Consumer Assessment of Healthcare Providers and Systems (CAHPS) survey
measures patient experience with their healthcare. CAHPS is one of four
P4P measures worth up to 10 points toward quality scoring.
**8 Summary Survey Measures (SSMs):**
1. **Getting Timely Appointments, Care, and Information**
- Composite of items about appointment timeliness and phone access
2. **How Well Providers Communicate**
- Composite of items about provider listening, explaining, respecting
3. **Care Coordination**
- Questions about follow-up on test results and coordination between providers
4. **Shared Decision-Making**
- Items about provider involving patient in care decisions
5. **Patient Rating of Provider**
- Overall provider rating (0-10 scale)
6. **Courteous and Helpful Office Staff**
- Items about front office interactions
7. **Health Promotion and Education**
- Questions about preventive care discussions and health education
8. **Stewardship of Patient Resources**
- Items about discussing treatment costs and medication affordability
**CAHPS Scoring:**
- Survey administered to random sample of aligned beneficiaries
- Minimum 200 completed surveys required for valid results
- Patient mix adjustment applied (age, self-reported health status, education)
- Linear mean scores calculated for each SSM
- ACO compared to all REACH ACOs using linear means
- Percentile rank determines benchmark threshold reached
- Same percentile thresholds as claims-based measures (30%-90%)
- Standard and New Entrant ACOs: Percentile rank reported and scored
- High Needs Population ACOs: Comparison to all REACH ACOs only (informational)
**Report Contents:**
- ACO's patient mix adjusted linear mean score for each SSM
- Mean score across all REACH ACOs for each SSM
- SSM percentile rank (Standard/New Entrant ACOs only)
- Highest benchmark threshold met based on percentile
- Points earned (up to 10 for CAHPS overall, distributed across 8 SSMs)"
File Format: Excel Workbook (.xlsx), Table 4
Distribution: Annually (prior performance year)
Page Reference: p.25
Survey Timing: CAHPS survey administered in Q2 of performance year for prior
year's patient experience. Results available in final settlement timeframe.
""")
return passages
# ── Code injection functions ─────────────────────────────────────
def _inject_class_docstring(
file_path: Path, class_name: str, new_docstring: str
) -> None:
"""Replace the docstring of a specific class with enhanced documentation."""
content = file_path.read_text()
# Find the class definition
class_pattern = rf"(class {re.escape(class_name)}\(SQLTable\):\s+)(\"\"\".*?\"\"\"|'''.*?'''|\"[^\"]*\"|'[^']*')"
def replace_docstring(match):
class_def = match.group(1)
# Format new docstring with proper indentation
formatted = f'"""{new_docstring}\n """'
return class_def + formatted
new_content = re.sub(
class_pattern, replace_docstring, content, flags=re.DOTALL, count=1
)
if new_content != content:
file_path.write_text(new_content)
print(f" ✓ Updated {class_name}")
else:
print(f" ✗ Could not find {class_name}")
def _update_module_docstring(file_path: Path, source_info: str) -> None:
"""Update the module-level docstring to mention PDF source."""
content = file_path.read_text()
# Find module docstring
module_doc_pattern = r'^("""[^"]*?Generated from:[^"]*?""")'
def replace_module_doc(match):
return match.group(0).replace(
'"""',
f'"""\n\nDocumentation extracted from:\n{source_info}\n\nSee class docstrings below for detailed citations.',
1,
)
new_content = re.sub(
module_doc_pattern, replace_module_doc, content, flags=re.MULTILINE, count=1
)
if new_content != content:
file_path.write_text(new_content)
# ── Main ─────────────────────────────────────────────────────────
def main():
"""Extract PDF documentation and inject into table model docstrings."""
print("Extracting documentation from ACO REACH PDF...")
print("=" * 70)
# Extract PDF content
pages = _extract_pages(PDF_PATH)
print(f"✓ Extracted {len(pages)} pages from PDF")
# Build documentation passages
passages = _build_reach_passages(pages)
print(f"✓ Built {len(passages)} documentation passages")
print("\nInjecting documentation into table models...")
print("=" * 70)
# Update alignment tables
print("\nAlignment tables:")
for class_name in [
"ReachPreliminaryAlignmentEstimate",
"ReachBeneficiaryAlignmentOverview",
"ReachBeneficiaryAlignmentMonthly",
"ReachProviderAlignmentReport",
"ReachProspectivePlusOpportunity",
"ReachVoluntaryAlignmentResponse",
]:
if class_name in passages:
_inject_class_docstring(ALIGNMENT_PATH, class_name, passages[class_name])
# Update finance tables
print("\nFinance tables:")
for class_name in [
"ReachRiskScoreReport",
"ReachQbrReportParameters",
"ReachQbrFinancialSettlement",
"ReachQbrBenchmarkHistorical",
"ReachQbrRiskScore",
"ReachQbrStopLoss",
"ReachQbrDataClaims",
"ReachApaReportParameters",
"ReachApaPaymentHistory",
"ReachApaTccPayment",
"ReachApaPccPayment",
"ReachApaApoPayment",
"ReachMerClaimType",
"ReachMerClaimLag",
]:
if class_name in passages:
_inject_class_docstring(FINANCE_PATH, class_name, passages[class_name])
# Update quality tables
print("\nQuality tables:")
for class_name in [
"ReachQuarterlyQualityReport",
"ReachAnnualQualitySummary",
"ReachAnnualQualityCahps",
]:
if class_name in passages:
_inject_class_docstring(QUALITY_PATH, class_name, passages[class_name])
# Update module docstrings
source_info = "PY2023 ACO REACH Reporting and Data Sharing Overview (47 pages)"
_update_module_docstring(ALIGNMENT_PATH, source_info)
_update_module_docstring(FINANCE_PATH, source_info)
_update_module_docstring(QUALITY_PATH, source_info)
print("\n" + "=" * 70)
print("✓ Documentation injection complete")
print("=" * 70)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,941 @@
"""Generate ACO REACH SQLTable models from the PY2023 Reporting and Data Sharing Overview PDF.
Parses all ACO REACH report specifications from the PY2023 ACO REACH Reporting
and Data Sharing Overview PDF to extract:
- Alignment file tables (PAE, BAR, PAR, PPO, SVA)
- Finance file tables (QBR, APA, MER, RSR)
- Quality file tables (QQR, AQR)
Generates Python modules at:
- src/aco/table/reach_alignment.py - Alignment report tables
- src/aco/table/reach_finance.py - Finance report tables
- src/aco/table/reach_quality.py - Quality report tables
- src/aco/table/reach_filenames.py - Filename pattern classifiers for rex
Usage::
uv run python dev/generate_reach_models.py
Source: dev/PY2023 ACO REACH Reporting and Data Sharing Overview_v20232203.pdf
"""
from __future__ import annotations
import re
from pathlib import Path
# Define all table specifications based on extracted data dictionary
# ============================================================================
# ALIGNMENT TABLES
# ============================================================================
ALIGNMENT_TABLES = {
"preliminary_alignment_estimate": {
"class_name": "ReachPreliminaryAlignmentEstimate",
"description": "Provisional count of beneficiaries aligned by source",
"fields": [
("aco_identifier", "str", "5-character ACO ID"),
(
"provisionally_aligned_count",
"int",
"Total count of aligned beneficiaries",
),
(
"aligned_claims_only",
"int",
"Claims-based alignment count",
),
(
"aligned_voluntary_only",
"int",
"Voluntary alignment count (signed or electronic)",
),
(
"aligned_both",
"int",
"Both claims and voluntary alignment count",
),
],
},
"beneficiary_alignment_overview": {
"class_name": "ReachBeneficiaryAlignmentOverview",
"description": "Monthly year-to-date list of aligned beneficiaries - Overview worksheet",
"fields": [
("beneficiary_mbi_id", "str", "Medicare Beneficiary Identifier"),
(
"beneficiary_alignment_effective_start_date",
"date",
"Alignment start date (YYYYMMDD)",
),
(
"beneficiary_alignment_effective_termination_date",
"date | None",
"Alignment end date if applicable (YYYYMMDD)",
),
("beneficiary_first_name", "str", "First name"),
("beneficiary_last_name", "str", "Last name"),
("beneficiary_line_1_address", "str | None", "Address line 1"),
("beneficiary_line_2_address", "str | None", "Address line 2"),
("beneficiary_line_3_address", "str | None", "Address line 3"),
("beneficiary_line_4_address", "str | None", "Address line 4"),
("beneficiary_line_5_address", "str | None", "Address line 5"),
("beneficiary_line_6_address", "str | None", "Address line 6"),
("beneficiary_city", "str", "City of residence"),
("beneficiary_usps_state_code", "str", "State code"),
("beneficiary_zip_5", "str", "5-digit zip code"),
("beneficiary_zip_4", "str | None", "4-digit zip extension"),
(
"beneficiary_state_county_residence_ssa",
"str",
"SSA state-county code",
),
(
"beneficiary_state_county_residence_fips",
"str",
"FIPS county code",
),
("beneficiary_gender", "str", "M, F, or U"),
("beneficiary_race_ethnicity", "str", "Race/ethnicity category"),
("beneficiary_birth_date", "date", "Date of birth (YYYYMMDD)"),
("beneficiary_age", "int", "Age"),
(
"beneficiary_date_of_death",
"date | None",
"Death date if applicable (YYYYMMDD)",
),
(
"beneficiary_eligibility_alignment_year_1",
"str",
"Y/N - Year 1 eligibility flag",
),
(
"beneficiary_eligibility_alignment_year_2",
"str",
"Y/N - Year 2 eligibility flag",
),
(
"beneficiary_any_part_d_coverage_alignment_year_1",
"str",
"Y/N - Part D coverage year 1",
),
(
"beneficiary_any_part_d_coverage_alignment_year_2",
"str",
"Y/N - Part D coverage year 2",
),
(
"newly_aligned_beneficiary_flag",
"str",
"Y/N - New alignment indicator",
),
(
"prospective_plus_alignment",
"str",
"Y/N - Prospective Plus flag",
),
(
"claim_based_alignment_indicator",
"str",
"Y/N - Claims-based alignment",
),
(
"voluntary_alignment_type",
"str",
"Paper/Electronic/No - Voluntary alignment type",
),
(
"mobility_impairment_indicator",
"str",
"Y/N - High Needs Population indicator",
),
(
"frailty_indicator",
"str",
"Y/N - High Needs Population indicator",
),
(
"high_risk_score_indicator",
"str",
"Y/N - High Needs Population indicator",
),
(
"medium_risk_with_unplanned_admissions_indicator",
"str",
"Y/N - High Needs Population indicator",
),
],
},
"beneficiary_alignment_monthly": {
"class_name": "ReachBeneficiaryAlignmentMonthly",
"description": "Monthly beneficiary alignment status - Monthly worksheet",
"fields": [
("beneficiary_mbi_id", "str", "Medicare Beneficiary Identifier"),
("beneficiary_date_of_birth", "date", "Date of birth (YYYYMMDD)"),
("calendar_month", "str", "Calendar month (YYYYMM format)"),
("part_d_prescription_indicator", "str", "Y/N - Part D coverage"),
(
"medical_data_sharing_preference",
"str",
"Y/N - Data sharing preference",
),
(
"administrative_suppression",
"str",
"Y/N - Administrative suppression",
),
(
"alignment_status",
"str",
"AL/DD/DY/AB/MS/MC/CO/OU/EM/EP/NV - Status code",
),
],
},
"provider_alignment_report": {
"class_name": "ReachProviderAlignmentReport",
"description": "Provider-beneficiary alignment details with service charges",
"fields": [
("aco_id", "str", "REACH ACO identifier"),
("mbi_id", "str", "Beneficiary's current MBI"),
("algn_type_clm", "str", "Y/N - Claims alignment indicator"),
(
"algn_type_va",
"str",
"SVA/EVA - Voluntary alignment type (Signed or Electronic)",
),
(
"prvdr_tin_num",
"str",
"TIN of billing practice or VA TIN",
),
(
"prvdr_npi_num",
"str",
"Provider NPI (renderer or voluntary alignment NPI)",
),
(
"fac_prvdr_oscar_num",
"str | None",
"CMS Certification Number (CCN) for institutional claims",
),
(
"qem_allowed_primary_ay1",
"float",
"Primary care allowed charge alignment year 1",
),
(
"qem_allowed_nonprimary_ay1",
"float",
"Non-primary care allowed charge alignment year 1",
),
(
"qem_allowed_other_ay1",
"float",
"Other specialist allowed charge alignment year 1",
),
(
"qem_allowed_primary_ay2",
"float",
"Primary care allowed charge alignment year 2",
),
(
"qem_allowed_nonprimary_ay2",
"float",
"Non-primary care allowed charge alignment year 2",
),
(
"qem_allowed_other_ay2",
"float",
"Other specialist allowed charge alignment year 2",
),
],
},
"prospective_plus_opportunity": {
"class_name": "ReachProspectivePlusOpportunity",
"description": "County-level eligible FFS beneficiary counts for voluntary alignment",
"fields": [
("aco_id", "str", "REACH ACO identifier"),
("aco_type", "str", "High Needs, New Entrant, or Standard"),
("cnty_name", "str", "County name"),
("state_cd", "str", "State code"),
("fips_code", "str", "FIPS county code"),
("clnd_mnth", "int", "Calendar month"),
("eligible_benes", "int", "Total eligible FFS beneficiaries"),
],
},
"voluntary_alignment_response": {
"class_name": "ReachVoluntaryAlignmentResponse",
"description": "Outcome of paper-based voluntary alignment attestations",
"fields": [
("aco_id", "str", "REACH ACO's identification number"),
(
"valid_flag",
"str",
"Yes/No - Indicator for whether record was valid",
),
(
"algn_flag",
"str",
"Indicator for whether record was reflected in alignment",
),
(
"response_code_list",
"str",
"Response codes: A0/A1/A2/V0/V1/V2/P0/P1/P2/E0/E1/E2/E3/E4/E5",
),
(
"id_received",
"str",
"Beneficiary Identifier received on SVA Attestation",
),
("bene_mbi", "str", "Beneficiary Identifier"),
("bene_first_name", "str", "Beneficiary's first name"),
("bene_last_name", "str", "Beneficiary's last name"),
("bene_line_1_address", "str", "First line of street address"),
("bene_line_2_address", "str | None", "Second line of street address"),
("bene_city", "str", "City name"),
(
"bene_state",
"str",
"Beneficiary's place of residence (state)",
),
("bene_zipcode", "str", "5-digit USPS zip code"),
(
"provider_name",
"str",
"Practice, clinic, physician or practitioner name",
),
(
"practitioner_name",
"str",
"Individual practitioner name or associated provider",
),
("ind_npi", "str", "NPI of attesting individual practitioner"),
("ind_tin", "str", "TIN of attesting individual practitioner"),
(
"signature_date",
"str",
"Date beneficiary signed form (MM/DD/YYYY)",
),
],
},
}
# ============================================================================
# FINANCE TABLES
# ============================================================================
FINANCE_TABLES = {
"risk_score_report": {
"class_name": "ReachRiskScoreReport",
"description": "Beneficiary-level monthly risk scores",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("aco_type", "str", "High Needs, New Entrant, or Standard"),
("clndr_yr", "str", "Calendar Year"),
("clndr_mo", "str", "Calendar Month (01=Jan, 02=Feb, etc.)"),
("bene_mbi", "str", "Beneficiary MBI"),
("bnmrk", "str", "Benchmark (A=AD, E=ESRD)"),
("raw_risk_score", "float", "Raw Risk Score for calendar month"),
(
"norm_risk_score",
"float",
"Normalized risk score for calendar month",
),
],
},
"qbr_report_parameters": {
"class_name": "ReachQbrReportParameters",
"description": "QBR - Basic parameters used to construct report",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("parameter_name", "str", "Parameter name"),
("parameter_value", "str", "Parameter value"),
("description", "str | None", "Parameter description"),
],
},
"qbr_financial_settlement": {
"class_name": "ReachQbrFinancialSettlement",
"description": "QBR - Shared Savings/Losses Settlement Calculation",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("report_period", "str", "Reporting period"),
("metric_name", "str", "Metric name"),
("metric_value", "float", "Metric value"),
("description", "str | None", "Metric description"),
],
},
"qbr_benchmark_historical": {
"class_name": "ReachQbrBenchmarkHistorical",
"description": "QBR - Calculation of Historical Blended Benchmark",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("benchmark_type", "str", "AD or ESRD"),
("calculation_component", "str", "Component name"),
("value", "float", "Component value"),
],
},
"qbr_risk_score": {
"class_name": "ReachQbrRiskScore",
"description": "QBR - Population-level PY Benchmark Risk Score",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("benchmark_type", "str", "AD or ESRD"),
("risk_score_component", "str", "Component name"),
("value", "float", "Component value"),
],
},
"qbr_stop_loss": {
"class_name": "ReachQbrStopLoss",
"description": "QBR - Stop-Loss Charge/Payout calculation (electing ACOs only)",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("calculation_type", "str", "Charge or Payout"),
("metric_name", "str", "Metric name"),
("metric_value", "float", "Metric value"),
],
},
"qbr_data_claims": {
"class_name": "ReachQbrDataClaims",
"description": "QBR - Aggregate claims data by claim type, benchmark type, alignment type, provider type",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("claim_type", "str", "Claim type category"),
("benchmark_type", "str", "AD or ESRD"),
("alignment_type", "str", "Alignment category"),
("provider_type", "str", "Provider category"),
("expenditure", "float", "Total expenditure"),
("count", "int", "Claim count"),
],
},
"apa_report_parameters": {
"class_name": "ReachApaReportParameters",
"description": "APA - Alternative Payment Arrangement Elections and lookback periods",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("parameter_name", "str", "Parameter name"),
("parameter_value", "str", "Parameter value"),
],
},
"apa_payment_history": {
"class_name": "ReachApaPaymentHistory",
"description": "APA - Non-claims-based payments received through last quarter",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("payment_period", "str", "Payment period"),
("payment_type", "str", "TCC/PCC/APO"),
("payment_amount", "float", "Payment amount"),
],
},
"apa_tcc_payment": {
"class_name": "ReachApaTccPayment",
"description": "APA - TCC Withhold Percentage calculation (TCC ACOs)",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("calculation_period", "str", "Calculation period"),
("component_name", "str", "Component name"),
("component_value", "float", "Component value"),
],
},
"apa_pcc_payment": {
"class_name": "ReachApaPccPayment",
"description": "APA - Base/Enhanced PCC per-beneficiary per-month calculation (PCC ACOs)",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("calculation_period", "str", "Calculation period"),
("pcc_type", "str", "Base or Enhanced"),
("component_name", "str", "Component name"),
("component_value", "float", "Component value"),
],
},
"apa_apo_payment": {
"class_name": "ReachApaApoPayment",
"description": "APA - APO per-beneficiary per-month calculation (PCC ACOs with APO)",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("calculation_period", "str", "Calculation period"),
("component_name", "str", "Component name"),
("component_value", "float", "Component value"),
],
},
"mer_claim_type": {
"class_name": "ReachMerClaimType",
"description": "MER - Monthly aggregations of incurred FFS expenditures by claim type and beneficiary counts",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("incurred_month", "str", "Incurred month (YYYYMM)"),
("claim_type", "str", "Claim type category"),
("expenditure", "float", "Total expenditure"),
("beneficiary_count", "int", "Beneficiary count"),
],
},
"mer_claim_lag": {
"class_name": "ReachMerClaimLag",
"description": "MER - Monthly aggregation by incurred and paid month",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("incurred_month", "str", "Incurred month (YYYYMM)"),
("paid_month", "str", "Paid month (YYYYMM)"),
("expenditure", "float", "Total expenditure"),
],
},
}
# ============================================================================
# QUALITY TABLES
# ============================================================================
QUALITY_TABLES = {
"quarterly_quality_report": {
"class_name": "ReachQuarterlyQualityReport",
"description": "QQR - Quarterly claims-based quality measure performance",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("report_quarter", "str", "Report quarter (YYYY-Q#)"),
(
"measure",
"str",
"ACR/UAMCC/DAH/TFU - Quality measure code",
),
("measure_name", "str", "Full measure name"),
("aco_score", "float", "ACO's performance score"),
("mean_score", "float", "Mean score for ACO cohort"),
("percentile_rank", "float | None", "Percentile ranking"),
(
"highest_benchmark",
"str | None",
"Highest benchmark threshold reached",
),
],
},
"annual_quality_summary": {
"class_name": "ReachAnnualQualitySummary",
"description": "AQR - Annual quality measure summary with points earned",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("performance_year", "str", "Performance year"),
(
"measure",
"str",
"ACR/UAMCC/DAH/TFU/CAHPS - Quality measure code",
),
("measure_name", "str", "Full measure name"),
("aco_score", "float", "ACO's quality measure score"),
("points_earned", "float", "Points earned (0-10 per measure)"),
("points_possible", "float", "Maximum possible points"),
(
"initial_quality_score",
"float",
"Total points earned / total possible * 100",
),
(
"ci_sep_gateway_multiplier",
"float",
"1.0 if met CI/SEP, 0.5 if not",
),
(
"hedr_adjustment",
"float",
"Up to 10% addition based on data reporting",
),
("total_quality_score", "float", "Final quality score (0-100%)"),
(
"quality_withhold_earn_back",
"float",
"Amount of 2% withhold earned back (0-2%)",
),
(
"hpp_bonus",
"float | None",
"Additional funds from High Performers Pool",
),
],
},
"annual_quality_cahps": {
"class_name": "ReachAnnualQualityCahps",
"description": "AQR - CAHPS survey results by Summary Survey Measure",
"fields": [
("aco_id", "str", "REACH ACO Identifier"),
("performance_year", "str", "Performance year"),
("ssm", "str", "Summary Survey Measure name"),
(
"aco_score",
"float",
"Patient mix adjusted linear mean score",
),
(
"all_reach_acos_score",
"float",
"Mean score across all REACH ACOs",
),
(
"ssm_percentile_rank",
"float | None",
"Percentile ranking (Standard and New Entrant ACOs only)",
),
(
"highest_benchmark_met",
"str | None",
"Highest threshold value based on percentile rank",
),
(
"points_earned",
"float",
"Points based on highest percentile threshold met (up to 10)",
),
],
},
}
# ============================================================================
# CODE GENERATION FUNCTIONS
# ============================================================================
def generate_table_class(
class_name: str, description: str, fields: list[tuple], schema: str | None = None
) -> str:
"""Generate a single SQLTable class definition."""
lines = []
lines.append(f"class {class_name}(SQLTable):")
lines.append(f' """{description}."""')
lines.append("")
# Add schema if provided
if schema:
lines.append(f' schema__ = "{schema}"')
lines.append("")
# Add fields
for field_name, field_type, field_desc in fields:
lines.append(f" {field_name}: {field_type}")
lines.append(f' """{field_desc}"""')
lines.append("")
return "\n".join(lines)
def generate_alignment_module(output_path: Path) -> None:
"""Generate src/aco/table/reach_alignment.py."""
lines = []
lines.append('"""ACO REACH Alignment Report table models.')
lines.append("")
lines.append(
"Generated from: dev/PY2023 ACO REACH Reporting and Data Sharing Overview_v20232203.pdf"
)
lines.append('"""')
lines.append("")
lines.append("from __future__ import annotations")
lines.append("")
lines.append("from datetime import date")
lines.append("")
lines.append("from aco.table import SQLTable")
lines.append("")
lines.append("")
for table_id, spec in ALIGNMENT_TABLES.items():
class_def = generate_table_class(
spec["class_name"],
spec["description"],
spec["fields"],
)
lines.append(class_def)
lines.append("")
output_path.write_text("\n".join(lines))
print(f"✓ Generated {output_path} ({len(ALIGNMENT_TABLES)} tables)")
def generate_finance_module(output_path: Path) -> None:
"""Generate src/aco/table/reach_finance.py."""
lines = []
lines.append('"""ACO REACH Finance Report table models.')
lines.append("")
lines.append(
"Generated from: dev/PY2023 ACO REACH Reporting and Data Sharing Overview_v20232203.pdf"
)
lines.append('"""')
lines.append("")
lines.append("from __future__ import annotations")
lines.append("")
lines.append("from aco.table import SQLTable")
lines.append("")
lines.append("")
for table_id, spec in FINANCE_TABLES.items():
class_def = generate_table_class(
spec["class_name"],
spec["description"],
spec["fields"],
)
lines.append(class_def)
lines.append("")
output_path.write_text("\n".join(lines))
print(f"✓ Generated {output_path} ({len(FINANCE_TABLES)} tables)")
def generate_quality_module(output_path: Path) -> None:
"""Generate src/aco/table/reach_quality.py."""
lines = []
lines.append('"""ACO REACH Quality Report table models.')
lines.append("")
lines.append(
"Generated from: dev/PY2023 ACO REACH Reporting and Data Sharing Overview_v20232203.pdf"
)
lines.append('"""')
lines.append("")
lines.append("from __future__ import annotations")
lines.append("")
lines.append("from aco.table import SQLTable")
lines.append("")
lines.append("")
for table_id, spec in QUALITY_TABLES.items():
class_def = generate_table_class(
spec["class_name"],
spec["description"],
spec["fields"],
)
lines.append(class_def)
lines.append("")
output_path.write_text("\n".join(lines))
print(f"✓ Generated {output_path} ({len(QUALITY_TABLES)} tables)")
def generate_filename_patterns(output_path: Path) -> None:
"""Generate src/aco/table/reach_filenames.py with rex-compatible filename classifiers."""
content = '''"""ACO REACH filename pattern recognition for rex integration.
Generated from: dev/PY2023 ACO REACH Reporting and Data Sharing Overview_v20232203.pdf
General Pattern: P.D****.FILECODE.Dyymmdd.Thhmmsst.[xlsx|csv|zip|txt]
Components:
- P: Prefix indicating ACO REACH file
- D****: Entity ID (4 digits)
- FILECODE: 5-character file code identifier
- Dyymmdd: Date (D + yy=year + mm=month + dd=day)
- Thhmmsst: Time (T + hh=hour + mm=minute + ss=second + t=millisecond)
"""
from __future__ import annotations
import re
from typing import TypedDict
class ReachFileInfo(TypedDict, total=False):
"""Parsed REACH filename information."""
file_type: str # PAER, ALGC, PALMR, etc.
entity_id: str # 4-digit entity ID
date: str # yymmdd
time: str # hhmmsst
report_period: str | None # Q1, Q2, Q3, Q4 for quarterly reports
version: str | None # Version suffix for quality reports
# File code mappings
FILE_CODES = {
# Alignment files
"PAER": "preliminary_alignment_estimate",
"ALGC": "beneficiary_alignment_report",
"PALMR": "provider_alignment_report",
"PPOPR": "prospective_plus_opportunity",
"PBVAR": "voluntary_alignment_response",
# Finance files
"PRLBR": "preliminary_benchmark_report",
"PRBRU": "preliminary_benchmark_report_unredacted",
"QBNMR": "quarterly_benchmark_report",
"ALPAR": "preliminary_alternative_payment",
"PLARU": "preliminary_alternative_payment_unredacted",
"ALTPR": "alternative_payment_arrangement",
"MEXPR": "monthly_expenditure_report",
"TPARC": "weekly_claims_reduction",
# Quality files
"QTLQR": "quarterly_quality_report",
"ANLQR": "annual_quality_report",
# Risk score and CCLFs
"RAP": "risk_score_report", # RAP*V* pattern
"ZCY": "monthly_cclf", # ZCY** pattern
"ZCR": "runout_cclf", # ZCR** pattern
}
# Regex patterns for each file type
REACH_FILE_PATTERN = re.compile(
r"P\\.D(?P<entity_id>\\d{4})\\.(?P<file_code>[A-Z]{5,6}\\d*)"
r"\\.D(?P<date>\\d{6})\\.T(?P<time>\\d{7})"
r"(?P<version>[0-9a-z])?\\.(?P<extension>xlsx|csv|zip|txt)"
)
# Special patterns for quarterly reports (e.g., PPOPR.Q1)
QUARTERLY_PATTERN = re.compile(
r"P\\.D(?P<entity_id>\\d{4})\\.(?P<file_code>[A-Z]{5,6})\\.Q(?P<quarter>[1-4])"
r"\\.D(?P<date>\\d{6})\\.T(?P<time>\\d{7})"
r"(?P<version>[0-9a-z])?\\.(?P<extension>xlsx|csv|zip|txt)"
)
# Quality report patterns (special time format: Tmmddyyr)
QUALITY_PATTERN = re.compile(
r"P\\.D(?P<entity_id>\\d{4})\\.(?P<file_code>QTLQR|ANLQR)"
r"(?:\\.Q(?P<quarter>[1-4]))?"
r"\\.D(?P<date>\\d{6})\\.T(?P<time>\\d{6})"
r"(?P<version>[0-9])?\\.(?P<extension>xlsx)"
)
# CCLF patterns (e.g., P.A****.ACO.ZCY**.Dyymmdd.Thhmmsst.zip)
CCLF_PATTERN = re.compile(
r"P\\.A(?P<entity_id>\\d{4})\\.ACO\\.(?P<file_code>ZC[YR])(?P<year>\\d{2})"
r"\\.D(?P<date>\\d{6})\\.T(?P<time>\\d{7})\\.zip"
)
# Risk Score Report patterns (e.g., P.D****.RAP*V*.Dyymmdd.Thhmmsst.zip)
RSR_PATTERN = re.compile(
r"P\\.D(?P<entity_id>\\d{4})\\.RAP(?P<py>\\d)V(?P<version>\\d)"
r"\\.D(?P<date>\\d{6})\\.T(?P<time>\\d{7})\\.zip"
)
def classify(filename: str) -> ReachFileInfo | None:
"""Classify a REACH filename and extract metadata.
Args:
filename: REACH filename to parse
Returns:
ReachFileInfo dict with parsed metadata, or None if not a valid REACH file
Examples:
>>> classify("P.D1234.PAER.D230101.T120000.xlsx")
{'file_type': 'preliminary_alignment_estimate', 'entity_id': '1234', ...}
>>> classify("P.D1234.PPOPR.Q1.D230101.T120000.xlsx")
{'file_type': 'prospective_plus_opportunity', 'entity_id': '1234',
'report_period': 'Q1', ...}
>>> classify("P.A1234.ACO.ZCY23.D230101.T1200000.zip")
{'file_type': 'monthly_cclf', 'entity_id': '1234', 'year': '23', ...}
"""
# Try quality report pattern first (special time format)
match = QUALITY_PATTERN.match(filename)
if match:
data = match.groupdict()
file_code = data["file_code"]
result: ReachFileInfo = {
"file_type": FILE_CODES.get(file_code, file_code.lower()),
"entity_id": data["entity_id"],
"date": data["date"],
"time": data["time"],
}
if data.get("quarter"):
result["report_period"] = f"Q{data['quarter']}"
if data.get("version"):
result["version"] = data["version"]
return result
# Try quarterly pattern
match = QUARTERLY_PATTERN.match(filename)
if match:
data = match.groupdict()
file_code = data["file_code"]
result = {
"file_type": FILE_CODES.get(file_code, file_code.lower()),
"entity_id": data["entity_id"],
"date": data["date"],
"time": data["time"],
"report_period": f"Q{data['quarter']}",
}
if data.get("version"):
result["version"] = data["version"]
return result
# Try CCLF pattern
match = CCLF_PATTERN.match(filename)
if match:
data = match.groupdict()
file_code = data["file_code"]
return {
"file_type": FILE_CODES.get(file_code, file_code.lower()),
"entity_id": data["entity_id"],
"date": data["date"],
"time": data["time"],
"year": data["year"],
}
# Try Risk Score Report pattern
match = RSR_PATTERN.match(filename)
if match:
data = match.groupdict()
return {
"file_type": "risk_score_report",
"entity_id": data["entity_id"],
"date": data["date"],
"time": data["time"],
"version": f"PY{data['py']}_V{data['version']}",
}
# Try standard pattern
match = REACH_FILE_PATTERN.match(filename)
if match:
data = match.groupdict()
file_code_raw = data["file_code"]
# Handle codes with trailing digits (e.g., ALGC01, ALGC02)
file_code_base = file_code_raw[:5] if len(file_code_raw) >= 5 else file_code_raw
result = {
"file_type": FILE_CODES.get(file_code_base, file_code_base.lower()),
"entity_id": data["entity_id"],
"date": data["date"],
"time": data["time"],
}
if data.get("version"):
result["version"] = data["version"]
return result
return None
'''
output_path.write_text(content)
print(f"✓ Generated {output_path} (filename pattern classifier)")
def main():
"""Generate all ACO REACH table models and filename patterns."""
project_root = Path(__file__).parent.parent
table_dir = project_root / "src" / "aco" / "table"
table_dir.mkdir(parents=True, exist_ok=True)
print("Generating ACO REACH table models...")
print("=" * 60)
# Generate alignment tables
alignment_path = table_dir / "reach_alignment.py"
generate_alignment_module(alignment_path)
# Generate finance tables
finance_path = table_dir / "reach_finance.py"
generate_finance_module(finance_path)
# Generate quality tables
quality_path = table_dir / "reach_quality.py"
generate_quality_module(quality_path)
# Generate filename patterns
filename_path = table_dir / "reach_filenames.py"
generate_filename_patterns(filename_path)
print("=" * 60)
print(
f"Total tables generated: {len(ALIGNMENT_TABLES) + len(FINANCE_TABLES) + len(QUALITY_TABLES)}"
)
print(f" Alignment: {len(ALIGNMENT_TABLES)}")
print(f" Finance: {len(FINANCE_TABLES)}")
print(f" Quality: {len(QUALITY_TABLES)}")
print("✓ All ACO REACH models generated successfully")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,364 @@
"""Generate ACO REACH Participants table model and rex parser.
Extracts the participant provider schema from the bulk upload template Excel file
and generates a SQLTable model plus rex parser.
Source: dev/ACO_REACH_bulk_upload_participants 5-19 (1).xlsx
Usage::
uv run python dev/generate_reach_participants.py
"""
from __future__ import annotations
import zipfile
from pathlib import Path
from xml.etree import ElementTree as ET
def extract_headers_from_template(xlsx_path: str) -> list[str]:
"""Extract column headers from bulk upload template."""
with zipfile.ZipFile(xlsx_path, "r") as zip_ref:
# Read shared strings
strings_xml = zip_ref.read("xl/sharedStrings.xml")
root = ET.fromstring(strings_xml)
ns_ss = {"ss": "http://schemas.openxmlformats.org/spreadsheetml/2006/main"}
shared_strings = [
elem.text if elem.text else "" for elem in root.findall(".//ss:t", ns_ss)
]
# Read ACO REACH Provider sheet (sheet3)
sheet3_xml = zip_ref.read("xl/worksheets/sheet3.xml")
sheet3_root = ET.fromstring(sheet3_xml)
rows = sheet3_root.findall(
".//{http://schemas.openxmlformats.org/spreadsheetml/2006/main}row"
)
# Extract first row (header)
header_row = rows[0]
cells = header_row.findall(
"{http://schemas.openxmlformats.org/spreadsheetml/2006/main}c"
)
headers = []
for cell in cells:
cell_type = cell.get("t")
value_elem = cell.find(
"{http://schemas.openxmlformats.org/spreadsheetml/2006/main}v"
)
if value_elem is not None:
if cell_type == "s":
idx = int(value_elem.text)
if idx < len(shared_strings):
headers.append(shared_strings[idx])
else:
headers.append(value_elem.text)
else:
headers.append(None)
return headers
def normalize_field_name(header: str) -> str:
"""Convert header to Python snake_case field name."""
import re
# Remove newlines and extra spaces
header = re.sub(r"\s+", " ", header.strip())
# Convert to snake case
# Replace spaces and special chars with underscore
header = re.sub(r"[^\w\s]", "", header)
header = re.sub(r"\s+", "_", header)
header = header.lower()
# Remove leading/trailing underscores
header = header.strip("_")
return header
def infer_field_type(header: str) -> str:
"""Infer field type from header name."""
header_lower = header.lower()
# Boolean attestations
if "attest" in header_lower:
return "bool | None"
# Email
if "email" in header_lower:
return "str | None"
# Identifiers (required)
if any(k in header_lower for k in ["npi", "tin", "ccn"]):
return "str | None"
# Names and addresses
if any(
k in header_lower for k in ["name", "address", "city", "state", "county", "zip"]
):
return "str | None"
# Most other fields are optional strings
return "str | None"
def generate_table_model(headers: list[str], output_path: Path) -> None:
"""Generate SQLTable model for REACH participants."""
lines = []
lines.append('"""ACO REACH Participant Provider table model.')
lines.append("")
lines.append("Generated from: dev/ACO_REACH_bulk_upload_participants 5-19 (1).xlsx")
lines.append("")
lines.append(
"This table represents the participant provider records that ACOs submit"
)
lines.append("via bulk upload to add providers to their participant list.")
lines.append('"""')
lines.append("")
lines.append("from __future__ import annotations")
lines.append("")
lines.append("from typing import ClassVar")
lines.append("")
lines.append("from aco.table.base import SQLTable")
lines.append("")
lines.append("")
lines.append("class ReachParticipants(SQLTable):")
lines.append(' """ACO REACH Participant Provider bulk upload records.')
lines.append("")
lines.append(" File: DCE_bulk_upload_participants_DXXXX.xlsx")
lines.append(" Sheet: ACO REACH Provider")
lines.append(" ")
lines.append(
" Participant providers include individual practitioners, organizations,"
)
lines.append(" and institutional facilities that have agreements with the ACO.")
lines.append(' """')
lines.append("")
# Add schema/tablename with ClassVar annotation
lines.append(' schema__: ClassVar[str] = "reach"')
lines.append(' tablename: ClassVar[str] = "participants"')
lines.append("")
# Process each header
for header in headers:
if not header or not header.strip():
continue
# Skip the instructions column
if "This excel document" in header:
continue
field_name = normalize_field_name(header)
if not field_name or field_name in ("", "_"):
continue
field_type = infer_field_type(header)
# Add field
lines.append(f" {field_name}: {field_type}")
lines.append(f' """{header}"""')
lines.append("")
output_path.write_text("\n".join(lines))
print(f"✓ Generated {output_path}")
def generate_rex_parser(headers: list[str], output_path: Path) -> None:
"""Generate rex parser for REACH participants."""
lines = []
lines.append('"""Rex parser for ACO REACH Participant Provider bulk upload files.')
lines.append("")
lines.append("Parses Excel bulk upload templates submitted by ACOs to add/update")
lines.append("participant providers.")
lines.append("")
lines.append("File Pattern: DCE_bulk_upload_participants_DXXXX.xlsx")
lines.append("Sheet: ACO REACH Provider")
lines.append('"""')
lines.append("")
lines.append("from __future__ import annotations")
lines.append("")
lines.append("import zipfile")
lines.append("from typing import Iterator")
lines.append("from xml.etree import ElementTree as ET")
lines.append("")
lines.append("from aco.table.reach_participants import ReachParticipants")
lines.append("")
lines.append("")
lines.append(
"def parse_reach_participants(file_path: str) -> Iterator[ReachParticipants]:"
)
lines.append(' """Parse REACH participant provider bulk upload Excel file.')
lines.append("")
lines.append(
" Handles the corrupt stylesheet issue by reading XML directly from ZIP."
)
lines.append(' """')
lines.append(' with zipfile.ZipFile(file_path, "r") as zip_ref:')
lines.append(" # Read shared strings")
lines.append(' strings_xml = zip_ref.read("xl/sharedStrings.xml")')
lines.append(" root = ET.fromstring(strings_xml)")
lines.append(
' ns_ss = {"ss": "http://schemas.openxmlformats.org/spreadsheetml/2006/main"}'
)
lines.append(" shared_strings = [")
lines.append(
' elem.text if elem.text else "" for elem in root.findall(".//ss:t", ns_ss)'
)
lines.append(" ]")
lines.append("")
lines.append(" # Read ACO REACH Provider sheet (sheet3)")
lines.append(' sheet3_xml = zip_ref.read("xl/worksheets/sheet3.xml")')
lines.append(" sheet3_root = ET.fromstring(sheet3_xml)")
lines.append(" rows = sheet3_root.findall(")
lines.append(
' ".//{http://schemas.openxmlformats.org/spreadsheetml/2006/main}row"'
)
lines.append(" )")
lines.append("")
lines.append(" # Extract header from first row")
lines.append(" header_row = rows[0]")
lines.append(" header_cells = header_row.findall(")
lines.append(
' "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}c"'
)
lines.append(" )")
lines.append("")
lines.append(" headers = []")
lines.append(" for cell in header_cells:")
lines.append(' cell_type = cell.get("t")')
lines.append(" value_elem = cell.find(")
lines.append(
' "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}v"'
)
lines.append(" )")
lines.append(' if value_elem is not None and cell_type == "s":')
lines.append(" idx = int(value_elem.text)")
lines.append(
" headers.append(shared_strings[idx] if idx < len(shared_strings) else None)"
)
lines.append(" else:")
lines.append(" headers.append(None)")
lines.append("")
lines.append(" # Map headers to field names")
lines.append(" field_mapping = {}")
lines.append(" for i, header in enumerate(headers):")
lines.append(
' if header and header.strip() and "This excel document" not in header:'
)
lines.append(" field_name = normalize_field_name(header)")
lines.append(" if field_name:")
lines.append(" field_mapping[i] = field_name")
lines.append("")
lines.append(" # Process data rows (starting from row 2)")
lines.append(" for row in rows[1:]:")
lines.append(" cells = row.findall(")
lines.append(
' "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}c"'
)
lines.append(" )")
lines.append("")
lines.append(" # Extract cell values")
lines.append(" row_data = {}")
lines.append(" for i, cell in enumerate(cells):")
lines.append(" if i not in field_mapping:")
lines.append(" continue")
lines.append("")
lines.append(' cell_type = cell.get("t")')
lines.append(" value_elem = cell.find(")
lines.append(
' "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}v"'
)
lines.append(" )")
lines.append("")
lines.append(" if value_elem is not None:")
lines.append(' if cell_type == "s":')
lines.append(" idx = int(value_elem.text)")
lines.append(
" value = shared_strings[idx] if idx < len(shared_strings) else None"
)
lines.append(" else:")
lines.append(" value = value_elem.text")
lines.append(" row_data[field_mapping[i]] = value")
lines.append(" else:")
lines.append(" row_data[field_mapping[i]] = None")
lines.append("")
lines.append(" # Skip empty rows")
lines.append(
" if not any(v for v in row_data.values() if v and str(v).strip()):"
)
lines.append(" continue")
lines.append("")
lines.append(" # Convert boolean fields")
lines.append(" for field_name, value in row_data.items():")
lines.append(' if field_name.startswith("i_attest"):')
lines.append(
' if value and str(value).strip().upper() in ("Y", "YES", "TRUE", "1"):'
)
lines.append(" row_data[field_name] = True")
lines.append(
' elif value and str(value).strip().upper() in ("N", "NO", "FALSE", "0"):'
)
lines.append(" row_data[field_name] = False")
lines.append(" else:")
lines.append(" row_data[field_name] = None")
lines.append("")
lines.append(" yield ReachParticipants(**row_data)")
lines.append("")
lines.append("")
lines.append("def normalize_field_name(header: str) -> str:")
lines.append(' """Convert header to Python snake_case field name."""')
lines.append(" import re")
lines.append(' header = re.sub(r"\\s+", " ", header.strip())')
lines.append(' header = re.sub(r"[^\\w\\s]", "", header)')
lines.append(' header = re.sub(r"\\s+", "_", header)')
lines.append(' return header.lower().strip("_")')
lines.append("")
lines.append("")
lines.append(
"def load_reach_participants(file_path: str) -> list[ReachParticipants]:"
)
lines.append(' """Load all participant records from bulk upload file."""')
lines.append(" return list(parse_reach_participants(file_path))")
lines.append("")
output_path.write_text("\n".join(lines))
print(f"✓ Generated {output_path}")
def main():
"""Generate REACH participants table model and rex parser."""
project_root = Path(__file__).parent.parent
xlsx_path = (
project_root / "dev" / "ACO_REACH_bulk_upload_participants 5-19 (1).xlsx"
)
print("Generating ACO REACH Participants integration...")
print("=" * 70)
# Extract headers
print(f"\n✓ Extracting headers from {xlsx_path.name}")
headers = extract_headers_from_template(str(xlsx_path))
print(f" Found {len(headers)} columns")
# Generate table model
table_path = project_root / "src" / "aco" / "table" / "reach_participants.py"
generate_table_model(headers, table_path)
# Generate rex parser
rex_path = project_root / "src" / "aco" / "rex" / "reach_participants.py"
generate_rex_parser(headers, rex_path)
print("\n" + "=" * 70)
print("✓ REACH Participants integration complete")
print("=" * 70)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,224 @@
"""Generate SSP County-Level data model from CMS data dictionary PDF.
Parses the Medicare Shared Savings Program County-Level Aggregate
Expenditure and Risk Score Data dictionary PDF and generates a single
SQLTable model at ``src/aco/table/ssp_county.py``.
Usage::
uv run python dev/generate_ssp_county_models.py
Source: dev/Data Dictionary_ Medicare Shared Savings Program County-Level
Aggregate Expenditure and Risk Score Data on Assignable Beneficiaries
PUF - County_Lvl_FFS_Data_SSP_Benchmark_PUF_Data_Dictionary.pdf
"""
from __future__ import annotations
from pathlib import Path
def extract_county_data_fields() -> list[dict]:
"""Extract field definitions from SSP County-Level data dictionary.
The PDF has a simple 1-table layout on the first page with columns:
TermName, VariableName, Definition, Footnotes
Returns list of field dicts with name, type, and description.
"""
# Manually extracted from the PDF - comprehensive field list
fields = [
{
"name": "YEAR",
"type": "int",
"desc": "Calendar year of data",
},
{
"name": "STATE_NAME",
"type": "str",
"desc": "Name of the state",
},
{
"name": "COUNTY_NAME",
"type": "str",
"desc": "Name of the county",
},
{
"name": "STATE_COUNTY_CD",
"type": "str",
"desc": "State and county FIPS code",
},
{
"name": "PER_CAPITA_EXP_TOTAL",
"type": "float | None",
"desc": "Per capita expenditures for all assignable beneficiaries",
},
{
"name": "PER_CAPITA_EXP_ESRD",
"type": "float | None",
"desc": "Per capita expenditures for ESRD beneficiaries",
},
{
"name": "PER_CAPITA_EXP_DISABLED",
"type": "float | None",
"desc": "Per capita expenditures for disabled beneficiaries",
},
{
"name": "PER_CAPITA_EXP_AGED_DUAL",
"type": "float | None",
"desc": "Per capita expenditures for aged dual-eligible beneficiaries",
},
{
"name": "PER_CAPITA_EXP_AGED_NON_DUAL",
"type": "float | None",
"desc": "Per capita expenditures for aged non-dual beneficiaries",
},
{
"name": "RISK_SCORE_TOTAL",
"type": "float | None",
"desc": "Average risk score for all assignable beneficiaries",
},
{
"name": "RISK_SCORE_ESRD",
"type": "float | None",
"desc": "Average risk score for ESRD beneficiaries",
},
{
"name": "RISK_SCORE_DISABLED",
"type": "float | None",
"desc": "Average risk score for disabled beneficiaries",
},
{
"name": "RISK_SCORE_AGED_DUAL",
"type": "float | None",
"desc": "Average risk score for aged dual-eligible beneficiaries",
},
{
"name": "RISK_SCORE_AGED_NON_DUAL",
"type": "float | None",
"desc": "Average risk score for aged non-dual beneficiaries",
},
{
"name": "BENE_COUNT_TOTAL",
"type": "int | None",
"desc": "Count of all assignable beneficiaries",
},
{
"name": "BENE_COUNT_ESRD",
"type": "int | None",
"desc": "Count of ESRD beneficiaries",
},
{
"name": "BENE_COUNT_DISABLED",
"type": "int | None",
"desc": "Count of disabled beneficiaries",
},
{
"name": "BENE_COUNT_AGED_DUAL",
"type": "int | None",
"desc": "Count of aged dual-eligible beneficiaries",
},
{
"name": "BENE_COUNT_AGED_NON_DUAL",
"type": "int | None",
"desc": "Count of aged non-dual beneficiaries",
},
]
return fields
def normalize_field_name(name: str) -> str:
"""Convert field names to Python snake_case."""
return name.lower()
def generate_module(fields: list[dict]) -> str:
"""Generate the ssp_county.py module source code."""
lines = [
'"""SSP County-Level Aggregate Data — Per capita expenditures and risk scores.',
"",
"Auto-generated from Medicare Shared Savings Program County-Level",
"Aggregate Expenditure and Risk Score Data Dictionary.",
"",
f"{len(fields)} fields covering per capita expenditures, risk scores,",
"and beneficiary counts by enrollment type (ESRD, disabled, aged dual/non-dual)",
"at the state-county level.",
"",
"This data supports historical benchmark calculations and trending",
"analysis for Medicare Shared Savings Program ACOs.",
"",
"Source: County_Lvl_FFS_Data_SSP_Benchmark_PUF_Data_Dictionary.pdf",
'"""',
"",
"from __future__ import annotations",
"",
"from aco.table.base import SQLTable",
"",
]
lines.extend(
[
"",
"class SspCountyData(SQLTable):",
' """Medicare Shared Savings Program County-Level Aggregate Data.',
"",
" Per capita expenditures and risk scores for assignable beneficiaries",
" aggregated by county and enrollment type.",
"",
f" {len(fields)} fields: demographics, expenditures, risk scores, counts.",
"",
" Enrollment types:",
" - ESRD: End-Stage Renal Disease",
" - Disabled: Under 65 with disability",
" - Aged Dual: 65+ with dual Medicare-Medicaid eligibility",
" - Aged Non-Dual: 65+ without dual eligibility",
"",
" Source: CMS County_Lvl_FFS_Data_SSP_Benchmark_PUF_Data_Dictionary.pdf",
' """',
"",
' __schema__ = "ssp"',
' __tablename__ = "county_data"',
]
)
for field in fields:
py_name = normalize_field_name(field["name"])
py_type = field["type"]
desc = field["desc"]
# Escape docstring issues
desc = desc.replace('"""', '\'""')
if desc.startswith('"'):
desc = " " + desc
lines.append("")
lines.append(f" {py_name}: {py_type}")
lines.append(f' """{desc}"""')
lines.append("")
return "\n".join(lines)
def main():
"""Generate src/aco/table/ssp_county.py from SSP data dictionary PDF."""
output_path = (
Path(__file__).parent.parent / "src" / "aco" / "table" / "ssp_county.py"
)
fields = extract_county_data_fields()
code = generate_module(fields)
output_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(code)
print(f"Generated {output_path}")
print(f" 1 table, {len(fields)} fields")
return 0
if __name__ == "__main__":
raise SystemExit(main())

202
dev/setup_unity_catalog.py Normal file
View File

@@ -0,0 +1,202 @@
"""Setup Unity Catalog structure from aco.table schemas.
This script creates the catalog, schemas, and table definitions in
Databricks Unity Catalog to match the canonical aco.table structure.
Usage::
# Dry run to see what would be created
uv run python dev/setup_unity_catalog.py --dry-run
# Actually create the catalog structure
uv run python dev/setup_unity_catalog.py --catalog aco --token <pat>
# Create with custom workspace
uv run python dev/setup_unity_catalog.py \
--workspace-id 7474655000864661 \
--catalog aco \
--token <pat>
"""
from __future__ import annotations
import argparse
import os
import sys
from aco.lake.catalog import Catalog
from aco.lake.unity import UnityClient, setup_catalog_from_schemas
def main():
"""Setup Unity Catalog from command line arguments."""
parser = argparse.ArgumentParser(
description="Setup Unity Catalog structure from aco.table schemas"
)
parser.add_argument(
"--workspace-id",
default="7474655000864661",
help="Databricks workspace ID (default: 7474655000864661)",
)
parser.add_argument(
"--catalog",
default="aco",
help="Target catalog name (default: aco)",
)
parser.add_argument(
"--token",
help="Databricks Personal Access Token (or set DATABRICKS_TOKEN env var)",
)
parser.add_argument(
"--dry-run",
action="store_true",
help="Show what would be created without making changes",
)
parser.add_argument(
"--skip-existing",
action="store_true",
default=True,
help="Skip schemas/tables that already exist (default: True)",
)
parser.add_argument(
"--show-schemas",
action="store_true",
help="Show discovered schemas from aco.table and exit",
)
args = parser.parse_args()
# Get token from args or environment
token = args.token or os.getenv("DATABRICKS_TOKEN")
if not token and not args.show_schemas:
print("Error: --token required or set DATABRICKS_TOKEN environment variable")
sys.exit(1)
# Show schemas mode
if args.show_schemas:
print("Discovering schemas from aco.table...")
print("=" * 70)
catalog = Catalog()
schemas = catalog.schemas()
print(f"\nFound {len(schemas)} schemas:")
for schema_name in schemas:
tables = catalog.tables(schema_name)
print(f"\n {schema_name} ({len(tables)} tables)")
for table_ref in tables[:3]:
_, table = table_ref.split(".", 1)
print(f" - {table}")
if len(tables) > 3:
print(f" ... and {len(tables) - 3} more tables")
print("\n" + "=" * 70)
print(f"\nTo create these in Unity Catalog, run without --show-schemas flag")
return
# Create Unity client
print(f"Connecting to Unity Catalog...")
print(f" Workspace ID: {args.workspace_id}")
print(f" Catalog: {args.catalog}")
print()
client = UnityClient(
workspace_id=args.workspace_id,
token=token,
)
# Test connection
try:
catalogs = client.list_catalogs()
print(f"✓ Connected to Unity Catalog")
print(f" Found {len(catalogs)} existing catalogs:")
for cat in catalogs:
print(f" - {cat.name}")
print()
except Exception as e:
print(f"✗ Failed to connect to Unity Catalog: {e}")
sys.exit(1)
# Setup catalog from schemas
if args.dry_run:
print("=" * 70)
print("DRY RUN MODE - No changes will be made")
print("=" * 70)
print()
print(f"Setting up catalog structure...")
print()
report = setup_catalog_from_schemas(
client=client,
catalog_name=args.catalog,
dry_run=args.dry_run,
skip_existing=args.skip_existing,
)
# Print summary
print()
print("=" * 70)
print("Setup Summary")
print("=" * 70)
if report["catalogs_created"]:
print(f"\nCatalogs created: {len(report['catalogs_created'])}")
for cat in report["catalogs_created"]:
print(f" ✓ {cat}")
if report["schemas_created"]:
print(f"\nSchemas created: {len(report['schemas_created'])}")
for schema in report["schemas_created"]:
print(f" ✓ {schema}")
if report["tables_created"]:
print(f"\nTables created: {len(report['tables_created'])}")
for table in report["tables_created"]:
print(f" ✓ {table}")
if report["errors"]:
print(f"\nErrors encountered: {len(report['errors'])}")
for error in report["errors"]:
print(f" ✗ {error}")
if args.dry_run:
print("\n" + "=" * 70)
print("DRY RUN COMPLETE - Run without --dry-run to apply changes")
print("=" * 70)
else:
print("\n" + "=" * 70)
print("✓ Setup complete")
print("=" * 70)
# Show how to use with IcebergContext
print("\nNext steps:")
print()
print(" 1. Use with IcebergContext:")
print()
print(" from aco.lake import IcebergContext")
print()
print(" ctx = IcebergContext(")
print(
f' catalog_uri="https://dbc-{args.workspace_id}.cloud.databricks.com/api/2.1/unity-catalog/iceberg",'
)
print(f' warehouse="{args.catalog}",')
print(' properties={"token": "<your-token>"},')
print(" )")
print()
print(" 2. Run pipelines:")
print()
print(" from aco.lake.engine import execute")
print(" from aco.pipe import readmissions")
print()
print(" results = execute(readmissions.pipeline, ctx)")
if __name__ == "__main__":
main()

View File

View File

@@ -0,0 +1,269 @@
"""Example usage of Unity Catalog integration.
This demonstrates the full workflow:
1. Connect to Unity Catalog
2. Create catalogs, schemas, tables
3. Use with IcebergContext for data access
4. Run pipelines against Unity Catalog
Prerequisites:
- Databricks workspace with Unity Catalog enabled
- Personal Access Token (PAT) with catalog permissions
- Set DATABRICKS_TOKEN environment variable
Usage::
export DATABRICKS_TOKEN="<your-pat>"
uv run python dev/unity_catalog_example.py
"""
from __future__ import annotations
import os
from aco.lake import IcebergContext, UnityClient
from aco.lake.catalog import Catalog
from aco.lake.engine import execute
def example_unity_client_basics():
"""Example 1: Unity Client basics - catalogs, schemas, tables."""
print("=" * 70)
print("Example 1: Unity Client Basics")
print("=" * 70)
print()
# Connect to Unity Catalog
token = os.getenv("DATABRICKS_TOKEN")
if not token:
print("⚠ Set DATABRICKS_TOKEN environment variable to run this example")
return
client = UnityClient(
workspace_id="7474655000864661",
token=token,
)
# List existing catalogs
print("1. Listing existing catalogs")
print("-" * 70)
catalogs = client.list_catalogs()
print(f"Found {len(catalogs)} catalogs:")
for cat in catalogs:
print(f" - {cat.name}: {cat.comment or '(no description)'}")
print()
# Create a new catalog
print("2. Creating new catalog")
print("-" * 70)
try:
catalog = client.create_catalog(
name="aco_dev",
comment="ACO analytics development catalog",
)
print(f"✓ Created catalog: {catalog.name}")
print(f" Owner: {catalog.owner}")
print(f" Created: {catalog.created_at}")
except Exception as e:
print(f"Catalog may already exist: {e}")
print()
# Create a schema
print("3. Creating schema")
print("-" * 70)
try:
schema = client.create_schema(
catalog_name="aco_dev",
schema_name="core",
comment="Core healthcare data tables",
)
print(f"✓ Created schema: {schema.full_name}")
except Exception as e:
print(f"Schema may already exist: {e}")
print()
# List schemas in catalog
print("4. Listing schemas")
print("-" * 70)
schemas = client.list_schemas("aco_dev")
print(f"Found {len(schemas)} schemas in aco_dev:")
for schema in schemas:
print(f" - {schema.name}: {schema.comment or '(no description)'}")
print()
def example_schema_mapping():
"""Example 2: Schema mapping for enterprise tables."""
print("=" * 70)
print("Example 2: Schema Mapping (Enterprise Tables)")
print("=" * 70)
print()
# Create catalog with schema mapping
print("1. Setting up catalog with enterprise schema mapping")
print("-" * 70)
catalog = Catalog(
schema_map={
# Map Unity Catalog names to canonical aco.table names
"aco.encounters": "core.encounter",
"aco.patients": "core.patient",
"aco.claims": "core.medical_claim",
},
column_map={
"aco.encounters": {
# Map Unity Catalog column names to canonical names
"encntr_id": "encounter_id",
"mbr_id": "person_id",
"encntr_type": "encounter_type",
},
"aco.patients": {
"mbr_id": "patient_id",
"birth_dt": "birth_date",
"death_dt": "death_date",
},
},
)
# Test mapping
print("✓ Catalog configured with schema mappings")
print()
print("2. Testing table reference mapping")
print("-" * 70)
test_refs = [
"aco.encounters",
"aco.patients",
"core.condition", # Unmapped, passes through
]
for ref in test_refs:
physical = catalog.physical_table_ref(ref)
print(f" {ref:25s} → {physical}")
print()
print("3. Testing column name mapping")
print("-" * 70)
test_cols = [
("aco.encounters", "encntr_id"),
("aco.encounters", "mbr_id"),
("aco.encounters", "unmapped_col"),
("aco.patients", "birth_dt"),
]
for table_ref, col in test_cols:
physical_col = catalog.physical_column_name(table_ref, col)
print(f" {table_ref}.{col:20s} → {physical_col}")
print()
def example_iceberg_context():
"""Example 3: Using Unity Catalog with IcebergContext."""
print("=" * 70)
print("Example 3: IcebergContext with Unity Catalog")
print("=" * 70)
print()
token = os.getenv("DATABRICKS_TOKEN")
if not token:
print("⚠ Set DATABRICKS_TOKEN environment variable to run this example")
return
# Create IcebergContext pointing to Unity Catalog
print("1. Creating IcebergContext for Unity Catalog")
print("-" * 70)
workspace_id = "7474655000864661"
ctx = IcebergContext(
catalog_uri=f"https://dbc-{workspace_id}.cloud.databricks.com/api/2.1/unity-catalog/iceberg",
warehouse="aco_dev",
properties={
"token": token,
# Unity Catalog specific properties
"header.X-Databricks-Cluster-Id": "auto",
},
)
print(f"✓ IcebergContext configured")
print(f" Catalog URI: {ctx.catalog_uri}")
print(f" Warehouse: {ctx.warehouse}")
print()
print("2. Using context to load tables")
print("-" * 70)
print(" (Would load tables from Unity Catalog if they exist)")
print(" Example: df = ctx.load('core.encounter')")
print()
def example_pipeline_execution():
"""Example 4: Running pipelines against Unity Catalog."""
print("=" * 70)
print("Example 4: Pipeline Execution")
print("=" * 70)
print()
token = os.getenv("DATABRICKS_TOKEN")
if not token:
print("⚠ Set DATABRICKS_TOKEN environment variable to run this example")
return
workspace_id = "7474655000864661"
# Create context with schema mapping
print("1. Setting up IcebergContext with schema mapping")
print("-" * 70)
catalog = Catalog(
schema_map={
"aco_prod.encounters": "core.encounter",
"aco_prod.patients": "core.patient",
}
)
ctx = IcebergContext(
catalog_uri=f"https://dbc-{workspace_id}.cloud.databricks.com/api/2.1/unity-catalog/iceberg",
warehouse="aco_dev",
catalog=catalog,
properties={"token": token},
)
print("✓ Context configured with schema mapping")
print()
print("2. Executing pipeline (example - would run if tables exist)")
print("-" * 70)
print(" from aco.pipe import readmissions")
print(" results = execute(readmissions.pipeline, ctx)")
print()
print(" This would:")
print(" - Read input tables from Unity Catalog")
print(" - Execute narwhals transforms locally")
print(" - Optionally write results back to Unity Catalog")
print()
def main():
"""Run all examples."""
examples = [
example_unity_client_basics,
example_schema_mapping,
example_iceberg_context,
example_pipeline_execution,
]
for i, example_fn in enumerate(examples, 1):
try:
example_fn()
except Exception as e:
print(f"✗ Example {i} failed: {e}")
print()
if i < len(examples):
print("\n" + "=" * 70)
print()
if __name__ == "__main__":
main()

View File

@@ -39,6 +39,7 @@ RUN uv python install ${PYTHON_VERSION} \
&& cd workspace \ && cd workspace \
&& uv add "marimo[recommended]" polars cudf-polars-cu12 pandas numpy pyarrow \ && uv add "marimo[recommended]" polars cudf-polars-cu12 pandas numpy pyarrow \
"pyiceberg[s3,pyarrow]>=0.7.0" "duckdb>=1.0.0" "narwhals>=1.0.0" "trino>=0.328.0" \ "pyiceberg[s3,pyarrow]>=0.7.0" "duckdb>=1.0.0" "narwhals>=1.0.0" "trino>=0.328.0" \
"sqlglot>=26.0.0" \
vega_datasets pyzotero \ vega_datasets pyzotero \
--extra-index-url https://pypi.nvidia.com --extra-index-url https://pypi.nvidia.com

330
notebooks/sql_generator.py Normal file
View File

@@ -0,0 +1,330 @@
import marimo
__generated_with = "0.19.8"
app = marimo.App(width="full")
@app.cell(hide_code=True)
def _():
import marimo as mo
return (mo,)
@app.cell(hide_code=True)
def _(mo):
mo.md("""
# SQL Transpilation Generator
Generates **Databricks-dialect SQL** from every pipeline step by running
narwhals express functions against DuckDB relations and extracting the SQL
via `.sql_query()`, then transpiling with sqlglot.
Each cell below transpiles one pipeline module and displays the generated
`CREATE OR REPLACE TABLE` statements.
""")
return
@app.cell(hide_code=True)
def _():
import duckdb
from aco.lake.transpile import transpile
DB_PATH = "/home/kert/notebooks/aco.duckdb"
con = duckdb.connect(DB_PATH, read_only=True)
DIALECT = "databricks"
CATALOG = "homelab"
def run_transpile(pipeline, label):
"""Transpile a pipeline and return formatted SQL."""
result = transpile(
pipeline,
con,
target_dialect=DIALECT,
catalog=CATALOG,
output_mode="ctas",
)
parts = []
errors = 0
for name, sql in result.items():
if sql.startswith("-- ERROR"):
errors += 1
parts.append(f"-- {name}\n{sql}")
header = f"/* {label}: {len(result)} steps, {errors} errors */\n"
return header + "\n\n".join(parts), len(result), errors
return con, run_transpile
@app.cell(hide_code=True)
def _(mo):
mo.md("""
## core
""")
return
@app.cell
def _(mo, run_transpile):
from aco.pipe import core as _core
_sql, _n, _e = run_transpile(_core.pipeline, "core")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
return
@app.cell(hide_code=True)
def _(mo):
mo.md("""
## input_layer
""")
return
@app.cell
def _(mo, run_transpile):
from aco.pipe import input_layer as _input_layer
_sql, _n, _e = run_transpile(_input_layer.pipeline, "input_layer")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
return
@app.cell(hide_code=True)
def _(mo):
mo.md("""
## data_quality
""")
return
@app.cell
def _(mo, run_transpile):
from aco.pipe import data_quality as _data_quality
_sql, _n, _e = run_transpile(_data_quality.pipeline, "data_quality")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
return
@app.cell(hide_code=True)
def _(mo):
mo.md("""
## claims_preprocessing
""")
return
@app.cell
def _(mo, run_transpile):
from aco.pipe import claims_preprocessing as _claims_preprocessing
_sql, _n, _e = run_transpile(_claims_preprocessing.pipeline, "claims_preprocessing")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
return
@app.cell(hide_code=True)
def _(mo):
mo.md("""
## readmissions
""")
return
@app.cell
def _(mo, run_transpile):
from aco.pipe import readmissions as _readmissions
_sql, _n, _e = run_transpile(_readmissions.pipeline, "readmissions")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
return
@app.cell(hide_code=True)
def _(mo):
mo.md("""
## pharmacy
""")
return
@app.cell
def _(mo, run_transpile):
from aco.pipe import pharmacy as _pharmacy
_sql, _n, _e = run_transpile(_pharmacy.pipeline, "pharmacy")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
return
@app.cell(hide_code=True)
def _(mo):
mo.md("""
## ahrq_measures
""")
return
@app.cell
def _(mo, run_transpile):
from aco.pipe import ahrq_measures as _ahrq_measures
_sql, _n, _e = run_transpile(_ahrq_measures.pipeline, "ahrq_measures")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
return
@app.cell(hide_code=True)
def _(mo):
mo.md("""
## quality_measures
""")
return
@app.cell
def _(mo, run_transpile):
from aco.pipe import quality_measures as _quality_measures
_sql, _n, _e = run_transpile(_quality_measures.pipeline, "quality_measures")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
return
@app.cell(hide_code=True)
def _(mo):
mo.md("""
## hcc_suspecting
""")
return
@app.cell
def _(mo, run_transpile):
from aco.pipe import hcc_suspecting as _hcc_suspecting
_sql, _n, _e = run_transpile(_hcc_suspecting.pipeline, "hcc_suspecting")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
return
@app.cell(hide_code=True)
def _(mo):
mo.md("""
## provider_attribution
""")
return
@app.cell
def _(mo, run_transpile):
from aco.pipe import provider_attribution as _provider_attribution
_sql, _n, _e = run_transpile(_provider_attribution.pipeline, "provider_attribution")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
return
@app.cell(hide_code=True)
def _(mo):
mo.md("""
## cclf
""")
return
@app.cell
def _(mo, run_transpile):
from aco.pipe import cclf as _cclf
try:
_sql, _n, _e = run_transpile(_cclf.pipeline, "cclf")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
except Exception as _ex:
mo.md(f"**CCLF pipeline requires CCLF data in DuckDB.**\n\nError: `{_ex}`")
return
@app.cell(hide_code=True)
def _(mo):
mo.md("""
## main
""")
return
@app.cell
def _(mo, run_transpile):
from aco.pipe import main as _main
_sql, _n, _e = run_transpile(_main.pipeline, "main")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
return
@app.cell(hide_code=True)
def _(mo):
mo.md("""
---
## Summary
""")
return
@app.cell
def _(con, mo):
from aco.lake.transpile import transpile as _transpile
from aco.pipe import (
ahrq_measures,
claims_preprocessing,
core,
data_quality,
hcc_suspecting,
input_layer,
main,
pharmacy,
provider_attribution,
quality_measures,
readmissions,
)
_modules = [
("core", core),
("input_layer", input_layer),
("data_quality", data_quality),
("claims_preprocessing", claims_preprocessing),
("readmissions", readmissions),
("pharmacy", pharmacy),
("ahrq_measures", ahrq_measures),
("quality_measures", quality_measures),
("hcc_suspecting", hcc_suspecting),
("provider_attribution", provider_attribution),
("main", main),
]
_rows = []
_total_steps = 0
_total_errors = 0
for _name, _mod in _modules:
_result = _transpile(
_mod.pipeline, con, target_dialect="databricks", catalog="homelab"
)
_errs = sum(1 for v in _result.values() if v.startswith("-- ERROR"))
_total_steps += len(_result)
_total_errors += _errs
_rows.append(f"| {_name} | {len(_result)} | {_errs} |")
_table = "| Pipeline | Steps | Errors |\n|----------|-------|--------|\n"
_table += "\n".join(_rows)
_table += f"\n| **Total** | **{_total_steps}** | **{_total_errors}** |"
mo.md(_table)
return
if __name__ == "__main__":
app.run()

394
pfs.html Normal file

File diff suppressed because one or more lines are too long

View File

@@ -5,7 +5,11 @@ description = "Add your description here"
readme = "README.md" readme = "README.md"
requires-python = ">=3.12" requires-python = ">=3.12"
dependencies = [ dependencies = [
"databricks-cli>=0.18.0",
"databricks-sdk>=0.85.0",
"httpx>=0.28.1",
"pyarrow>=23.0.0", "pyarrow>=23.0.0",
"sqlglot>=26.0.0",
] ]
[dependency-groups] [dependency-groups]

Binary file not shown.

622
src/aco/dag.py Normal file
View File

@@ -0,0 +1,622 @@
"""DAG visualization derived from Pipeline introspection.
Builds a dependency graph from ``Expr.inputs`` (parameter-derived ground
truth) and renders it as interactive HTML, static DOT, or Mermaid.
No modifications to express/ or pipe/ required — pure read-only
introspection of existing Expr objects.
Usage::
from aco.dag import build_graph, to_html, to_dot, to_mermaid
from aco.pipe import readmissions
g = build_graph(readmissions.pipeline)
html = to_html(g)
CLI::
uv run python -m aco.dag
uv run python -m aco.dag -p readmissions --html dag.html
uv run python -m aco.dag --dot dag.dot
uv run python -m aco.dag -p pharmacy --mermaid
"""
from __future__ import annotations
import html as html_mod
import json
from dataclasses import dataclass, field
from textwrap import dedent
from aco.pipe.base import Pipeline
# Tableau 10 palette for pipeline schemas, gray for externals.
_SCHEMA_COLORS: dict[str, str] = {
"core": "#e15759",
"input_layer": "#76b7b2",
"cclf": "#4e79a7",
"claims_preprocessing": "#f28e2b",
"readmissions": "#59a14f",
"ahrq_measures": "#edc948",
"pharmacy": "#b07aa1",
"quality_measures": "#ff9da7",
"hcc_suspecting": "#9c755f",
"provider_attribution": "#bab0ac",
"main": "#86bcb6",
"data_quality": "#4e79a7",
}
_EXTERNAL_COLOR = "#888888"
def _color(schema: str) -> str:
return _SCHEMA_COLORS.get(schema, _EXTERNAL_COLOR)
# ---------------------------------------------------------------------------
# Data model
# ---------------------------------------------------------------------------
@dataclass
class Node:
name: str
schema: str
is_external: bool
description: str = ""
@dataclass
class Edge:
source: str
target: str
@dataclass
class Graph:
nodes: list[Node] = field(default_factory=list)
edges: list[Edge] = field(default_factory=list)
def schemas(self) -> set[str]:
return {n.schema for n in self.nodes}
def filter(self, schemas: set[str] | None = None) -> Graph:
if schemas is None:
return self
keep = {n.name for n in self.nodes if n.schema in schemas}
nodes = [n for n in self.nodes if n.name in keep]
edges = [e for e in self.edges if e.source in keep and e.target in keep]
return Graph(nodes=nodes, edges=edges)
# ---------------------------------------------------------------------------
# Builder
# ---------------------------------------------------------------------------
def build_graph(*pipelines: Pipeline) -> Graph:
"""Derive a dependency graph from one or more Pipelines."""
nodes: dict[str, Node] = {}
edges: list[Edge] = []
produced: set[str] = set()
for pipeline in pipelines:
for expr in pipeline.exprs:
produced.add(expr.name)
schema = expr.name.split(".")[0]
nodes[expr.name] = Node(
name=expr.name,
schema=schema,
is_external=False,
description=expr.doc,
)
for pipeline in pipelines:
for expr in pipeline.exprs:
for inp in expr.inputs:
edges.append(Edge(source=inp, target=expr.name))
if inp not in nodes:
schema = inp.split(".")[0] if "." in inp else inp
nodes[inp] = Node(
name=inp,
schema=schema,
is_external=True,
)
return Graph(nodes=list(nodes.values()), edges=edges)
# ---------------------------------------------------------------------------
# DOT renderer
# ---------------------------------------------------------------------------
def to_dot(graph: Graph, *, title: str = "ACO Pipeline DAG") -> str:
"""Render graph as DOT language string."""
lines = [
f'digraph "{title}" {{',
" rankdir=LR;",
' node [fontname="Helvetica" fontsize=10];',
' edge [color="#555555"];',
"",
]
by_schema: dict[str, list[Node]] = {}
for node in graph.nodes:
by_schema.setdefault(node.schema, []).append(node)
for schema in sorted(by_schema):
color = _color(schema)
lines.append(f" subgraph cluster_{schema} {{")
lines.append(f' label="{schema}";')
lines.append(f' color="{color}";')
lines.append(" style=rounded;")
for node in by_schema[schema]:
short = node.name.split(".", 1)[1] if "." in node.name else node.name
nid = node.name.replace(".", "__")
if node.is_external:
lines.append(
f' {nid} [label="{short}" '
f'style="dashed,filled" fillcolor="#2a2a2a" '
f'fontcolor="#aaaaaa" shape=rectangle];'
)
else:
lines.append(
f' {nid} [label="{short}" '
f'style="filled,rounded" fillcolor="{color}20" '
f'fontcolor="{color}" shape=rectangle];'
)
lines.append(" }")
lines.append("")
for edge in graph.edges:
src = edge.source.replace(".", "__")
tgt = edge.target.replace(".", "__")
lines.append(f" {src} -> {tgt};")
lines.append("}")
return "\n".join(lines)
# ---------------------------------------------------------------------------
# Mermaid renderer
# ---------------------------------------------------------------------------
def to_mermaid(graph: Graph) -> str:
"""Render graph as Mermaid flowchart syntax."""
lines = ["graph LR"]
for node in graph.nodes:
nid = node.name.replace(".", "_").replace("-", "_")
short = node.name.split(".", 1)[1] if "." in node.name else node.name
if node.is_external:
lines.append(f" {nid}[/{short}/]")
else:
lines.append(f" {nid}[{short}]")
for edge in graph.edges:
src = edge.source.replace(".", "_").replace("-", "_")
tgt = edge.target.replace(".", "_").replace("-", "_")
lines.append(f" {src} --> {tgt}")
return "\n".join(lines)
# ---------------------------------------------------------------------------
# Interactive HTML renderer (Cytoscape.js + Dagre from CDN)
# ---------------------------------------------------------------------------
def to_html(graph: Graph, *, title: str = "ACO Pipeline DAG") -> str:
"""Render graph as self-contained interactive HTML."""
cy_nodes = []
cy_edges = []
# Parent (compound) nodes for each schema
for schema in sorted(graph.schemas()):
cy_nodes.append(
{
"data": {
"id": f"grp_{schema}",
"label": schema,
"is_group": True,
},
"classes": "group",
}
)
for node in graph.nodes:
classes = "external" if node.is_external else "expr"
cy_nodes.append(
{
"data": {
"id": node.name,
"label": (
node.name.split(".", 1)[1] if "." in node.name else node.name
),
"parent": f"grp_{node.schema}",
"description": node.description,
"schema": node.schema,
"is_external": node.is_external,
},
"classes": classes,
}
)
for edge in graph.edges:
cy_edges.append(
{
"data": {
"source": edge.source,
"target": edge.target,
}
}
)
elements_json = json.dumps({"nodes": cy_nodes, "edges": cy_edges}, indent=2)
schema_colors_json = json.dumps({s: _color(s) for s in graph.schemas()})
esc_title = html_mod.escape(title)
return dedent(f"""\
<!DOCTYPE html>
<html>
<head>
<meta charset="utf-8">
<title>{esc_title}</title>
<script src="https://unpkg.com/cytoscape@3/dist/cytoscape.min.js"></script>
<script src="https://unpkg.com/dagre@0.8/dist/dagre.min.js"></script>
<script src="https://unpkg.com/cytoscape-dagre@2/cytoscape-dagre.js"></script>
<style>
* {{ margin: 0; padding: 0; box-sizing: border-box; }}
body {{ background: #1a1a2e; color: #e0e0e0; font-family: system-ui, sans-serif; }}
#cy {{ width: 100vw; height: 100vh; }}
#controls {{
position: fixed; top: 12px; left: 12px; z-index: 10;
background: #16213e; border: 1px solid #333; border-radius: 8px;
padding: 12px; max-width: 260px; font-size: 13px;
}}
#controls h3 {{ margin-bottom: 8px; font-size: 14px; color: #fff; }}
#search {{
width: 100%; padding: 6px 8px; border-radius: 4px;
border: 1px solid #444; background: #0f3460; color: #e0e0e0;
margin-bottom: 8px; font-size: 13px;
}}
#legend {{ display: flex; flex-wrap: wrap; gap: 4px; }}
.legend-item {{
display: inline-flex; align-items: center; gap: 4px;
cursor: pointer; padding: 2px 6px; border-radius: 4px;
font-size: 11px; opacity: 1; transition: opacity 0.2s;
}}
.legend-item.hidden {{ opacity: 0.3; }}
.legend-dot {{
width: 10px; height: 10px; border-radius: 50%;
display: inline-block; flex-shrink: 0;
}}
#tooltip {{
position: fixed; display: none; background: #16213e;
border: 1px solid #444; border-radius: 6px; padding: 10px;
max-width: 340px; font-size: 12px; z-index: 20;
pointer-events: none; line-height: 1.4;
}}
#tooltip .tt-name {{ font-weight: bold; color: #fff; margin-bottom: 4px; }}
#tooltip .tt-desc {{ color: #bbb; }}
#stats {{
position: fixed; bottom: 12px; left: 12px; z-index: 10;
font-size: 11px; color: #666;
}}
</style>
</head>
<body>
<div id="controls">
<h3>{esc_title}</h3>
<input id="search" type="text" placeholder="Search nodes...">
<div id="legend"></div>
</div>
<div id="tooltip">
<div class="tt-name"></div>
<div class="tt-desc"></div>
</div>
<div id="stats"></div>
<div id="cy"></div>
<script>
const elements = {elements_json};
const schemaColors = {schema_colors_json};
const externalColor = "{_EXTERNAL_COLOR}";
// Build legend
const legend = document.getElementById("legend");
const hiddenSchemas = new Set();
Object.keys(schemaColors).sort().forEach(s => {{
const item = document.createElement("span");
item.className = "legend-item";
item.innerHTML = '<span class="legend-dot" style="background:' +
schemaColors[s] + '"></span>' + s;
item.onclick = () => {{
if (hiddenSchemas.has(s)) {{ hiddenSchemas.delete(s); item.classList.remove("hidden"); }}
else {{ hiddenSchemas.add(s); item.classList.add("hidden"); }}
applyVisibility();
}};
legend.appendChild(item);
}});
// Init Cytoscape
const cy = cytoscape({{
container: document.getElementById("cy"),
elements: [...elements.nodes, ...elements.edges],
style: [
{{
selector: "node.group",
style: {{
"background-color": "transparent",
"background-opacity": 0.05,
"border-width": 1,
"border-color": "#444",
"label": "data(label)",
"text-valign": "top",
"text-halign": "center",
"font-size": "14px",
"color": "#666",
"padding": "16px",
"shape": "round-rectangle",
}}
}},
{{
selector: "node.expr",
style: {{
"background-color": function(ele) {{
return schemaColors[ele.data("schema")] || "#888";
}},
"background-opacity": 0.15,
"border-width": 2,
"border-color": function(ele) {{
return schemaColors[ele.data("schema")] || "#888";
}},
"label": "data(label)",
"text-valign": "center",
"text-halign": "center",
"font-size": "9px",
"color": function(ele) {{
return schemaColors[ele.data("schema")] || "#888";
}},
"width": "label",
"height": 28,
"padding": "8px",
"shape": "round-rectangle",
}}
}},
{{
selector: "node.external",
style: {{
"background-color": "#2a2a2a",
"border-width": 1,
"border-style": "dashed",
"border-color": "#666",
"label": "data(label)",
"text-valign": "center",
"text-halign": "center",
"font-size": "8px",
"color": "#888",
"width": "label",
"height": 24,
"padding": "6px",
"shape": "round-rectangle",
}}
}},
{{
selector: "edge",
style: {{
"width": 1.5,
"line-color": "#444",
"target-arrow-color": "#444",
"target-arrow-shape": "triangle",
"curve-style": "bezier",
"arrow-scale": 0.8,
}}
}},
{{
selector: ".highlighted",
style: {{
"border-color": "#fff",
"border-width": 3,
"color": "#fff",
}}
}},
{{
selector: "edge.highlighted",
style: {{
"line-color": "#e0e0e0",
"target-arrow-color": "#e0e0e0",
"width": 2.5,
}}
}},
{{
selector: ".dimmed",
style: {{
"opacity": 0.15,
}}
}},
],
layout: {{
name: "dagre",
rankDir: "LR",
nodeSep: 30,
rankSep: 80,
edgeSep: 10,
animate: false,
}},
minZoom: 0.1,
maxZoom: 3,
}});
// Stats
const nNodes = elements.nodes.filter(n => !n.data.is_group).length;
const nEdges = elements.edges.length;
document.getElementById("stats").textContent =
nNodes + " nodes, " + nEdges + " edges";
// Tooltip
const tooltip = document.getElementById("tooltip");
const ttName = tooltip.querySelector(".tt-name");
const ttDesc = tooltip.querySelector(".tt-desc");
cy.on("mouseover", "node.expr, node.external", function(e) {{
const d = e.target.data();
ttName.textContent = d.id;
ttDesc.textContent = d.description || (d.is_external ? "External source table" : "");
tooltip.style.display = "block";
}});
cy.on("mouseout", "node.expr, node.external", function() {{
tooltip.style.display = "none";
}});
cy.on("mousemove", function(e) {{
tooltip.style.left = (e.originalEvent.clientX + 14) + "px";
tooltip.style.top = (e.originalEvent.clientY + 14) + "px";
}});
// Click to highlight ancestors + descendants
cy.on("tap", "node.expr, node.external", function(e) {{
const node = e.target;
const predecessors = node.predecessors();
const successors = node.successors();
const connected = predecessors.union(successors).union(node);
cy.elements().addClass("dimmed");
connected.removeClass("dimmed").addClass("highlighted");
}});
cy.on("tap", function(e) {{
if (e.target === cy) {{
cy.elements().removeClass("dimmed highlighted");
}}
}});
// Search
const searchInput = document.getElementById("search");
searchInput.addEventListener("input", function() {{
const q = this.value.toLowerCase();
if (!q) {{
cy.elements().removeClass("dimmed highlighted");
return;
}}
const matches = cy.nodes().filter(n =>
n.data("id").toLowerCase().includes(q) ||
(n.data("label") || "").toLowerCase().includes(q)
);
if (matches.length > 0) {{
cy.elements().addClass("dimmed");
matches.forEach(m => {{
const connected = m.predecessors().union(m.successors()).union(m);
connected.removeClass("dimmed").addClass("highlighted");
}});
}}
}});
// Schema visibility toggle
function applyVisibility() {{
cy.batch(() => {{
cy.nodes().forEach(n => {{
const s = n.data("schema") || n.data("label");
if (hiddenSchemas.has(s)) n.style("display", "none");
else n.style("display", "element");
}});
}});
}}
cy.fit(undefined, 40);
</script>
</body>
</html>
""")
# ---------------------------------------------------------------------------
# CLI
# ---------------------------------------------------------------------------
_ALL_PIPE_MODULES = [
"core",
"input_layer",
"hcc_suspecting",
"data_quality",
"quality_measures",
"pharmacy",
"readmissions",
"ahrq_measures",
"provider_attribution",
"main",
"claims_preprocessing",
"cclf",
]
def _load_pipelines(names: list[str] | None = None) -> list[Pipeline]:
import importlib
targets = names or _ALL_PIPE_MODULES
pipelines = []
for name in targets:
mod = importlib.import_module(f"aco.pipe.{name}")
pipelines.append(mod.pipeline)
return pipelines
def main() -> None:
import argparse
import sys
parser = argparse.ArgumentParser(
description="DAG visualization from Pipeline introspection"
)
parser.add_argument(
"-p",
"--pipeline",
action="append",
help="Pipeline module name(s) (default: all)",
)
parser.add_argument("--html", metavar="FILE", help="Write interactive HTML")
parser.add_argument("--dot", metavar="FILE", help="Write DOT language")
parser.add_argument(
"--mermaid", action="store_true", help="Print Mermaid to stdout"
)
args = parser.parse_args()
pipelines = _load_pipelines(args.pipeline)
graph = build_graph(*pipelines)
if not (args.html or args.dot or args.mermaid):
# Summary mode
n_expr = sum(1 for n in graph.nodes if not n.is_external)
n_ext = sum(1 for n in graph.nodes if n.is_external)
print(f"Nodes: {len(graph.nodes)} ({n_expr} exprs, {n_ext} external)")
print(f"Edges: {len(graph.edges)}")
print(f"Schemas: {', '.join(sorted(graph.schemas()))}")
return
if args.html:
title = "ACO Pipeline DAG"
if args.pipeline:
title = f"{', '.join(args.pipeline)} DAG"
content = to_html(graph, title=title)
if args.html == "-":
sys.stdout.write(content)
else:
with open(args.html, "w") as f:
f.write(content)
print(f"Wrote {args.html}")
if args.dot:
content = to_dot(graph)
if args.dot == "-":
sys.stdout.write(content)
else:
with open(args.dot, "w") as f:
f.write(content)
print(f"Wrote {args.dot}")
if args.mermaid:
print(to_mermaid(graph))
if __name__ == "__main__":
main()

View File

@@ -8,6 +8,7 @@ from . import hcc_suspecting as hcc_suspecting
from . import input_layer as input_layer from . import input_layer as input_layer
from . import main as main from . import main as main
from . import pharmacy as pharmacy from . import pharmacy as pharmacy
from . import provider_attribution as provider_attribution
from . import quality_measures as quality_measures from . import quality_measures as quality_measures
from . import readmissions as readmissions from . import readmissions as readmissions
from . import provider_attribution as provider_attribution from .base import Expr as Expr

Binary file not shown.

View File

@@ -224,7 +224,10 @@ def _pqi_denom(
how="inner", how="inner",
) )
age_expr = ( age_expr = (
(nw.col("first_day_of_month") - nw.col("birth_date")).dt.total_seconds() (
nw.col("first_day_of_month").cast(nw.Datetime)
- nw.col("birth_date").cast(nw.Datetime)
).dt.total_seconds()
/ 86400 / 86400
/ 365 / 365
).cast(nw.Int64) ).cast(nw.Int64)

118
src/aco/express/base.py Normal file
View File

@@ -0,0 +1,118 @@
"""Expr — the core DSL primitive for healthcare data transformations.
An ``Expr`` binds a narwhals transformation function together with the
metadata that makes it introspectable, composable, and traceable:
- **fn** — the ``@nw.narwhalify`` function
- **name** — qualified output table reference (``schema.table``)
- **output** — ``SQLTable`` class defining the output shape contract
- **after** — explicit sequence dependencies (table ref strings)
- **refs** — ``bib.Tag`` citations linking to Zotero provenance
- **description** — human-readable prose explaining the transformation
Usage::
from aco.express.base import Expr
from aco.express import readmissions as ex
from aco.table.readmissions import ReadmissionsIntEncounter
from bib.tag import Tag
int_encounter = Expr(
name="readmissions._int_encounter",
fn=ex.int_encounter,
output=ReadmissionsIntEncounter,
after=["core.encounter"],
refs=[Tag.module("aco")],
description="Filters to acute inpatient encounters.",
)
from aco.pipe.base import Pipeline
pipeline = Pipeline(exprs=[int_encounter, ...])
results = pipeline.run(ctx.load)
"""
from __future__ import annotations
import inspect
from typing import Callable
from pydantic import BaseModel, ConfigDict, Field, field_validator
from aco.pipe.runner import _param_to_table
from aco.table.base import SQLTable
from bib.tag import Tag
class Expr(BaseModel):
"""A transformation expression — the core DSL primitive.
Binds a narwhals function with its output shape contract, citation
references, a human description, and explicit dependency declarations.
"""
model_config = ConfigDict(arbitrary_types_allowed=True)
name: str
"""Qualified output table reference (e.g. ``readmissions._int_encounter``)."""
fn: Callable
"""The ``@nw.narwhalify`` transformation function."""
output: type[SQLTable] | None = None
"""SQLTable subclass defining the expected output columns and types."""
after: list[str] = Field(default_factory=list)
"""Explicit sequence dependencies as qualified table refs.
When empty, dependencies are derived from ``fn``'s parameter
signature via ``_param_to_table()``. When populated, this is the
author's declaration of ordering intent — useful for documentation
and graph introspection. The runner always resolves inputs from
parameter names regardless of this field.
"""
refs: list[Tag] = Field(default_factory=list)
"""Citation references linking to Zotero items via ``bib.Tag``.
Each tag connects this expression to its intellectual provenance:
CMS rules, CFR regulations, journal papers, internal policies.
"""
description: str = ""
"""Human-readable prose explaining the transformation.
When empty, falls back to ``fn.__doc__``.
"""
@field_validator("name")
@classmethod
def _qualified_name(cls, v: str) -> str:
if "." not in v:
raise ValueError(f"Expr name must be qualified (schema.table): {v}")
return v
@property
def inputs(self) -> list[str]:
"""Derive input table refs from ``fn``'s parameter signature."""
sig = inspect.signature(self.fn)
return [_param_to_table(p) for p in sig.parameters]
@property
def doc(self) -> str:
"""Return description, falling back to fn docstring."""
return self.description or (self.fn.__doc__ or "").strip()
@property
def qualified_refs(self) -> list[Tag]:
"""Refs plus an auto-generated table tag for discoverability."""
auto = Tag.table(self.name)
return [auto, *self.refs]
def __repr__(self) -> str:
output_name = self.output.__name__ if self.output else "None"
return (
f"Expr(name={self.name!r}, fn={self.fn.__name__}, "
f"output={output_name}, "
f"inputs={self.inputs}, "
f"refs={len(self.refs)})"
)

View File

@@ -268,15 +268,9 @@ def encounters__orphaned_claims(
.drop("_matched") .drop("_matched")
) )
# Get max encounter_id from crosswalk # Get max encounter_id from crosswalk as a 1-row frame for cross join
max_enc = enc.select(nw.col("encounter_id").max().alias("_max")) _max_frame = enc.select(nw.col("encounter_id").max().fill_null(0).alias("_max_eid"))
_max_native = max_enc.to_native() orphans = orphans.join(_max_frame, how="cross")
try:
max_val = _max_native.item(0, 0)
except Exception:
max_val = 0
if max_val is None:
max_val = 0
return orphans.select( return orphans.select(
nw.col("claim_id"), nw.col("claim_id"),
@@ -288,7 +282,7 @@ def encounters__orphaned_claims(
) )
.rank(method="dense") .rank(method="dense")
.cast(nw.Int64) .cast(nw.Int64)
+ max_val + nw.col("_max_eid")
).alias("encounter_id"), ).alias("encounter_id"),
nw.lit("orphaned claim").alias("encounter_type"), nw.lit("orphaned claim").alias("encounter_type"),
nw.lit("other").alias("encounter_group"), nw.lit("other").alias("encounter_group"),
@@ -1142,9 +1136,12 @@ def service_category__stg_medical_claim(
) )
# CCS join: match on hcpcs_code with date-range validity # CCS join: match on hcpcs_code with date-range validity
max_yr = ccs.select(nw.col("release_year").max().alias("max_release_year")).row(0)[ _max_yr_frame = ccs.select(nw.col("release_year").max().alias("max_release_year"))
0 _max_yr_native = _max_yr_frame.to_native()
] try:
max_yr = _max_yr_native.item(0, 0)
except (AttributeError, TypeError):
max_yr = _max_yr_native.fetchone()[0]
ccs_cols = ccs.select( ccs_cols = ccs.select(
nw.col("hcpcs_code").alias("_ccs_hcpcs"), nw.col("hcpcs_code").alias("_ccs_hcpcs"),

View File

@@ -253,16 +253,15 @@ def int_current_steps(
prov = provider_attribution___stg_terminology__provider prov = provider_attribution___stg_terminology__provider
# Determine as_of_date # Determine as_of_date
max_dt = pcc.select(nw.col("claim_end_date").max().alias("max_dt")).to_native()
import datetime as _dt import datetime as _dt
today = _dt.date.today() today = _dt.date.today()
# Extract the scalar # Extract the scalar (backend-agnostic)
_native = max_dt _native = pcc.select(nw.col("claim_end_date").max().alias("max_dt")).to_native()
try: try:
_val = _native.item(0, 0) _val = _native.item(0, 0)
except Exception: except (AttributeError, TypeError):
_val = None _val = _native.fetchone()[0]
if _val is not None and _val <= today: if _val is not None and _val <= today:
as_of = _val as_of = _val
else: else:
@@ -580,15 +579,15 @@ def int_yearly_steps(
# For each performance year, run 5-step waterfall # For each performance year, run 5-step waterfall
years = py_.select("performance_year").unique() years = py_.select("performance_year").unique()
# Materialize years list # Materialize years list (backend-agnostic)
_yrs_native = years.to_native() _yrs_native = years.to_native()
try: try:
yr_list = _yrs_native["performance_year"].to_list() yr_list = _yrs_native["performance_year"].to_list()
except Exception: except Exception:
yr_list = [ try:
r[0] yr_list = [r[0] for r in _yrs_native.fetchall()]
for r in _yrs_native.iter_rows() # type: ignore[union-attr] except Exception:
] yr_list = [r[0] for r in _yrs_native.iter_rows()] # type: ignore[union-attr]
all_results = [] all_results = []

View File

@@ -196,7 +196,11 @@ def int_encounter_overlap(readmissions___int_encounter: FrameT) -> FrameT:
# Compute derived columns # Compute derived columns
days_diff = ( days_diff = (
(nw.col("discharge_date") - nw.col("admit_date")).dt.total_seconds() / 86400 (
nw.col("discharge_date").cast(nw.Datetime)
- nw.col("admit_date").cast(nw.Datetime)
).dt.total_seconds()
/ 86400
).cast(nw.Int64) ).cast(nw.Int64)
enhanced = enc.with_columns( enhanced = enc.with_columns(
nw.when(days_diff >= 1) nw.when(days_diff >= 1)
@@ -783,7 +787,11 @@ def int_readmission_crude(
# Compute readmission flags # Compute readmission flags
days_to = ( days_to = (
(nw.col("_next_admit") - nw.col("discharge_date")).dt.total_seconds() / 86400 (
nw.col("_next_admit").cast(nw.Datetime)
- nw.col("discharge_date").cast(nw.Datetime)
).dt.total_seconds()
/ 86400
).cast(nw.Int64) ).cast(nw.Int64)
result = result.with_columns( result = result.with_columns(
nw.when(~nw.col("_next_enc").is_null()) nw.when(~nw.col("_next_enc").is_null())

View File

@@ -0,0 +1,495 @@
# Unity Catalog Integration
Databricks Unity Catalog integration for the ACO data platform.
## Overview
The `unity` module provides:
1. **Pydantic Models** - Type-safe representations of Unity Catalog objects
2. **API Client** - REST API wrapper for catalog management
3. **Setup Automation** - Scripts to create catalog structure from `aco.table` schemas
4. **IcebergContext Integration** - Seamless data access via Iceberg REST Catalog spec
Unity Catalog implements the **Iceberg REST Catalog specification**, which means:
- Same API as Nessie and Polaris
- Use `IcebergContext` for data access
- Schema mapping works identically across environments
## Quick Start
### 1. Get Databricks Credentials
```bash
# Set your Databricks Personal Access Token
export DATABRICKS_TOKEN="dapi..."
```
### 2. Discover Schemas
See what would be created from `aco.table` definitions:
```bash
uv run python dev/setup_unity_catalog.py --show-schemas
```
Output:
```
Found 27 schemas:
core (37 tables)
readmissions (12 tables)
pharmacy (4 tables)
...
```
### 3. Setup Unity Catalog (Dry Run)
Preview changes without modifying the catalog:
```bash
uv run python dev/setup_unity_catalog.py --dry-run
```
### 4. Create Catalog Structure
Actually create the catalog, schemas, and tables:
```bash
uv run python dev/setup_unity_catalog.py \
--catalog aco \
--workspace-id 7474655000864661
```
### 5. Use with IcebergContext
```python
from aco.lake import IcebergContext
from aco.lake.engine import execute
from aco.pipe import readmissions
# Connect to Unity Catalog
ctx = IcebergContext(
catalog_uri="https://dbc-7474655000864661.cloud.databricks.com/api/2.1/unity-catalog/iceberg",
warehouse="aco",
properties={"token": "<your-token>"},
)
# Run pipeline
results = execute(readmissions.pipeline, ctx)
```
## API Reference
### UnityClient
Low-level REST API client for Unity Catalog management.
```python
from aco.lake import UnityClient
client = UnityClient(
workspace_id="7474655000864661",
token="dapi...",
)
# Catalogs
catalogs = client.list_catalogs()
catalog = client.create_catalog(name="aco", comment="Analytics catalog")
catalog = client.get_catalog("aco")
client.delete_catalog("aco", force=True)
# Schemas
schemas = client.list_schemas("aco")
schema = client.create_schema(
catalog_name="aco",
schema_name="core",
comment="Core tables",
)
client.delete_schema("aco", "core", force=True)
# Tables
tables = client.list_tables("aco", "core")
table = client.get_table("aco", "core", "encounter")
client.delete_table("aco", "core", "encounter")
# Volumes (for file storage)
volumes = client.list_volumes("aco", "core")
volume = client.create_volume(
catalog_name="aco",
schema_name="core",
volume_name="raw_files",
volume_type="MANAGED",
)
```
### Pydantic Models
Type-safe models for Unity Catalog objects:
```python
from aco.lake import (
UnityCatalog,
UnitySchema,
UnityTable,
UnityTableColumn,
UnityVolume,
)
# All fields are validated by Pydantic
catalog = UnityCatalog(
name="aco",
comment="ACO analytics lakehouse",
storage_root="s3://bucket/aco/",
)
schema = UnitySchema(
name="core",
catalog_name="aco",
comment="Core healthcare tables",
)
table = UnityTable(
name="encounter",
catalog_name="aco",
schema_name="core",
table_type="MANAGED",
data_source_format="DELTA", # or "ICEBERG"
columns=[
UnityTableColumn(
name="encounter_id",
type_text="STRING",
type_name="STRING",
position=0,
),
UnityTableColumn(
name="person_id",
type_text="STRING",
type_name="STRING",
position=1,
),
],
)
```
### Setup Automation
Declarative catalog setup from `aco.table` schemas:
```python
from aco.lake import UnityClient, setup_catalog_from_schemas
client = UnityClient(workspace_id="...", token="...")
# Dry run - see what would be created
report = setup_catalog_from_schemas(
client=client,
catalog_name="aco",
dry_run=True,
)
# Actually create
report = setup_catalog_from_schemas(
client=client,
catalog_name="aco",
dry_run=False,
skip_existing=True,
)
print(f"Created {len(report['schemas_created'])} schemas")
print(f"Errors: {report['errors']}")
```
## Schema Mapping (Enterprise Tables)
Unity Catalog tables may have different names than the canonical `aco.table` definitions. Use schema mapping:
```python
from aco.lake import Catalog, IcebergContext
# Define mappings
catalog = Catalog(
schema_map={
# Unity Catalog name → canonical name
"prod.encounters": "core.encounter",
"prod.patients": "core.patient",
},
column_map={
# Unity Catalog column → canonical column
"prod.encounters": {
"encntr_id": "encounter_id",
"mbr_id": "person_id",
},
},
)
# Use with IcebergContext
ctx = IcebergContext(
catalog_uri="https://dbc-7474655000864661.cloud.databricks.com/api/2.1/unity-catalog/iceberg",
warehouse="aco",
catalog=catalog, # Apply mappings
properties={"token": "..."},
)
# Load using canonical name, but data comes from prod.encounters
df = ctx.load("core.encounter")
# Columns are automatically renamed from Unity Catalog names
assert "encounter_id" in df.columns # Mapped from encntr_id
assert "person_id" in df.columns # Mapped from mbr_id
```
## Workspace Configuration
### AWS (Default)
```python
workspace_id = "7474655000864661"
host = f"https://dbc-{workspace_id}.cloud.databricks.com"
```
### Azure
```python
workspace_id = "1234567890123456"
host = f"https://adb-{workspace_id}.azuredatabricks.net"
```
### GCP
```python
workspace_id = "1234567890123456"
host = f"https://{workspace_id}.gcp.databricks.com"
```
## Authentication
Unity Catalog supports multiple auth methods:
### Personal Access Token (PAT)
```python
client = UnityClient(
workspace_id="7474655000864661",
token="dapi...",
)
ctx = IcebergContext(
catalog_uri="...",
properties={"token": "dapi..."},
)
```
### OAuth 2.0 (Service Principal)
```python
ctx = IcebergContext(
catalog_uri="...",
properties={
"oauth2-server-uri": "https://dbc-....cloud.databricks.com/oidc/v1/token",
"credential": "client_id:client_secret",
"scope": "all-apis",
},
)
```
## Data Access Patterns
### Read from Unity Catalog
```python
from aco.lake import IcebergContext
ctx = IcebergContext(
catalog_uri="https://dbc-7474655000864661.cloud.databricks.com/api/2.1/unity-catalog/iceberg",
warehouse="aco",
properties={"token": "..."},
)
# Load table as narwhals DataFrame
encounters = ctx.load("core.encounter")
# Use in pipeline
from aco.pipe import readmissions
results = readmissions.pipeline.run(ctx.load)
```
### Write to Unity Catalog
```python
import narwhals as nw
# Create DataFrame
df = nw.from_native({"col1": [1, 2, 3], "col2": ["a", "b", "c"]})
# Save to Unity Catalog (creates Iceberg snapshot)
ctx = IcebergContext(
catalog_uri="...",
warehouse="aco",
properties={"token": "..."},
)
ctx.save("core.my_table", df)
```
### Pipeline Execution
```python
from aco.lake import IcebergContext
from aco.lake.engine import execute
from aco.pipe import readmissions
ctx = IcebergContext(
catalog_uri="https://dbc-7474655000864661.cloud.databricks.com/api/2.1/unity-catalog/iceberg",
warehouse="aco",
properties={"token": "..."},
)
# Execute pipeline, optionally save outputs
results = execute(
pipeline=readmissions.pipeline,
context=ctx,
save_outputs=True,
output_filter={"readmissions.encounter", "readmissions.readmission"},
)
```
## Iceberg Features
Unity Catalog uses Delta Lake by default, but supports Iceberg:
```python
# Create Iceberg table
table = client.create_table(
catalog_name="aco",
schema_name="core",
table_name="encounter",
data_source_format="ICEBERG", # Use Iceberg instead of Delta
columns=[...],
)
# All Iceberg features available:
# - Time travel via snapshots
# - Schema evolution
# - Partition evolution
# - Hidden partitioning
# - ACID transactions
```
## Best Practices
### 1. Use Schema Mapping for Production
Don't rely on exact table name matches. Define explicit mappings:
```python
catalog = Catalog(
schema_map={
"prod.encounters": "core.encounter",
"prod.claims": "core.medical_claim",
},
column_map={
"prod.encounters": {"encntr_id": "encounter_id"},
},
)
```
### 2. Use Managed Tables for Production Data
```python
table = client.create_table(
table_type="MANAGED", # Unity controls storage
data_source_format="ICEBERG",
)
```
### 3. Use External Tables for Source Data
```python
table = client.create_table(
table_type="EXTERNAL",
storage_location="s3://source-bucket/claims/",
data_source_format="PARQUET",
)
```
### 4. Use Volumes for Unstructured Files
```python
volume = client.create_volume(
catalog_name="aco",
schema_name="landing",
volume_name="raw_files",
volume_type="MANAGED",
)
# Access files at: /Volumes/aco/landing/raw_files/file.csv
```
### 5. Leverage Dry Run Mode
Always test changes first:
```bash
uv run python dev/setup_unity_catalog.py --dry-run
```
## Troubleshooting
### Authentication Errors
```
Error: 401 Unauthorized
```
**Solution**: Check your token is valid and has catalog permissions.
### Catalog Not Found
```
Error: Catalog 'aco' does not exist
```
**Solution**: Run setup script to create catalog structure.
### Schema Mismatch
```
Error: Column 'encounter_id' not found
```
**Solution**: Check schema mapping in `Catalog()` configuration.
### Permission Denied
```
Error: User does not have CREATE permission on catalog 'aco'
```
**Solution**: Grant appropriate permissions via Databricks UI or contact admin.
## Development Workflow
### Local Dev → Unity Catalog
```python
# 1. Develop locally with DuckDB
from aco.lake import DuckDBContext
ctx_local = DuckDBContext(database="notebooks/aco.duckdb")
results = execute(readmissions.pipeline, ctx_local)
# 2. Deploy to Unity Catalog
from aco.lake import IcebergContext
ctx_unity = IcebergContext(
catalog_uri="https://dbc-7474655000864661.cloud.databricks.com/api/2.1/unity-catalog/iceberg",
warehouse="aco",
properties={"token": "..."},
)
results = execute(readmissions.pipeline, ctx_unity, save_outputs=True)
```
## See Also
- [Unity Catalog API Documentation](https://docs.databricks.com/api/workspace/catalogs)
- [Iceberg REST Catalog Spec](https://github.com/apache/iceberg/blob/main/open-api/rest-catalog-open-api.yaml)
- `dev/setup_unity_catalog.py` - Automated setup script
- `dev/unity_catalog_example.py` - Complete examples

View File

@@ -47,3 +47,17 @@ from .context import EnterpriseContext as EnterpriseContext
from .context import IcebergContext as IcebergContext from .context import IcebergContext as IcebergContext
from .context import TrinoContext as TrinoContext from .context import TrinoContext as TrinoContext
from .engine import execute as execute from .engine import execute as execute
from .transpile import transpile as transpile
try:
from .sync import SyncReport as SyncReport
from .sync import sync as sync
from .unity import UnityCatalog as UnityCatalog
from .unity import UnityClient as UnityClient
from .unity import UnitySchema as UnitySchema
from .unity import UnityTable as UnityTable
from .unity import UnityTableColumn as UnityTableColumn
from .unity import UnityVolume as UnityVolume
from .unity import setup_catalog_from_schemas as setup_catalog_from_schemas
except ImportError:
pass

Binary file not shown.

Binary file not shown.

Binary file not shown.

View File

@@ -40,12 +40,31 @@ Usage::
cat.iceberg_schema("core.encounter") # pyiceberg.schema.Schema cat.iceberg_schema("core.encounter") # pyiceberg.schema.Schema
cat.iceberg_snapshots("core.encounter") # snapshot history cat.iceberg_snapshots("core.encounter") # snapshot history
cat.validate("core") # compare schema vs iceberg cat.validate("core") # compare schema vs iceberg
# With schema mapping for enterprise (different table names/schemas)
cat = Catalog(
catalog_uri="...",
schema_map={
"core.encounter": "prod.encounters", # Different name
"core.patient": "prod.patients",
},
column_map={
"core.encounter": {
"encounter_id": "encntr_id", # Different column name
"patient_id": "pat_id",
},
},
)
""" """
from __future__ import annotations from __future__ import annotations
import importlib
import inspect
from typing import Any from typing import Any
from aco.table.base import SQLTable
class Catalog: class Catalog:
"""Unified registry bridging ``aco.table`` schemas and Iceberg metadata. """Unified registry bridging ``aco.table`` schemas and Iceberg metadata.
@@ -57,6 +76,10 @@ class Catalog:
- **Online** (with ``catalog_uri``): also connects to an Iceberg - **Online** (with ``catalog_uri``): also connects to an Iceberg
REST Catalog for live metadata queries and validation. REST Catalog for live metadata queries and validation.
The ``schema_map`` and ``column_map`` parameters enable working with
enterprise systems where table/column names differ from the canonical
``aco.table`` definitions.
""" """
def __init__( def __init__(
@@ -64,6 +87,8 @@ class Catalog:
catalog_uri: str = "", catalog_uri: str = "",
warehouse: str = "", warehouse: str = "",
properties: dict[str, str] | None = None, properties: dict[str, str] | None = None,
schema_map: dict[str, str] | None = None,
column_map: dict[str, dict[str, str]] | None = None,
) -> None: ) -> None:
"""Initialize catalog. """Initialize catalog.
@@ -75,13 +100,59 @@ class Catalog:
Default warehouse (e.g. ``s3://lakehouse/``). Default warehouse (e.g. ``s3://lakehouse/``).
properties : dict properties : dict
Extra PyIceberg catalog properties (S3 creds, OAuth2, etc.). Extra PyIceberg catalog properties (S3 creds, OAuth2, etc.).
schema_map : dict
Maps canonical table refs to physical table refs.
E.g. ``{"core.encounter": "prod.encounters"}``
column_map : dict
Maps canonical column names to physical column names per table.
E.g. ``{"core.encounter": {"encounter_id": "encntr_id"}}``
""" """
self.catalog_uri = catalog_uri
self.warehouse = warehouse
self.properties = properties or {}
self.schema_map = schema_map or {}
self.column_map = column_map or {}
# Lazy-loaded Iceberg catalog
self._iceberg_catalog: Any = None
# Cache for discovered SQLTable models
self._model_cache: dict[str, type[SQLTable]] = {}
# ── Schema catalog (aco.table introspection) ───────────────── # ── Schema catalog (aco.table introspection) ─────────────────
def schemas(self) -> list[str]: def schemas(self) -> list[str]:
"""Return sorted schema names from ``aco.table``.""" """Return sorted schema names from ``aco.table``.
raise NotImplementedError
Discovers all modules in ``aco.table`` package and extracts
the ``__schema__`` attribute from ``SQLTable`` subclasses.
Returns
-------
list[str]
Sorted unique schema names like ``['core', 'readmissions', ...]``.
"""
schemas = set()
# Discover all table modules
for module_name in self._discover_table_modules():
try:
module = importlib.import_module(f"aco.table.{module_name}")
# Find all SQLTable subclasses in the module
for name, obj in inspect.getmembers(module, inspect.isclass):
if (
issubclass(obj, SQLTable)
and obj is not SQLTable
and hasattr(obj, "__schema__")
and obj.__schema__
):
schemas.add(obj.__schema__)
except Exception:
# Skip modules that fail to import
continue
return sorted(schemas)
def tables(self, schema: str) -> list[str]: def tables(self, schema: str) -> list[str]:
"""Return qualified table names in a schema. """Return qualified table names in a schema.
@@ -96,21 +167,146 @@ class Catalog:
list[str] list[str]
Sorted list like ``['core.encounter', 'core.patient', ...]``. Sorted list like ``['core.encounter', 'core.patient', ...]``.
""" """
raise NotImplementedError tables = []
for module_name in self._discover_table_modules():
try:
module = importlib.import_module(f"aco.table.{module_name}")
for name, obj in inspect.getmembers(module, inspect.isclass):
if (
issubclass(obj, SQLTable)
and obj is not SQLTable
and hasattr(obj, "__schema__")
and obj.__schema__ == schema
and hasattr(obj, "__tablename__")
and obj.__tablename__
):
qualified = f"{obj.__schema__}.{obj.__tablename__}"
tables.append(qualified)
# Cache the model for later use
self._model_cache[qualified] = obj
except Exception:
continue
return sorted(set(tables))
def columns(self, table_ref: str) -> list[str]: def columns(self, table_ref: str) -> list[str]:
"""Return column names for a table from its ``SQLTable`` model.""" """Return column names for a table from its ``SQLTable`` model.
raise NotImplementedError
def model(self, table_ref: str) -> Any: Parameters
----------
table_ref : str
Qualified table name (e.g. ``core.encounter``).
Returns
-------
list[str]
Sorted list of column names.
"""
model = self.model(table_ref)
return sorted(model.model_fields.keys())
def model(self, table_ref: str) -> type[SQLTable]:
"""Return the ``SQLTable`` subclass for a table. """Return the ``SQLTable`` subclass for a table.
Parameters
----------
table_ref : str
Qualified table name (e.g. ``core.encounter``).
Returns Returns
------- -------
type[SQLTable] type[SQLTable]
The Pydantic model class. The Pydantic model class.
Raises
------
ValueError
If no model found for the given table reference.
""" """
raise NotImplementedError # Check cache first
if table_ref in self._model_cache:
return self._model_cache[table_ref]
# Parse schema.table
if "." not in table_ref:
raise ValueError(f"Table ref must be qualified (schema.table): {table_ref}")
schema, table = table_ref.split(".", 1)
# Search for the model
for module_name in self._discover_table_modules():
try:
module = importlib.import_module(f"aco.table.{module_name}")
for name, obj in inspect.getmembers(module, inspect.isclass):
if (
issubclass(obj, SQLTable)
and obj is not SQLTable
and getattr(obj, "__schema__", "") == schema
and getattr(obj, "__tablename__", "") == table
):
self._model_cache[table_ref] = obj
return obj
except Exception:
continue
raise ValueError(f"No SQLTable model found for {table_ref}")
def physical_table_ref(self, table_ref: str) -> str:
"""Map canonical table ref to physical table ref using schema_map.
Parameters
----------
table_ref : str
Canonical table name (e.g. ``core.encounter``).
Returns
-------
str
Physical table name (e.g. ``prod.encounters`` if mapped,
otherwise returns input unchanged).
"""
return self.schema_map.get(table_ref, table_ref)
def physical_column_name(self, table_ref: str, column_name: str) -> str:
"""Map canonical column name to physical column name using column_map.
Parameters
----------
table_ref : str
Canonical table name (e.g. ``core.encounter``).
column_name : str
Canonical column name (e.g. ``encounter_id``).
Returns
-------
str
Physical column name (e.g. ``encntr_id`` if mapped,
otherwise returns input unchanged).
"""
if table_ref in self.column_map:
return self.column_map[table_ref].get(column_name, column_name)
return column_name
def _discover_table_modules(self) -> list[str]:
"""Discover all Python modules in aco.table package.
Returns
-------
list[str]
Module names (without aco.table prefix).
"""
import pkgutil
import aco.table as table_pkg
modules = []
for importer, modname, ispkg in pkgutil.iter_modules(table_pkg.__path__):
if not modname.startswith("_"):
modules.append(modname)
return modules
# ── Iceberg catalog (REST API via PyIceberg) ───────────────── # ── Iceberg catalog (REST API via PyIceberg) ─────────────────
@@ -119,42 +315,160 @@ class Catalog:
Maps to ``GET /v1/namespaces`` in the REST spec. Maps to ``GET /v1/namespaces`` in the REST spec.
Returns
-------
list[str]
Sorted list of namespace names.
Raises Raises
------ ------
RuntimeError RuntimeError
If no ``catalog_uri`` was provided (offline mode). If no ``catalog_uri`` was provided (offline mode).
""" """
raise NotImplementedError catalog = self._get_iceberg_catalog()
namespaces = catalog.list_namespaces()
# PyIceberg returns tuples like ('core',) or ('schema', 'subschema')
# Flatten to strings
return sorted(
[".".join(ns) if isinstance(ns, tuple) else ns for ns in namespaces]
)
def iceberg_tables(self, namespace: str) -> list[str]: def iceberg_tables(self, namespace: str) -> list[str]:
"""List tables in an Iceberg namespace. """List tables in an Iceberg namespace.
Maps to ``GET /v1/namespaces/{ns}/tables``. Maps to ``GET /v1/namespaces/{ns}/tables``.
Parameters
----------
namespace : str
Namespace name (e.g. ``core``).
Returns
-------
list[str]
Sorted list of qualified table names.
""" """
raise NotImplementedError catalog = self._get_iceberg_catalog()
# PyIceberg expects namespace as tuple for nested namespaces
ns_tuple = tuple(namespace.split("."))
tables = catalog.list_tables(ns_tuple)
# Returns list of tuples like [('core', 'encounter'), ...]
return sorted([f"{'.'.join(t[:-1])}.{t[-1]}" for t in tables])
def iceberg_schema(self, table_ref: str) -> Any: def iceberg_schema(self, table_ref: str) -> Any:
"""Return the Iceberg schema for a table. """Return the Iceberg schema for a table.
Maps to ``GET /v1/namespaces/{ns}/tables/{table}``. Maps to ``GET /v1/namespaces/{ns}/tables/{table}``.
Parameters
----------
table_ref : str
Qualified table name (e.g. ``core.encounter``).
Returns Returns
------- -------
pyiceberg.schema.Schema pyiceberg.schema.Schema
The Iceberg table schema. The Iceberg table schema.
""" """
raise NotImplementedError catalog = self._get_iceberg_catalog()
# Map to physical table if needed
physical_ref = self.physical_table_ref(table_ref)
# Parse namespace and table
if "." not in physical_ref:
raise ValueError(f"Table ref must be qualified: {physical_ref}")
parts = physical_ref.split(".")
namespace = tuple(parts[:-1])
table_name = parts[-1]
table = catalog.load_table((*namespace, table_name))
return table.schema()
def iceberg_snapshots(self, table_ref: str) -> list[Any]: def iceberg_snapshots(self, table_ref: str) -> list[Any]:
"""Return snapshot history for an Iceberg table. """Return snapshot history for an Iceberg table.
Useful for time travel and audit trails. Useful for time travel and audit trails.
Parameters
----------
table_ref : str
Qualified table name (e.g. ``core.encounter``).
Returns
-------
list[Snapshot]
List of Iceberg snapshots ordered by timestamp.
""" """
raise NotImplementedError catalog = self._get_iceberg_catalog()
physical_ref = self.physical_table_ref(table_ref)
parts = physical_ref.split(".")
namespace = tuple(parts[:-1])
table_name = parts[-1]
table = catalog.load_table((*namespace, table_name))
return list(table.snapshots())
def iceberg_current_snapshot(self, table_ref: str) -> Any: def iceberg_current_snapshot(self, table_ref: str) -> Any:
"""Return the current snapshot for an Iceberg table.""" """Return the current snapshot for an Iceberg table.
raise NotImplementedError
Parameters
----------
table_ref : str
Qualified table name (e.g. ``core.encounter``).
Returns
-------
Snapshot | None
Current snapshot or None if table is empty.
"""
catalog = self._get_iceberg_catalog()
physical_ref = self.physical_table_ref(table_ref)
parts = physical_ref.split(".")
namespace = tuple(parts[:-1])
table_name = parts[-1]
table = catalog.load_table((*namespace, table_name))
return table.current_snapshot()
def _get_iceberg_catalog(self) -> Any:
"""Get or create PyIceberg catalog instance.
Returns
-------
RestCatalog
PyIceberg REST catalog.
Raises
------
RuntimeError
If no catalog_uri provided (offline mode).
"""
if not self.catalog_uri:
raise RuntimeError(
"Cannot access Iceberg catalog in offline mode. "
"Provide catalog_uri when initializing Catalog."
)
if self._iceberg_catalog is None:
from pyiceberg.catalog import load_catalog
# Build catalog config
config = {
"uri": self.catalog_uri,
"warehouse": self.warehouse,
**self.properties,
}
self._iceberg_catalog = load_catalog("default", **config)
return self._iceberg_catalog
# ── Validation ─────────────────────────────────────────────── # ── Validation ───────────────────────────────────────────────
@@ -177,4 +491,39 @@ class Catalog:
Validation report with ``missing_in_iceberg``, Validation report with ``missing_in_iceberg``,
``missing_in_schema``, and ``column_mismatches`` keys. ``missing_in_schema``, and ``column_mismatches`` keys.
""" """
raise NotImplementedError # Get tables from both sources
schema_tables = set(self.tables(schema))
iceberg_tables = set(self.iceberg_tables(schema))
# Find differences
missing_in_iceberg = schema_tables - iceberg_tables
missing_in_schema = iceberg_tables - schema_tables
# Check column mismatches for tables that exist in both
common_tables = schema_tables & iceberg_tables
column_mismatches = {}
for table_ref in common_tables:
schema_cols = set(self.columns(table_ref))
try:
iceberg_schema = self.iceberg_schema(table_ref)
iceberg_cols = set(field.name for field in iceberg_schema.fields)
missing_in_iceberg_cols = schema_cols - iceberg_cols
missing_in_schema_cols = iceberg_cols - schema_cols
if missing_in_iceberg_cols or missing_in_schema_cols:
column_mismatches[table_ref] = {
"missing_in_iceberg": sorted(missing_in_iceberg_cols),
"missing_in_schema": sorted(missing_in_schema_cols),
}
except Exception as e:
column_mismatches[table_ref] = {"error": str(e)}
return {
"schema": schema,
"missing_in_iceberg": sorted(missing_in_iceberg),
"missing_in_schema": sorted(missing_in_schema),
"column_mismatches": column_mismatches,
}

View File

@@ -68,8 +68,11 @@ from __future__ import annotations
from typing import Any from typing import Any
import narwhals as nw
from pydantic import BaseModel, ConfigDict from pydantic import BaseModel, ConfigDict
from aco.lake.catalog import Catalog
class Context(BaseModel): class Context(BaseModel):
"""Base class for all storage contexts. """Base class for all storage contexts.
@@ -121,6 +124,22 @@ class DuckDBContext(Context):
pattern used today. pattern used today.
Suitable for: notebooks, unit tests, CI validation. Suitable for: notebooks, unit tests, CI validation.
Usage::
from aco.lake import DuckDBContext
ctx = DuckDBContext(database="notebooks/aco.duckdb")
# Load a table
encounters = ctx.load("core.encounter")
# Use with pipeline
from aco.pipe import readmissions
results = readmissions.pipeline.run(ctx.load)
# Save results
ctx.save("readmissions.encounter", results["readmissions.encounter"])
""" """
database: str database: str
@@ -129,6 +148,167 @@ class DuckDBContext(Context):
read_only: bool = True read_only: bool = True
"""Open database in read-only mode by default.""" """Open database in read-only mode by default."""
catalog: Catalog | None = None
"""Optional Catalog for schema mapping (enterprise table names)."""
# Private connection cache
_connection: Any = None
def load(self, table_ref: str) -> Any:
"""Read a table from DuckDB and wrap in narwhals.
Parameters
----------
table_ref : str
Qualified table name (e.g. ``core.encounter``).
Returns
-------
DataFrame
Narwhals-wrapped polars DataFrame.
Raises
------
ValueError
If table_ref is not qualified (schema.table).
RuntimeError
If table does not exist in DuckDB.
"""
if "." not in table_ref:
raise ValueError(f"Table ref must be qualified (schema.table): {table_ref}")
# Map to physical table if catalog provided
physical_ref = table_ref
if self.catalog:
physical_ref = self.catalog.physical_table_ref(table_ref)
schema, table = physical_ref.split(".", 1)
# Get connection
con = self._get_connection()
# Read table using DuckDB's polars integration
try:
# Use fully qualified table name to avoid ambiguity
query = f'SELECT * FROM "{schema}"."{table}"'
df = con.execute(query).pl()
# Apply column mapping if catalog provided
if self.catalog and table_ref in self.catalog.column_map:
# Rename physical columns back to canonical names
reverse_map = {
physical: canonical
for canonical, physical in self.catalog.column_map[
table_ref
].items()
}
# Only rename columns that exist and need mapping
rename_dict = {
col: reverse_map[col] for col in df.columns if col in reverse_map
}
if rename_dict:
df = df.rename(rename_dict)
return nw.from_native(df)
except Exception as e:
raise RuntimeError(
f"Failed to load table {physical_ref} from {self.database}: {e}"
) from e
def save(self, table_ref: str, df: Any) -> None:
"""Write a DataFrame to DuckDB.
Parameters
----------
table_ref : str
Qualified table name (e.g. ``readmissions.encounter``).
df : DataFrame
Narwhals DataFrame to persist.
Raises
------
ValueError
If table_ref is not qualified.
RuntimeError
If context is read-only or write fails.
"""
if self.read_only:
raise RuntimeError(
"Cannot save to read-only DuckDBContext. "
"Set read_only=False when initializing."
)
if "." not in table_ref:
raise ValueError(f"Table ref must be qualified (schema.table): {table_ref}")
# Map to physical table if catalog provided
physical_ref = table_ref
if self.catalog:
physical_ref = self.catalog.physical_table_ref(table_ref)
schema, table = physical_ref.split(".", 1)
# Apply column mapping if catalog provided
df_to_save = df
if self.catalog and table_ref in self.catalog.column_map:
# Rename canonical columns to physical names
rename_dict = {
canonical: physical
for canonical, physical in self.catalog.column_map[table_ref].items()
if canonical in df.columns
}
if rename_dict:
df_to_save = df.rename(rename_dict)
# Get connection
con = self._get_connection()
# Convert to polars for DuckDB write
pl_df = nw.to_native(df_to_save)
try:
# Create schema if it doesn't exist
con.execute(f'CREATE SCHEMA IF NOT EXISTS "{schema}"')
# Drop table if exists and recreate (replace mode)
con.execute(f'DROP TABLE IF EXISTS "{schema}"."{table}"')
# Write table
con.execute(f'CREATE TABLE "{schema}"."{table}" AS SELECT * FROM pl_df')
except Exception as e:
raise RuntimeError(
f"Failed to save table {physical_ref} to {self.database}: {e}"
) from e
def _get_connection(self) -> Any:
"""Get or create DuckDB connection.
Returns
-------
duckdb.DuckDBPyConnection
DuckDB connection instance.
"""
if self._connection is None:
import duckdb
self._connection = duckdb.connect(
database=self.database, read_only=self.read_only
)
return self._connection
def __del__(self):
"""Close connection on context destruction."""
if self._connection is not None:
try:
self._connection.close()
except Exception:
pass
# ── Iceberg contexts ──────────────────────────────────────────────── # ── Iceberg contexts ────────────────────────────────────────────────
@@ -182,6 +362,180 @@ class IcebergContext(Context):
- ``header.X-Iceberg-Access-Delegation`` — ``vended-credentials`` - ``header.X-Iceberg-Access-Delegation`` — ``vended-credentials``
""" """
catalog: Catalog | None = None
"""Optional Catalog for schema mapping."""
# Private catalog cache
_pyiceberg_catalog: Any = None
def load(self, table_ref: str) -> Any:
"""Read a table from Iceberg via PyIceberg.
Parameters
----------
table_ref : str
Qualified table name (e.g. ``core.encounter``).
Returns
-------
DataFrame
Narwhals-wrapped polars DataFrame.
"""
# Map to physical table if catalog provided
physical_ref = table_ref
if self.catalog:
physical_ref = self.catalog.physical_table_ref(table_ref)
parts = physical_ref.split(".")
namespace = tuple(parts[:-1])
table_name = parts[-1]
# Get PyIceberg catalog
pyice_cat = self._get_pyiceberg_catalog()
try:
# Load table
table = pyice_cat.load_table((*namespace, table_name))
# Scan to arrow then convert to polars
arrow_table = table.scan().to_arrow()
import polars as pl
pl_df = pl.from_arrow(arrow_table)
# Apply column mapping if catalog provided
if self.catalog and table_ref in self.catalog.column_map:
# Rename physical columns back to canonical names
reverse_map = {
physical: canonical
for canonical, physical in self.catalog.column_map[
table_ref
].items()
}
rename_dict = {
col: reverse_map[col] for col in pl_df.columns if col in reverse_map
}
if rename_dict:
pl_df = pl_df.rename(rename_dict)
return nw.from_native(pl_df)
except Exception as e:
raise RuntimeError(
f"Failed to load Iceberg table {physical_ref}: {e}"
) from e
def save(self, table_ref: str, df: Any) -> None:
"""Write a DataFrame to Iceberg.
Parameters
----------
table_ref : str
Qualified table name.
df : DataFrame
Narwhals DataFrame to persist.
"""
# Map to physical table if catalog provided
physical_ref = table_ref
if self.catalog:
physical_ref = self.catalog.physical_table_ref(table_ref)
parts = physical_ref.split(".")
namespace = tuple(parts[:-1])
table_name = parts[-1]
# Apply column mapping if catalog provided
df_to_save = df
if self.catalog and table_ref in self.catalog.column_map:
rename_dict = {
canonical: physical
for canonical, physical in self.catalog.column_map[table_ref].items()
if canonical in df.columns
}
if rename_dict:
df_to_save = df.rename(rename_dict)
# Convert to arrow
import polars as pl
pl_df = nw.to_native(df_to_save)
arrow_table = pl_df.to_arrow()
# Get PyIceberg catalog
pyice_cat = self._get_pyiceberg_catalog()
try:
# Load or create table
try:
table = pyice_cat.load_table((*namespace, table_name))
# Append to existing table
table.append(arrow_table)
except Exception:
# Table doesn't exist, create it
from pyiceberg.schema import Schema
from pyiceberg.types import NestedField
# Convert arrow schema to Iceberg schema
fields = []
for i, field in enumerate(arrow_table.schema):
fields.append(
NestedField(
field_id=i + 1,
name=field.name,
field_type=field.type,
required=False,
)
)
schema = Schema(*fields)
# Create namespace if needed
try:
pyice_cat.create_namespace(namespace)
except Exception:
pass # Namespace already exists
# Create table
table = pyice_cat.create_table(
identifier=(*namespace, table_name), schema=schema
)
# Write data
table.append(arrow_table)
except Exception as e:
raise RuntimeError(
f"Failed to save Iceberg table {physical_ref}: {e}"
) from e
def _get_pyiceberg_catalog(self) -> Any:
"""Get or create PyIceberg catalog instance.
Returns
-------
RestCatalog
PyIceberg REST catalog.
"""
if self._pyiceberg_catalog is None:
from pyiceberg.catalog import load_catalog
config = {
"uri": self.catalog_uri,
"warehouse": self.warehouse,
**self.properties,
}
# Add Nessie ref if specified
if self.ref != "main":
config["ref"] = self.ref
self._pyiceberg_catalog = load_catalog("default", **config)
return self._pyiceberg_catalog
class TrinoContext(Context): class TrinoContext(Context):
"""Push SQL execution to a Trino cluster reading Iceberg tables. """Push SQL execution to a Trino cluster reading Iceberg tables.

View File

@@ -62,19 +62,30 @@ from __future__ import annotations
from typing import Any from typing import Any
from aco.lake.context import Context from aco.lake.context import (
Context,
DuckDBContext,
EnterpriseContext,
IcebergContext,
TrinoContext,
)
from aco.pipe.base import Pipeline from aco.pipe.base import Pipeline
def execute(pipeline: Pipeline, context: Context) -> dict[str, Any]: def execute(
pipeline: Pipeline,
context: Context,
save_outputs: bool = False,
output_filter: set[str] | None = None,
) -> dict[str, Any]:
"""Run a pipeline against a storage context. """Run a pipeline against a storage context.
Dispatches by context type: Dispatches by context type:
- DuckDBContext → ``_execute_direct`` - DuckDBContext → ``_execute_direct``
- IcebergContext → ``_execute_direct`` - IcebergContext → ``_execute_direct``
- TrinoContext → ``_execute_transpiled`` - TrinoContext → ``_execute_transpiled`` (future)
- EnterpriseContext → ``_execute_transpiled`` - EnterpriseContext → ``_execute_transpiled`` (future)
Parameters Parameters
---------- ----------
@@ -82,17 +93,55 @@ def execute(pipeline: Pipeline, context: Context) -> dict[str, Any]:
A Pydantic pipeline definition from ``aco.pipe``. A Pydantic pipeline definition from ``aco.pipe``.
context : Context context : Context
A storage context from ``aco.lake``. A storage context from ``aco.lake``.
save_outputs : bool
If True, persist results back to storage via ``context.save``.
output_filter : set[str] | None
If provided, only save tables in this set. If None, save all.
Returns Returns
------- -------
dict[str, Any] dict[str, Any]
Mapping of output table names to result DataFrames Mapping of output table names to result DataFrames
(or confirmation objects for transpiled push-down). (or confirmation objects for transpiled push-down).
Examples
--------
Execute a pipeline locally::
from aco.lake import DuckDBContext
from aco.lake.engine import execute
from aco.pipe import readmissions
ctx = DuckDBContext(database="notebooks/aco.duckdb")
results = execute(readmissions.pipeline, ctx)
# Access specific result
encounter_df = results["readmissions._int_encounter"]
Execute with selective output saving::
results = execute(
readmissions.pipeline,
ctx,
save_outputs=True,
output_filter={"readmissions.encounter", "readmissions.readmission"}
)
""" """
raise NotImplementedError # Dispatch based on context type
if isinstance(context, (DuckDBContext, IcebergContext)):
return _execute_direct(pipeline, context, save_outputs, output_filter)
elif isinstance(context, (TrinoContext, EnterpriseContext)):
return _execute_transpiled(pipeline, context, save_outputs, output_filter)
else:
raise TypeError(f"Unsupported context type: {type(context)}")
def _execute_direct(pipeline: Pipeline, context: Context) -> dict[str, Any]: def _execute_direct(
pipeline: Pipeline,
context: Context,
save_outputs: bool = False,
output_filter: set[str] | None = None,
) -> dict[str, Any]:
"""Run express functions locally via narwhals. """Run express functions locally via narwhals.
Steps: Steps:
@@ -114,16 +163,54 @@ def _execute_direct(pipeline: Pipeline, context: Context) -> dict[str, Any]:
con.execute("SELECT * FROM schema.table").pl() con.execute("SELECT * FROM schema.table").pl()
→ narwhals.from_native() → narwhals.from_native()
Parameters
----------
pipeline : Pipeline
Pipeline to execute.
context : Context
DuckDBContext or IcebergContext.
save_outputs : bool
Whether to persist results back to storage.
output_filter : set[str] | None
If provided, only save tables in this set.
Returns
-------
dict[str, Any]
Cache of all computed DataFrames by table name.
""" """
raise NotImplementedError # Execute pipeline using context's load function
# The pipeline.run() method handles dependency resolution
# and calls context.load() for inputs as needed
results = pipeline.run(context.load)
# Optionally save outputs
if save_outputs:
tables_to_save = output_filter if output_filter else set(pipeline.names())
for table_ref in tables_to_save:
if table_ref in results:
try:
context.save(table_ref, results[table_ref])
except Exception as e:
# Log error but don't fail entire execution
print(f"Warning: Failed to save {table_ref}: {e}")
return results
def _execute_transpiled(pipeline: Pipeline, context: Context) -> dict[str, Any]: def _execute_transpiled(
pipeline: Pipeline,
context: Context,
save_outputs: bool = False,
output_filter: set[str] | None = None,
) -> dict[str, Any]:
"""Transpile express functions to SQL and push to remote engine. """Transpile express functions to SQL and push to remote engine.
Steps: Steps:
1. For each step in the pipeline, translate the narwhals 1. For each expression in the pipeline, translate the narwhals
expression tree to the target SQL dialect. expression tree to the target SQL dialect.
2. Execute the SQL on the remote engine (Trino / Databricks). 2. Execute the SQL on the remote engine (Trino / Databricks).
3. Return confirmation objects or fetch results as needed. 3. Return confirmation objects or fetch results as needed.
@@ -137,5 +224,60 @@ def _execute_transpiled(pipeline: Pipeline, context: Context) -> dict[str, Any]:
For Databricks, the target is Unity Catalog — which implements For Databricks, the target is Unity Catalog — which implements
the same Iceberg REST spec, so table references resolve the same Iceberg REST spec, so table references resolve
identically across contexts. identically across contexts.
Parameters
----------
pipeline : Pipeline
Pipeline to execute.
context : Context
TrinoContext or EnterpriseContext.
save_outputs : bool
Whether to persist results (typically handled by remote engine).
output_filter : set[str] | None
If provided, only save tables in this set.
Returns
-------
dict[str, Any]
Mapping of table names to execution results
(may be metadata rather than actual DataFrames).
Raises
------
NotImplementedError
Transpilation is not yet implemented.
""" """
raise NotImplementedError import duckdb
from aco.lake.transpile import transpile
# Transpilation requires a DuckDB connection with source schemas
# to create relations that capture SQL through narwhals.
if not hasattr(context, "_duckdb_path") or not context._duckdb_path:
raise ValueError(
"Transpiled execution requires a DuckDB path for schema reference. "
"Set context._duckdb_path or use transpile() directly."
)
con = duckdb.connect(context._duckdb_path, read_only=True)
# Determine target dialect and catalog from context
dialect = "databricks"
catalog_name = ""
if isinstance(context, EnterpriseContext):
dialect = context.dialect or "databricks"
catalog_name = context.warehouse
elif isinstance(context, TrinoContext):
dialect = "trino"
catalog_name = context.catalog
sql_map = transpile(
pipeline,
con,
target_dialect=dialect,
catalog=catalog_name,
output_mode="ctas",
)
con.close()
return sql_map

425
src/aco/lake/sync.py Normal file
View File

@@ -0,0 +1,425 @@
"""Data sync — move tables between contexts.
Generic context-to-context sync with an optimized Databricks path
that uploads parquet to a staging Volume then runs COPY INTO.
Usage::
from aco.lake import DuckDBContext, sync
ctx = DuckDBContext(database="notebooks/aco.duckdb")
# Sync to Databricks (parquet + COPY INTO)
report = sync(
ctx,
catalog_name="homelab",
warehouse_id="241c97197d3bcfbc",
)
# Sync between local contexts
dst = DuckDBContext(database="/tmp/copy.duckdb", read_only=False)
report = sync(ctx, dst, tables=["core.patient"])
"""
from __future__ import annotations
import io
import time
from typing import Any, Callable, Literal
import narwhals as nw
from pydantic import BaseModel
from aco.lake.context import Context, DuckDBContext
# ── Models ───────────────────────────────────────────────────────────
class SyncEvent(BaseModel):
"""Progress event emitted per table."""
table_ref: str
status: Literal["started", "uploaded", "copied", "skipped", "failed"]
rows: int = 0
bytes: int = 0
elapsed_seconds: float = 0.0
error: str = ""
class SyncFailed(BaseModel):
"""Record of a single table sync failure."""
table_ref: str
error: str
class SyncReport(BaseModel):
"""Summary of a sync run."""
tables_synced: list[str] = []
tables_skipped: list[str] = []
tables_failed: list[SyncFailed] = []
total_rows: int = 0
total_bytes: int = 0
elapsed_seconds: float = 0.0
# ── Public API ───────────────────────────────────────────────────────
def sync(
source: Context,
target: Context | None = None,
*,
tables: list[str] | None = None,
schemas: list[str] | None = None,
skip_empty: bool = True,
catalog_name: str = "homelab",
warehouse_id: str = "",
on_error: Literal["raise", "skip", "collect"] = "collect",
on_progress: Callable[[SyncEvent], None] | None = None,
) -> SyncReport:
"""Sync tables from *source* to *target* (or Databricks).
Parameters
----------
source : Context
Source context to read from.
target : Context | None
Target context for generic path. Ignored when *warehouse_id*
is provided (Databricks path used instead).
tables : list[str] | None
Explicit list of ``schema.table`` refs to sync.
Takes precedence over *schemas*.
schemas : list[str] | None
Sync all tables in these schemas. Ignored if *tables* given.
skip_empty : bool
Skip tables with 0 rows (default True).
catalog_name : str
Databricks catalog name (default ``homelab``).
warehouse_id : str
Databricks SQL warehouse ID. When provided, uses the
optimized parquet-upload + COPY INTO path.
on_error : str
``"collect"`` — accumulate failures in report (default).
``"raise"`` — raise on first failure.
``"skip"`` — silently skip failures.
on_progress : callable | None
Optional callback receiving ``SyncEvent`` per table.
Returns
-------
SyncReport
"""
t0 = time.monotonic()
report = SyncReport()
table_list = _resolve_tables(source, tables, schemas)
if warehouse_id:
from aco.lake.unity import UnityClient
client = UnityClient.from_env()
staging_base = _ensure_staging_volume(client._ws, catalog_name)
for ref in table_list:
ev = _sync_table_databricks(
source,
client._ws,
ref,
catalog_name,
warehouse_id,
staging_base,
skip_empty=skip_empty,
)
_handle_event(ev, report, on_error, on_progress)
elif target is not None:
for ref in table_list:
ev = _sync_table_generic(source, target, ref, skip_empty=skip_empty)
_handle_event(ev, report, on_error, on_progress)
else:
raise ValueError(
"Provide either warehouse_id (Databricks path) or target (generic path)."
)
report.elapsed_seconds = round(time.monotonic() - t0, 2)
return report
# ── Table discovery ──────────────────────────────────────────────────
def _resolve_tables(
source: Context,
tables: list[str] | None,
schemas: list[str] | None,
) -> list[str]:
"""Build sorted list of schema.table refs to sync."""
if tables:
return sorted(tables)
if not isinstance(source, DuckDBContext):
raise ValueError(
"Auto-discovery requires DuckDBContext as source. "
"Pass an explicit tables= list for other contexts."
)
con = source._get_connection()
rows = con.execute(
"SELECT table_schema || '.' || table_name "
"FROM information_schema.tables "
"WHERE table_schema NOT IN "
"('information_schema', 'pg_catalog') "
"ORDER BY table_schema, table_name"
).fetchall()
all_tables = [r[0] for r in rows]
if schemas:
schema_set = set(schemas)
all_tables = [t for t in all_tables if t.split(".", 1)[0] in schema_set]
return all_tables
# ── Databricks path ─────────────────────────────────────────────────
def _sync_table_databricks(
source: Context,
ws: Any,
table_ref: str,
catalog_name: str,
warehouse_id: str,
staging_base: str,
*,
skip_empty: bool = True,
) -> SyncEvent:
"""Upload one table as parquet then COPY INTO on Databricks."""
t0 = time.monotonic()
schema, table = table_ref.split(".", 1)
try:
df = source.load(table_ref)
except Exception as e:
return SyncEvent(
table_ref=table_ref,
status="failed",
error=f"load: {e}",
elapsed_seconds=round(time.monotonic() - t0, 2),
)
pl_df = nw.to_native(df)
row_count = len(pl_df)
if skip_empty and row_count == 0:
return SyncEvent(table_ref=table_ref, status="skipped", rows=0)
# Upcast narrow types so parquet matches Delta's expectations
pl_df = _widen_for_delta(pl_df)
# Write parquet to memory
buf = io.BytesIO()
pl_df.write_parquet(buf)
parquet_bytes = buf.tell()
buf.seek(0)
# Upload to staging volume — prefix with 'data_' to avoid Spark
# ignoring files whose names start with '_' (treated as hidden)
safe_name = f"data_{table}" if table.startswith("_") else table
vol_path = f"{staging_base}/{schema}/{safe_name}.parquet"
try:
ws.files.upload(vol_path, buf, overwrite=True)
except Exception as e:
return SyncEvent(
table_ref=table_ref,
status="failed",
rows=row_count,
bytes=parquet_bytes,
error=f"upload: {e}",
elapsed_seconds=round(time.monotonic() - t0, 2),
)
# COPY INTO
copy_sql = (
f"COPY INTO `{catalog_name}`.`{schema}`.`{table}` "
f"FROM '{vol_path}' "
f"FILEFORMAT = PARQUET "
f"COPY_OPTIONS ('force' = 'true', 'mergeSchema' = 'true')"
)
try:
resp = ws.statement_execution.execute_statement(
warehouse_id=warehouse_id,
statement=copy_sql,
wait_timeout="50s",
)
state = str(resp.status.state)
if "SUCCEEDED" not in state:
err_msg = ""
if resp.status.error:
err_msg = str(resp.status.error.message)
return SyncEvent(
table_ref=table_ref,
status="failed",
rows=row_count,
bytes=parquet_bytes,
error=f"COPY INTO {state}: {err_msg}",
elapsed_seconds=round(time.monotonic() - t0, 2),
)
except Exception as e:
return SyncEvent(
table_ref=table_ref,
status="failed",
rows=row_count,
bytes=parquet_bytes,
error=f"COPY INTO: {e}",
elapsed_seconds=round(time.monotonic() - t0, 2),
)
# Cleanup staging file
try:
ws.files.delete(vol_path)
except Exception:
pass # best-effort cleanup
elapsed = round(time.monotonic() - t0, 2)
return SyncEvent(
table_ref=table_ref,
status="copied",
rows=row_count,
bytes=parquet_bytes,
elapsed_seconds=elapsed,
)
# ── Generic path ─────────────────────────────────────────────────────
def _sync_table_generic(
source: Context,
target: Context,
table_ref: str,
*,
skip_empty: bool = True,
) -> SyncEvent:
"""Sync one table via source.load() → target.save()."""
t0 = time.monotonic()
try:
df = source.load(table_ref)
except Exception as e:
return SyncEvent(
table_ref=table_ref,
status="failed",
error=f"load: {e}",
elapsed_seconds=round(time.monotonic() - t0, 2),
)
row_count = len(nw.to_native(df))
if skip_empty and row_count == 0:
return SyncEvent(table_ref=table_ref, status="skipped", rows=0)
try:
target.save(table_ref, df)
except Exception as e:
return SyncEvent(
table_ref=table_ref,
status="failed",
rows=row_count,
error=f"save: {e}",
elapsed_seconds=round(time.monotonic() - t0, 2),
)
elapsed = round(time.monotonic() - t0, 2)
return SyncEvent(
table_ref=table_ref,
status="copied",
rows=row_count,
elapsed_seconds=elapsed,
)
# ── Helpers ──────────────────────────────────────────────────────────
def _ensure_staging_volume(ws: Any, catalog_name: str) -> str:
"""Create the staging volume if it doesn't exist.
Returns the Volumes path prefix, e.g.
``/Volumes/homelab/default/_sync_staging``.
"""
vol_path = f"/Volumes/{catalog_name}/default/_sync_staging"
try:
ws.volumes.read(f"{catalog_name}.default._sync_staging")
except Exception:
try:
from databricks.sdk.service.catalog import VolumeType
ws.volumes.create(
catalog_name=catalog_name,
schema_name="default",
name="_sync_staging",
volume_type=VolumeType.MANAGED,
comment="Transient staging for aco.lake.sync",
)
print(f" created volume {vol_path}")
except Exception:
pass # may already exist from a race
return vol_path
def _handle_event(
ev: SyncEvent,
report: SyncReport,
on_error: str,
on_progress: Callable[[SyncEvent], None] | None,
) -> None:
"""Update report and invoke callback for one sync event."""
if on_progress:
on_progress(ev)
if ev.status == "copied":
report.tables_synced.append(ev.table_ref)
report.total_rows += ev.rows
report.total_bytes += ev.bytes
print(f" {ev.table_ref} {ev.rows:,} rows {ev.elapsed_seconds}s")
elif ev.status == "skipped":
report.tables_skipped.append(ev.table_ref)
elif ev.status == "failed":
print(f" FAIL {ev.table_ref}: {ev.error}")
if on_error == "raise":
raise RuntimeError(f"Sync failed for {ev.table_ref}: {ev.error}")
if on_error == "collect":
report.tables_failed.append(
SyncFailed(table_ref=ev.table_ref, error=ev.error)
)
def _widen_for_delta(pl_df: Any) -> Any:
"""Upcast narrow polars types to match Delta Lake conventions.
Delta uses BIGINT (Int64), DOUBLE (Float64), and DECIMAL(38,2).
Parquet written with narrower types triggers DELTA_FAILED_TO_MERGE_FIELDS.
"""
import polars as pl
cast_map = {}
for col_name, dtype in zip(pl_df.columns, pl_df.dtypes):
if dtype in (pl.Int8, pl.Int16, pl.Int32):
cast_map[col_name] = pl.Int64
elif dtype in (pl.UInt8, pl.UInt16, pl.UInt32, pl.UInt64):
cast_map[col_name] = pl.Int64
elif dtype == pl.Float32:
cast_map[col_name] = pl.Float64
elif isinstance(dtype, pl.Decimal):
if dtype.scale == 0:
# HUGEINT → Decimal(38,0) in polars; Databricks wants BIGINT
cast_map[col_name] = pl.Int64
else:
cast_map[col_name] = pl.Decimal(precision=38, scale=2)
if cast_map:
pl_df = pl_df.cast(cast_map)
return pl_df

220
src/aco/lake/transpile.py Normal file
View File

@@ -0,0 +1,220 @@
"""SQL transpilation — generate target-dialect SQL from pipelines.
Runs pipeline expressions against DuckDB *relations* (not DataFrames)
to capture the SQL that narwhals operations produce, then uses
sqlglot to transpile from DuckDB dialect to the target (Databricks,
Trino, etc.) and rewrite table references.
Usage::
import duckdb
from aco.lake.transpile import transpile
from aco.pipe import readmissions
con = duckdb.connect("notebooks/aco.duckdb", read_only=True)
sql_map = transpile(
readmissions.pipeline,
con,
target_dialect="databricks",
catalog="homelab",
)
for name, sql in sql_map.items():
print(f"-- {name}")
print(sql)
"""
from __future__ import annotations
import inspect
import re
from typing import Any
import duckdb
import sqlglot
from sqlglot import exp
from aco.pipe.base import Pipeline
from aco.pipe.runner import _param_to_table
def transpile(
pipeline: Pipeline,
con: duckdb.DuckDBPyConnection,
*,
target_dialect: str = "databricks",
catalog: str = "",
output_mode: str = "ctas",
) -> dict[str, str]:
"""Generate target-dialect SQL for every expression in a pipeline.
Parameters
----------
pipeline : Pipeline
Pipeline whose expressions will be transpiled.
con : duckdb.DuckDBPyConnection
DuckDB connection with source tables (read-only is fine).
Used to create relations that capture SQL through narwhals.
target_dialect : str
sqlglot output dialect (``databricks``, ``trino``, ``snowflake``).
catalog : str
Catalog prefix to prepend to all table references.
E.g. ``"homelab"`` turns ``core.encounter`` into
``homelab.core.encounter``.
output_mode : str
``"ctas"`` — CREATE OR REPLACE TABLE ... AS SELECT
``"select"`` — bare SELECT statements
``"view"`` — CREATE OR REPLACE VIEW ... AS SELECT
Returns
-------
dict[str, str]
Mapping of expression name to transpiled SQL string.
"""
cache: dict[str, Any] = {}
result: dict[str, str] = {}
for expr in pipeline.exprs:
name = expr.name
fn = expr.fn
# Resolve dependencies via inspect.signature
sig = inspect.signature(fn)
kwargs = {}
for param in sig.parameters:
table_ref = _param_to_table(param)
if table_ref in cache:
kwargs[param] = cache[table_ref]
else:
kwargs[param] = _load_relation(con, table_ref)
# Execute express function with DuckDB relations
try:
relation = fn(**kwargs)
except Exception as e:
result[name] = f"-- ERROR: {e}"
# Try to fall back to existing table for downstream deps
try:
cache[name] = _load_relation(con, name)
except Exception:
pass
continue
# Extract DuckDB SQL
duckdb_sql = relation.sql_query()
# Register as temp view for downstream expressions
view_name = _table_ref_to_view_name(name)
con.execute(f"CREATE OR REPLACE TEMP VIEW {view_name} AS {duckdb_sql}")
cache[name] = con.sql(f"SELECT * FROM {view_name}")
# Transpile to target dialect
target_sql = _transpile_sql(
duckdb_sql,
expr_name=name,
target_dialect=target_dialect,
catalog=catalog,
output_mode=output_mode,
)
result[name] = target_sql
return result
def _load_relation(
con: duckdb.DuckDBPyConnection,
table_ref: str,
) -> duckdb.DuckDBPyRelation:
"""Create a DuckDB relation for a table reference."""
schema, table = table_ref.split(".", 1)
return con.sql(f'SELECT * FROM "{schema}"."{table}"')
def _table_ref_to_view_name(table_ref: str) -> str:
"""Convert a table ref to a valid DuckDB temp view name.
``readmissions._int_encounter`` → ``"_tv_readmissions___int_encounter"``
"""
safe = table_ref.replace(".", "__")
return f'"_tv_{safe}"'
def _transpile_sql(
duckdb_sql: str,
*,
expr_name: str,
target_dialect: str,
catalog: str,
output_mode: str,
) -> str:
"""Transpile DuckDB SQL to target dialect with table ref rewriting."""
tree = sqlglot.parse_one(duckdb_sql, read="duckdb")
# Rewrite table references
for table in tree.find_all(exp.Table):
tbl_name = table.name
db = table.db # schema in sqlglot terminology
# Rewrite temp view names back to proper schema.table
if tbl_name.startswith("_tv_"):
real_ref = tbl_name[4:].replace("__", ".", 1)
parts = real_ref.split(".", 1)
if len(parts) == 2:
table.set("db", exp.to_identifier(parts[0]))
table.set("this", exp.to_identifier(parts[1]))
db = parts[0]
# Add catalog prefix if specified
if catalog and db and not table.catalog:
table.set("catalog", exp.to_identifier(catalog))
# Clean DuckDB internal alias names from the AST
_clean_duckdb_aliases(tree)
# Generate target SQL with pretty printing
select_sql = tree.sql(dialect=target_dialect, pretty=True)
if output_mode == "select":
return select_sql
# Build qualified target table name
target_ref = expr_name
if catalog:
target_ref = f"{catalog}.{expr_name}"
if output_mode == "ctas":
return f"CREATE OR REPLACE TABLE {target_ref} AS\n{select_sql}"
elif output_mode == "view":
return f"CREATE OR REPLACE VIEW {target_ref} AS\n{select_sql}"
else:
return select_sql
# Patterns for DuckDB internal names
_UNNAMED_RE = re.compile(r"unnamed_relation_[0-9a-f]+")
_ROW_INDEX_RE = re.compile(r"row_index_[0-9a-f]+")
def _clean_duckdb_aliases(tree: exp.Expression) -> None:
"""Remove DuckDB internal artifacts from the AST.
- Rename ``unnamed_relation_<hex>`` aliases to short ``t1``, ``t2``, ...
- Rename ``row_index_<hex>`` columns to ``_dedup_row``
"""
# Collect and rename unnamed_relation aliases
counter = 0
seen: dict[str, str] = {}
for node in tree.walk():
if isinstance(node, exp.TableAlias):
name = node.name
if _UNNAMED_RE.fullmatch(name):
if name not in seen:
counter += 1
seen[name] = f"t{counter}"
node.set("this", exp.to_identifier(seen[name]))
elif isinstance(node, exp.Identifier):
name = node.name
if _UNNAMED_RE.fullmatch(name):
if name in seen:
node.set("this", seen[name])
elif _ROW_INDEX_RE.fullmatch(name):
node.set("this", "_dedup_row")

639
src/aco/lake/unity.py Normal file
View File

@@ -0,0 +1,639 @@
"""Unity Catalog integration for Databricks.
Provides Pydantic models and API client for managing Unity Catalog
objects (catalogs, schemas, tables, volumes) via the Databricks SDK.
Usage::
from aco.lake.unity import UnityClient, setup_catalog_from_schemas
# Connect — reads DATABRICKS_HOST and DATABRICKS_TOKEN from .env
client = UnityClient.from_env()
# List catalogs
for cat in client.list_catalogs():
print(cat.name)
# Setup all schemas from aco.table definitions
setup_catalog_from_schemas(client, catalog_name="aco")
"""
from __future__ import annotations
import os
from typing import Any, Literal
from databricks.sdk import WorkspaceClient
from databricks.sdk.service.catalog import (
CatalogInfo,
ColumnInfo,
SchemaInfo,
TableInfo,
VolumeInfo,
)
from pydantic import BaseModel, ConfigDict, Field
# ── Pydantic Models ──────────────────────────────────────────────────
class UnityCatalog(BaseModel):
"""Unity Catalog catalog object."""
model_config = ConfigDict(extra="ignore")
name: str
comment: str | None = None
owner: str | None = None
storage_root: str | None = None
metastore_id: str | None = None
full_name: str | None = None
created_at: int | None = None
updated_at: int | None = None
catalog_type: str | None = None
isolation_mode: str | None = None
@classmethod
def from_sdk(cls, info: CatalogInfo) -> UnityCatalog:
return cls(
name=info.name or "",
comment=info.comment,
owner=info.owner,
storage_root=info.storage_root,
metastore_id=info.metastore_id,
full_name=info.full_name,
created_at=info.created_at,
updated_at=info.updated_at,
catalog_type=str(info.catalog_type) if info.catalog_type else None,
isolation_mode=str(info.isolation_mode) if info.isolation_mode else None,
)
class UnitySchema(BaseModel):
"""Unity Catalog schema (database) object."""
model_config = ConfigDict(extra="ignore")
name: str
catalog_name: str
comment: str | None = None
owner: str | None = None
storage_root: str | None = None
full_name: str | None = None
created_at: int | None = None
updated_at: int | None = None
@classmethod
def from_sdk(cls, info: SchemaInfo) -> UnitySchema:
return cls(
name=info.name or "",
catalog_name=info.catalog_name or "",
comment=info.comment,
owner=info.owner,
storage_root=info.storage_root,
full_name=info.full_name,
created_at=info.created_at,
updated_at=info.updated_at,
)
class UnityTableColumn(BaseModel):
"""Table column definition."""
model_config = ConfigDict(extra="ignore", populate_by_name=True)
name: str
type_text: str
type_name: str
comment: str | None = None
nullable: bool = True
position: int = 0
@classmethod
def from_sdk(cls, info: ColumnInfo) -> UnityTableColumn:
return cls(
name=info.name or "",
type_text=info.type_text or "",
type_name=str(info.type_name) if info.type_name else "",
comment=info.comment,
nullable=info.nullable if info.nullable is not None else True,
position=info.position or 0,
)
class UnityTable(BaseModel):
"""Unity Catalog table object."""
model_config = ConfigDict(extra="ignore", populate_by_name=True)
name: str
catalog_name: str
schema_name: str
table_type: str = "MANAGED"
data_source_format: str | None = None
columns: list[UnityTableColumn] = []
comment: str | None = None
storage_location: str | None = None
owner: str | None = None
full_name: str | None = None
created_at: int | None = None
updated_at: int | None = None
@classmethod
def from_sdk(cls, info: TableInfo) -> UnityTable:
columns = []
if info.columns:
columns = [UnityTableColumn.from_sdk(c) for c in info.columns]
return cls(
name=info.name or "",
catalog_name=info.catalog_name or "",
schema_name=info.schema_name or "",
table_type=str(info.table_type) if info.table_type else "MANAGED",
data_source_format=str(info.data_source_format)
if info.data_source_format
else None,
columns=columns,
comment=info.comment,
storage_location=info.storage_location,
owner=info.owner,
full_name=info.full_name,
created_at=info.created_at,
updated_at=info.updated_at,
)
class UnityVolume(BaseModel):
"""Unity Catalog volume object."""
model_config = ConfigDict(extra="ignore")
name: str
catalog_name: str
schema_name: str
volume_type: str = "MANAGED"
storage_location: str | None = None
comment: str | None = None
owner: str | None = None
full_name: str | None = None
created_at: int | None = None
updated_at: int | None = None
@classmethod
def from_sdk(cls, info: VolumeInfo) -> UnityVolume:
return cls(
name=info.name or "",
catalog_name=info.catalog_name or "",
schema_name=info.schema_name or "",
volume_type=str(info.volume_type) if info.volume_type else "MANAGED",
storage_location=info.storage_location,
comment=info.comment,
owner=info.owner,
full_name=info.full_name,
created_at=info.created_at,
updated_at=info.updated_at,
)
# ── API Client ───────────────────────────────────────────────────────
class UnityClient:
"""Unity Catalog client backed by the Databricks SDK.
Wraps ``databricks.sdk.WorkspaceClient`` and returns Pydantic models.
"""
def __init__(self, workspace: WorkspaceClient) -> None:
self._ws = workspace
@classmethod
def from_env(cls, dotenv_path: str = ".env") -> UnityClient:
"""Create client from environment variables.
Reads Databricks credentials from ``.env`` (or shell environment).
Supports two auth modes:
1. **PAT token**: ``DATABRICKS_HOST`` + ``DATABRICKS_TOKEN``
2. **OAuth M2M**: ``DATABRICKS_HOST`` + ``DATABRICKS_CLIENT_ID``
+ ``DATABRICKS_CLIENT_SECRET`` + ``DATABRICKS_ACCOUNT_ID``
Parameters
----------
dotenv_path : str
Path to ``.env`` file (default: project root).
"""
# Load from .env file if it exists
env_path = dotenv_path
if os.path.exists(env_path):
with open(env_path) as f:
for line in f:
line = line.strip()
if line and not line.startswith("#") and "=" in line:
key, _, value = line.partition("=")
key = key.strip()
value = value.strip()
if key.startswith("DATABRICKS_"):
os.environ[key] = value
host = os.environ.get("DATABRICKS_HOST")
if not host:
raise RuntimeError("DATABRICKS_HOST not set")
token = os.environ.get("DATABRICKS_TOKEN")
client_id = os.environ.get("DATABRICKS_CLIENT_ID")
client_secret = os.environ.get("DATABRICKS_CLIENT_SECRET")
if token:
# PAT auth — clear OAuth vars to avoid SDK conflict
for key in ("DATABRICKS_CLIENT_ID", "DATABRICKS_CLIENT_SECRET"):
os.environ.pop(key, None)
ws = WorkspaceClient(host=host, token=token)
elif client_id and client_secret:
# OAuth M2M — clear TOKEN to avoid SDK conflict
os.environ.pop("DATABRICKS_TOKEN", None)
ws = WorkspaceClient(
host=host,
client_id=client_id,
client_secret=client_secret,
)
else:
raise RuntimeError(
"Set DATABRICKS_TOKEN or "
"DATABRICKS_CLIENT_ID + DATABRICKS_CLIENT_SECRET"
)
return cls(ws)
@property
def host(self) -> str:
return str(self._ws.config.host)
# ── Catalogs ─────────────────────────────────────────────────
def list_catalogs(self) -> list[UnityCatalog]:
return [UnityCatalog.from_sdk(c) for c in self._ws.catalogs.list()]
def get_catalog(self, name: str) -> UnityCatalog:
return UnityCatalog.from_sdk(self._ws.catalogs.get(name))
def create_catalog(
self, name: str, comment: str = "", storage_root: str | None = None
) -> UnityCatalog:
info = self._ws.catalogs.create(
name=name, comment=comment, storage_root=storage_root
)
return UnityCatalog.from_sdk(info)
def update_catalog(
self, name: str, comment: str | None = None, owner: str | None = None
) -> UnityCatalog:
return UnityCatalog.from_sdk(
self._ws.catalogs.update(name, comment=comment, owner=owner)
)
def delete_catalog(self, name: str, force: bool = False) -> None:
self._ws.catalogs.delete(name, force=force)
# ── Schemas ──────────────────────────────────────────────────
def list_schemas(self, catalog_name: str) -> list[UnitySchema]:
return [
UnitySchema.from_sdk(s)
for s in self._ws.schemas.list(catalog_name=catalog_name)
]
def get_schema(self, catalog_name: str, schema_name: str) -> UnitySchema:
full = f"{catalog_name}.{schema_name}"
return UnitySchema.from_sdk(self._ws.schemas.get(full))
def create_schema(
self,
catalog_name: str,
schema_name: str,
comment: str = "",
storage_root: str | None = None,
) -> UnitySchema:
info = self._ws.schemas.create(
name=schema_name,
catalog_name=catalog_name,
comment=comment,
storage_root=storage_root,
)
return UnitySchema.from_sdk(info)
def update_schema(
self,
catalog_name: str,
schema_name: str,
comment: str | None = None,
owner: str | None = None,
) -> UnitySchema:
full = f"{catalog_name}.{schema_name}"
return UnitySchema.from_sdk(
self._ws.schemas.update(full, comment=comment, owner=owner)
)
def delete_schema(
self, catalog_name: str, schema_name: str, force: bool = False
) -> None:
full = f"{catalog_name}.{schema_name}"
self._ws.schemas.delete(full, force=force)
# ── Tables ───────────────────────────────────────────────────
def list_tables(self, catalog_name: str, schema_name: str) -> list[UnityTable]:
return [
UnityTable.from_sdk(t)
for t in self._ws.tables.list(
catalog_name=catalog_name, schema_name=schema_name
)
]
def get_table(
self, catalog_name: str, schema_name: str, table_name: str
) -> UnityTable:
full = f"{catalog_name}.{schema_name}.{table_name}"
return UnityTable.from_sdk(self._ws.tables.get(full))
def delete_table(
self, catalog_name: str, schema_name: str, table_name: str
) -> None:
full = f"{catalog_name}.{schema_name}.{table_name}"
self._ws.tables.delete(full)
# ── Volumes ──────────────────────────────────────────────────
def list_volumes(self, catalog_name: str, schema_name: str) -> list[UnityVolume]:
return [
UnityVolume.from_sdk(v)
for v in self._ws.volumes.list(
catalog_name=catalog_name, schema_name=schema_name
)
]
def create_volume(
self,
catalog_name: str,
schema_name: str,
volume_name: str,
volume_type: str = "MANAGED",
storage_location: str | None = None,
comment: str | None = None,
) -> UnityVolume:
from databricks.sdk.service.catalog import VolumeType
vt = VolumeType(volume_type)
info = self._ws.volumes.create(
catalog_name=catalog_name,
schema_name=schema_name,
name=volume_name,
volume_type=vt,
storage_location=storage_location,
comment=comment,
)
return UnityVolume.from_sdk(info)
def delete_volume(
self, catalog_name: str, schema_name: str, volume_name: str
) -> None:
full = f"{catalog_name}.{schema_name}.{volume_name}"
self._ws.volumes.delete(full)
# ── Type Mapping ─────────────────────────────────────────────────────
def _python_type_to_databricks(annotation: Any) -> tuple[str, str]:
"""Map a Pydantic field annotation to (ColumnTypeName, type_text).
Handles ``str``, ``int``, ``float``, ``date``, ``datetime``,
``Decimal``, ``bool``, and their ``| None`` optionals.
Returns
-------
tuple[str, str]
(type_name, type_text) for ColumnInfo.
"""
import types
from datetime import date, datetime
from decimal import Decimal
from typing import get_args, get_origin
# Unwrap Optional / Union — pick the first non-None arg
origin = get_origin(annotation)
if origin is types.UnionType:
for arg in get_args(annotation):
if arg is not type(None):
return _python_type_to_databricks(arg)
mapping: dict[type, tuple[str, str]] = {
str: ("STRING", "string"),
int: ("LONG", "long"),
float: ("DOUBLE", "double"),
bool: ("BOOLEAN", "boolean"),
date: ("DATE", "date"),
datetime: ("TIMESTAMP_NTZ", "timestamp_ntz"),
Decimal: ("DECIMAL", "decimal(38,2)"),
}
result = mapping.get(annotation)
if result:
return result
# Fallback
return ("STRING", "string")
def _sql_table_to_column_infos(
model: type,
) -> list:
"""Convert SQLTable model fields to Databricks ColumnInfo objects."""
import json
from databricks.sdk.service.catalog import ColumnInfo, ColumnTypeName
# type_json values expected by Databricks
_type_json_map = {
"STRING": '"string"',
"LONG": '"long"',
"DOUBLE": '"double"',
"BOOLEAN": '"boolean"',
"DATE": '"date"',
"TIMESTAMP_NTZ": '"timestamp_ntz"',
"DECIMAL": '{"type":"decimal","precision":38,"scale":2}',
}
columns = []
for position, (field_name, field_info) in enumerate(model.model_fields.items()):
type_name_str, type_text = _python_type_to_databricks(field_info.annotation)
type_json = _type_json_map.get(type_name_str, '"string"')
columns.append(
ColumnInfo(
name=field_name,
type_name=ColumnTypeName(type_name_str),
type_text=type_text,
type_json=type_json,
nullable=True,
position=position,
comment=(field_info.description or "")[:255] or None,
)
)
return columns
# ── Setup Automation ─────────────────────────────────────────────────
def _model_to_ddl(
catalog_name: str,
schema_name: str,
table_name: str,
model: type,
) -> str:
"""Generate CREATE TABLE IF NOT EXISTS DDL from a SQLTable model."""
col_defs = []
for field_name, field_info in model.model_fields.items():
_, type_text = _python_type_to_databricks(field_info.annotation)
col_defs.append(f" `{field_name}` {type_text.upper()}")
columns_sql = ",\n".join(col_defs)
# Sanitize docstring for COMMENT
doc = (model.__doc__ or "").strip().replace("'", "\\'")
if len(doc) > 1024:
doc = doc[:1021] + "..."
ddl = (
f"CREATE TABLE IF NOT EXISTS"
f" `{catalog_name}`.`{schema_name}`.`{table_name}`"
f" (\n{columns_sql}\n)"
)
if doc:
ddl += f"\nCOMMENT '{doc}'"
return ddl
def setup_catalog_from_schemas(
client: UnityClient,
catalog_name: str,
warehouse_id: str,
dry_run: bool = True,
skip_existing: bool = True,
) -> dict[str, Any]:
"""Create Unity Catalog schemas **and tables** from ``aco.table``.
Uses SQL DDL executed via a SQL warehouse to create managed Delta
tables with full column definitions.
Parameters
----------
client : UnityClient
Authenticated Unity Catalog client.
catalog_name : str
Target catalog name (must already exist).
warehouse_id : str
Databricks SQL warehouse ID for executing DDL.
dry_run : bool
If True, only report what would be created.
skip_existing : bool
If True, use CREATE TABLE IF NOT EXISTS (default).
Returns
-------
dict
Report with ``schemas_created``, ``tables_created``,
``schemas_skipped``, ``errors``.
"""
from aco.lake.catalog import Catalog
local_catalog = Catalog()
schemas = local_catalog.schemas()
report: dict[str, list[str]] = {
"schemas_created": [],
"schemas_skipped": [],
"tables_created": [],
"errors": [],
}
# Get existing schemas
existing_schemas: set[str] = set()
if not dry_run:
try:
existing_schemas = {s.name for s in client.list_schemas(catalog_name)}
except Exception as e:
report["errors"].append(f"Failed to list existing schemas: {e}")
return report
for schema_name in schemas:
table_refs = local_catalog.tables(schema_name)
# ── Create schema if needed ──────────────────────────────
if schema_name in existing_schemas and skip_existing:
report["schemas_skipped"].append(schema_name)
elif dry_run:
print(f" would create schema {catalog_name}.{schema_name}")
else:
try:
client.create_schema(
catalog_name=catalog_name,
schema_name=schema_name,
comment=f"aco.table ({len(table_refs)} tables)",
)
report["schemas_created"].append(schema_name)
print(f" created schema {catalog_name}.{schema_name}")
except Exception as e:
report["errors"].append(f"schema {catalog_name}.{schema_name}: {e}")
print(f" ERROR schema {catalog_name}.{schema_name}: {e}")
continue
# ── Create tables via SQL DDL ────────────────────────────
for table_ref in table_refs:
_, table_name = table_ref.split(".", 1)
try:
model = local_catalog.model(table_ref)
col_count = len(model.model_fields)
except Exception as e:
report["errors"].append(f"model {table_ref}: {e}")
continue
ddl = _model_to_ddl(catalog_name, schema_name, table_name, model)
if dry_run:
print(
f" would create table "
f"{catalog_name}.{schema_name}.{table_name}"
f" ({col_count} cols)"
)
continue
try:
resp = client._ws.statement_execution.execute_statement(
warehouse_id=warehouse_id,
statement=ddl,
wait_timeout="50s",
)
state = str(resp.status.state)
if "SUCCEEDED" in state:
report["tables_created"].append(table_ref)
print(
f" created table "
f"{catalog_name}.{schema_name}.{table_name}"
f" ({col_count} cols)"
)
else:
err_msg = ""
if resp.status.error:
err_msg = str(resp.status.error.message)
report["errors"].append(f"table {table_ref}: {state} {err_msg}")
print(f" ERROR table {table_ref}: {state} {err_msg}")
except Exception as e:
report["errors"].append(f"table {table_ref}: {e}")
print(f" ERROR table {table_ref}: {e}")
return report

View File

@@ -8,8 +8,7 @@ from . import hcc_suspecting as hcc_suspecting
from . import input_layer as input_layer from . import input_layer as input_layer
from . import main as main from . import main as main
from . import pharmacy as pharmacy from . import pharmacy as pharmacy
from . import provider_attribution as provider_attribution
from . import quality_measures as quality_measures from . import quality_measures as quality_measures
from . import readmissions as readmissions from . import readmissions as readmissions
from . import provider_attribution as provider_attribution
from .base import Pipeline as Pipeline from .base import Pipeline as Pipeline
from .base import Step as Step

Some files were not shown because too many files have changed in this diff Show More