update sql_generator notebook: schema-only mode, all 12 pipelines

- Switch from DuckDB file to schema-only transpilation (no data needed)
- Add cclf pipeline (previously excluded due to transpile errors)
- Include cclf in summary table
- Remove try/except wrapper and hardcoded DB path
This commit is contained in:
kert
2026-03-11 15:06:36 -04:00
parent 16e006fb5e
commit aa9840ae1d

View File

@@ -20,21 +20,17 @@ def _(mo):
narwhals express functions against DuckDB relations and extracting the SQL
via `.sql_query()`, then transpiling with sqlglot.
Each cell below transpiles one pipeline module and displays the generated
`CREATE OR REPLACE TABLE` statements.
Uses schema-only DuckDB (no data file needed). Each cell below transpiles
one pipeline module and displays the generated `CREATE OR REPLACE TABLE`
statements.
""")
return
@app.cell(hide_code=True)
def _():
import duckdb
from aco.lake.transpile import transpile
DB_PATH = "/home/kert/notebooks/aco.duckdb"
con = duckdb.connect(DB_PATH, read_only=True)
DIALECT = "databricks"
CATALOG = "homelab"
@@ -42,7 +38,6 @@ def _():
"""Transpile a pipeline and return formatted SQL."""
result = transpile(
pipeline,
con,
target_dialect=DIALECT,
catalog=CATALOG,
output_mode="ctas",
@@ -56,7 +51,7 @@ def _():
header = f"/* {label}: {len(result)} steps, {errors} errors */\n"
return header + "\n\n".join(parts), len(result), errors
return con, run_transpile
return (run_transpile,)
@app.cell(hide_code=True)
@@ -241,13 +236,8 @@ def _(mo):
def _(mo, run_transpile):
from aco.pipe import cclf as _cclf
try:
_sql, _n, _e = run_transpile(_cclf.pipeline, "cclf")
_msg = f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```"
except Exception as _ex:
_msg = f"**CCLF pipeline requires CCLF data in DuckDB.**\n\nError: `{_ex}`"
mo.md(_msg)
_sql, _n, _e = run_transpile(_cclf.pipeline, "cclf")
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
return
@@ -278,10 +268,10 @@ def _(mo):
@app.cell
def _(con, mo):
from aco.lake.transpile import transpile as _transpile
def _(mo, run_transpile):
from aco.pipe import (
ahrq_measures,
cclf,
claims_preprocessing,
core,
data_quality,
@@ -305,6 +295,7 @@ def _(con, mo):
("quality_measures", quality_measures),
("hcc_suspecting", hcc_suspecting),
("provider_attribution", provider_attribution),
("cclf", cclf),
("main", main),
]
@@ -312,13 +303,10 @@ def _(con, mo):
_total_steps = 0
_total_errors = 0
for _name, _mod in _modules:
_result = _transpile(
_mod.pipeline, con, target_dialect="databricks", catalog="homelab"
)
_errs = sum(1 for v in _result.values() if v.startswith("-- ERROR"))
_total_steps += len(_result)
_total_errors += _errs
_rows.append(f"| {_name} | {len(_result)} | {_errs} |")
_, _n, _e = run_transpile(_mod.pipeline, _name)
_total_steps += _n
_total_errors += _e
_rows.append(f"| {_name} | {_n} | {_e} |")
_table = "| Pipeline | Steps | Errors |\n|----------|-------|--------|\n"
_table += "\n".join(_rows)