update sql_generator notebook: schema-only mode, all 12 pipelines
- Switch from DuckDB file to schema-only transpilation (no data needed) - Add cclf pipeline (previously excluded due to transpile errors) - Include cclf in summary table - Remove try/except wrapper and hardcoded DB path
This commit is contained in:
@@ -20,21 +20,17 @@ def _(mo):
|
||||
narwhals express functions against DuckDB relations and extracting the SQL
|
||||
via `.sql_query()`, then transpiling with sqlglot.
|
||||
|
||||
Each cell below transpiles one pipeline module and displays the generated
|
||||
`CREATE OR REPLACE TABLE` statements.
|
||||
Uses schema-only DuckDB (no data file needed). Each cell below transpiles
|
||||
one pipeline module and displays the generated `CREATE OR REPLACE TABLE`
|
||||
statements.
|
||||
""")
|
||||
return
|
||||
|
||||
|
||||
@app.cell(hide_code=True)
|
||||
def _():
|
||||
import duckdb
|
||||
|
||||
from aco.lake.transpile import transpile
|
||||
|
||||
DB_PATH = "/home/kert/notebooks/aco.duckdb"
|
||||
con = duckdb.connect(DB_PATH, read_only=True)
|
||||
|
||||
DIALECT = "databricks"
|
||||
CATALOG = "homelab"
|
||||
|
||||
@@ -42,7 +38,6 @@ def _():
|
||||
"""Transpile a pipeline and return formatted SQL."""
|
||||
result = transpile(
|
||||
pipeline,
|
||||
con,
|
||||
target_dialect=DIALECT,
|
||||
catalog=CATALOG,
|
||||
output_mode="ctas",
|
||||
@@ -56,7 +51,7 @@ def _():
|
||||
header = f"/* {label}: {len(result)} steps, {errors} errors */\n"
|
||||
return header + "\n\n".join(parts), len(result), errors
|
||||
|
||||
return con, run_transpile
|
||||
return (run_transpile,)
|
||||
|
||||
|
||||
@app.cell(hide_code=True)
|
||||
@@ -241,13 +236,8 @@ def _(mo):
|
||||
def _(mo, run_transpile):
|
||||
from aco.pipe import cclf as _cclf
|
||||
|
||||
try:
|
||||
_sql, _n, _e = run_transpile(_cclf.pipeline, "cclf")
|
||||
_msg = f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```"
|
||||
except Exception as _ex:
|
||||
_msg = f"**CCLF pipeline requires CCLF data in DuckDB.**\n\nError: `{_ex}`"
|
||||
|
||||
mo.md(_msg)
|
||||
_sql, _n, _e = run_transpile(_cclf.pipeline, "cclf")
|
||||
mo.md(f"**{_n} steps, {_e} errors**\n\n```sql\n{_sql}\n```")
|
||||
return
|
||||
|
||||
|
||||
@@ -278,10 +268,10 @@ def _(mo):
|
||||
|
||||
|
||||
@app.cell
|
||||
def _(con, mo):
|
||||
from aco.lake.transpile import transpile as _transpile
|
||||
def _(mo, run_transpile):
|
||||
from aco.pipe import (
|
||||
ahrq_measures,
|
||||
cclf,
|
||||
claims_preprocessing,
|
||||
core,
|
||||
data_quality,
|
||||
@@ -305,6 +295,7 @@ def _(con, mo):
|
||||
("quality_measures", quality_measures),
|
||||
("hcc_suspecting", hcc_suspecting),
|
||||
("provider_attribution", provider_attribution),
|
||||
("cclf", cclf),
|
||||
("main", main),
|
||||
]
|
||||
|
||||
@@ -312,13 +303,10 @@ def _(con, mo):
|
||||
_total_steps = 0
|
||||
_total_errors = 0
|
||||
for _name, _mod in _modules:
|
||||
_result = _transpile(
|
||||
_mod.pipeline, con, target_dialect="databricks", catalog="homelab"
|
||||
)
|
||||
_errs = sum(1 for v in _result.values() if v.startswith("-- ERROR"))
|
||||
_total_steps += len(_result)
|
||||
_total_errors += _errs
|
||||
_rows.append(f"| {_name} | {len(_result)} | {_errs} |")
|
||||
_, _n, _e = run_transpile(_mod.pipeline, _name)
|
||||
_total_steps += _n
|
||||
_total_errors += _e
|
||||
_rows.append(f"| {_name} | {_n} | {_e} |")
|
||||
|
||||
_table = "| Pipeline | Steps | Errors |\n|----------|-------|--------|\n"
|
||||
_table += "\n".join(_rows)
|
||||
|
||||
Reference in New Issue
Block a user