data: geographic, setting, anomaly scoring, enforcement correlation (refs #240, #241, #242, #243)

Geographic analysis (skin_subs.geographic_summary, mac_summary):
- 16 states, 8 MAC jurisdictions
- FL highest spend/bene ($2,142), First Coast MAC #1 overall
- Palmetto states (TN, VA, NC) have highest per-bene spend

Setting analysis (skin_subs.setting_analysis):
- Office setting: 68% of product spend, lowest oversight
- Podiatry×office: 45% of total spend — wound care mill pattern
- 169 setting×specialty×product combinations

Anomaly scoring (skin_subs.anomaly_scores):
- Composite risk: volume z-score, intensity z-score, setting risk, geo risk
- 7 critical, 17 high, 21 moderate, 155 low risk providers
- Top risk: TX podiatrist, $45K, office, vol_z=+5.69 — Jenson pattern

Enforcement correlation (skin_subs.provider_risk_profile):
- 23 providers match "jenson-like" pattern (TX podiatry, office)
- Benford chi² analysis (no synthetic data anomalies expected)
- Fraud-pattern matched providers slightly higher volume z-scores

17 tables in skin_subs schema.
This commit is contained in:
kert
2026-03-25 15:02:05 -04:00
parent e5ec7a22a1
commit 88025d5976
2 changed files with 548 additions and 0 deletions

View File

@@ -0,0 +1,310 @@
"""Build anomaly scoring and provider risk profiles for skin substitutes.
Addresses #242 (predatory pattern detection) and #243 (enforcement
action correlation).
Usage:
uv run python dev/scripts/build_skin_subs_anomaly.py
"""
from __future__ import annotations
import math
from pathlib import Path
import duckdb
ROOT = Path(__file__).resolve().parents[2]
DUCKDB_PATH = ROOT / "data" / "aco.duckdb"
# Known enforcement cases — providers we can match against
KNOWN_FRAUD_PATTERNS = {
"jenson": {
"specialty": "podiatry",
"state": "TX",
"setting": "office",
"description": "USA v. Jenson — $90M, S.D. Tex., podiatry clinic",
},
"gehrke": {
"specialty": "nurse_practitioner",
"state": "AZ",
"setting": "snf",
"description": "USA v. Gehrke/King — $1.2B, D. Ariz., mobile wound care",
},
"vohra": {
"specialty": "other",
"state": "FL",
"setting": "snf",
"description": "Vohra Wound Physicians — $45M, S.D. Fla., EMR upcoding",
},
}
# Setting risk weights (higher = more fraud-prone based on OIG findings)
SETTING_RISK = {
"office": 1.5, # Lowest oversight, wound care mills
"snf": 1.3, # Kickback-prone, captive patients
"asc": 0.8, # Moderate oversight
"hopd": 0.5, # Institutional controls, auditable
}
def build_anomaly_scores(con: duckdb.DuckDBPyConnection) -> None:
"""#242 — Composite anomaly scoring per provider."""
print("\n=== Anomaly Scoring (#242) ===")
con.execute("DROP TABLE IF EXISTS skin_subs.anomaly_scores")
con.execute(f"""
CREATE TABLE skin_subs.anomaly_scores AS
WITH provider_stats AS (
SELECT
p.rendering_npi,
p.provider_specialty,
p.state,
p.total_paid,
p.unique_patients,
p.paid_per_patient,
p.total_claim_lines,
p.unique_products,
-- Dominant setting for this provider
(SELECT place_of_service_description
FROM skin_subs.claims_synthetic c
WHERE c.rendering_npi = p.rendering_npi
GROUP BY place_of_service_description
ORDER BY sum(paid_amount) DESC LIMIT 1
) AS primary_setting,
-- Avg units per product claim
(SELECT round(avg(units), 1)
FROM skin_subs.claims_synthetic c
WHERE c.rendering_npi = p.rendering_npi
AND c.claim_type = 'product'
) AS avg_units
FROM skin_subs.providers p
),
peer_stats AS (
SELECT
provider_specialty,
avg(total_paid) AS peer_avg_paid,
stddev(total_paid) AS peer_std_paid,
avg(paid_per_patient) AS peer_avg_per_patient,
stddev(paid_per_patient) AS peer_std_per_patient,
avg(unique_patients) AS peer_avg_patients,
stddev(unique_patients) AS peer_std_patients,
avg(avg_units) AS peer_avg_units,
stddev(avg_units) AS peer_std_units
FROM provider_stats
GROUP BY provider_specialty
)
SELECT
ps.rendering_npi,
ps.provider_specialty,
ps.state,
ps.primary_setting,
ps.total_paid,
ps.unique_patients,
ps.paid_per_patient,
ps.total_claim_lines,
ps.unique_products,
ps.avg_units,
-- Z-scores vs specialty peers
CASE WHEN peer.peer_std_paid > 0
THEN round((ps.total_paid - peer.peer_avg_paid)
/ peer.peer_std_paid, 2)
ELSE 0 END AS volume_zscore,
CASE WHEN peer.peer_std_per_patient > 0
THEN round((ps.paid_per_patient - peer.peer_avg_per_patient)
/ peer.peer_std_per_patient, 2)
ELSE 0 END AS intensity_zscore,
CASE WHEN peer.peer_std_units > 0
THEN round((ps.avg_units - peer.peer_avg_units)
/ peer.peer_std_units, 2)
ELSE 0 END AS units_zscore,
-- Setting risk weight
CASE ps.primary_setting
WHEN 'office' THEN {SETTING_RISK['office']}
WHEN 'snf' THEN {SETTING_RISK['snf']}
WHEN 'asc' THEN {SETTING_RISK['asc']}
WHEN 'hopd' THEN {SETTING_RISK['hopd']}
ELSE 1.0 END AS setting_risk,
-- Geographic risk (state per-bene spend z-score)
(SELECT round((g.paid_per_bene - agg.m) / nullif(agg.s, 0), 2)
FROM skin_subs.geographic_summary g,
(SELECT avg(paid_per_bene) AS m, stddev(paid_per_bene) AS s
FROM skin_subs.geographic_summary) agg
WHERE g.state = ps.state
LIMIT 1
) AS geo_risk_zscore
FROM provider_stats ps
LEFT JOIN peer_stats peer ON ps.provider_specialty = peer.provider_specialty
ORDER BY volume_zscore DESC
""")
# Add composite risk score
con.execute("""
ALTER TABLE skin_subs.anomaly_scores
ADD COLUMN composite_risk DOUBLE;
UPDATE skin_subs.anomaly_scores
SET composite_risk = round(
(GREATEST(volume_zscore, 0) * 0.3
+ GREATEST(intensity_zscore, 0) * 0.3
+ GREATEST(units_zscore, 0) * 0.1
+ setting_risk * 0.2
+ GREATEST(COALESCE(geo_risk_zscore, 0), 0) * 0.1
), 2);
""")
# Add risk tier
con.execute("""
ALTER TABLE skin_subs.anomaly_scores
ADD COLUMN risk_tier VARCHAR;
UPDATE skin_subs.anomaly_scores
SET risk_tier = CASE
WHEN composite_risk >= 2.0 THEN 'critical'
WHEN composite_risk >= 1.0 THEN 'high'
WHEN composite_risk >= 0.5 THEN 'moderate'
ELSE 'low' END;
""")
count = con.execute("SELECT count(*) FROM skin_subs.anomaly_scores").fetchone()[0]
print(f" Providers scored: {count}")
# Tier distribution
print("\n Risk tier distribution:")
for r in con.execute("""
SELECT risk_tier, count(*) as n,
round(avg(total_paid), 2) as avg_paid,
round(avg(composite_risk), 2) as avg_score
FROM skin_subs.anomaly_scores
GROUP BY risk_tier ORDER BY avg_score DESC
""").fetchall():
print(f" {r[0]:10s} n={r[1]:3d} avg_paid=${r[2]:>10,.2f} avg_score={r[3]}")
# Top risk providers
print("\n Top 10 highest-risk providers:")
for r in con.execute("""
SELECT rendering_npi, provider_specialty, state, primary_setting,
total_paid, paid_per_patient, composite_risk, risk_tier,
volume_zscore, intensity_zscore
FROM skin_subs.anomaly_scores
ORDER BY composite_risk DESC LIMIT 10
""").fetchall():
print(f" NPI {r[0]} {r[1]:15s} {r[2]} {r[3]:8s} "
f"paid=${r[4]:>10,.2f} per_pt=${r[5]:>8,.2f} "
f"risk={r[6]:5.2f} ({r[7]}) vol_z={r[8]:+5.2f} int_z={r[9]:+5.2f}")
def build_enforcement_correlation(con: duckdb.DuckDBPyConnection) -> None:
"""#243 — Correlate anomaly scores with known fraud patterns."""
print("\n=== Enforcement Correlation (#243) ===")
con.execute("DROP TABLE IF EXISTS skin_subs.provider_risk_profile")
con.execute("""
CREATE TABLE skin_subs.provider_risk_profile AS
SELECT
a.*,
-- Match against known fraud patterns
CASE
WHEN a.provider_specialty = 'podiatry'
AND a.state = 'TX'
AND a.primary_setting = 'office'
THEN 'jenson-like'
WHEN a.primary_setting = 'snf'
AND a.provider_specialty IN ('nurse_practitioner', 'other')
THEN 'gehrke-like'
WHEN a.primary_setting = 'snf'
AND a.state = 'FL'
THEN 'vohra-like'
ELSE NULL
END AS fraud_pattern_match,
-- Benford analysis: first-digit distribution of paid amounts
-- (precomputed per provider from claims)
(SELECT round(
-- Chi-squared statistic for Benford's law
sum(power(observed - expected, 2) / expected), 4)
FROM (
SELECT
digit,
count(*) * 1.0 / nullif(total, 0) AS observed,
log10(1.0 + 1.0/digit) AS expected
FROM (
SELECT
CAST(substr(CAST(CAST(round(abs(paid_amount)) AS INTEGER) AS VARCHAR), 1, 1) AS INTEGER) AS digit,
count(*) OVER () AS total
FROM skin_subs.claims_synthetic
WHERE rendering_npi = a.rendering_npi
AND paid_amount > 0
) digits
WHERE digit BETWEEN 1 AND 9
GROUP BY digit, total
) benford
) AS benford_chi2
FROM skin_subs.anomaly_scores a
ORDER BY composite_risk DESC
""")
count = con.execute(
"SELECT count(*) FROM skin_subs.provider_risk_profile"
).fetchone()[0]
print(f" Risk profiles: {count}")
# Fraud pattern matches
print("\n Fraud pattern matches:")
for r in con.execute("""
SELECT fraud_pattern_match, count(*) as n,
round(avg(composite_risk), 2) as avg_risk,
round(avg(total_paid), 2) as avg_paid
FROM skin_subs.provider_risk_profile
WHERE fraud_pattern_match IS NOT NULL
GROUP BY fraud_pattern_match ORDER BY avg_risk DESC
""").fetchall():
print(f" {r[0]:15s} n={r[1]:3d} avg_risk={r[2]:5.2f} avg_paid=${r[3]:>10,.2f}")
# Do fraud-pattern providers score higher than peers?
print("\n Risk scores: fraud-pattern vs non-match:")
for r in con.execute("""
SELECT
CASE WHEN fraud_pattern_match IS NOT NULL THEN 'matched'
ELSE 'unmatched' END AS group_name,
count(*) as n,
round(avg(composite_risk), 3) as avg_risk,
round(avg(volume_zscore), 3) as avg_vol_z,
round(avg(intensity_zscore), 3) as avg_int_z
FROM skin_subs.provider_risk_profile
GROUP BY group_name
""").fetchall():
print(f" {r[0]:10s} n={r[1]:3d} risk={r[2]:6.3f} vol_z={r[3]:+6.3f} int_z={r[4]:+6.3f}")
# Benford analysis summary
print("\n Benford's law chi² (lower = more conformant, >15.5 = suspicious at p<0.05):")
for r in con.execute("""
SELECT risk_tier, count(*) as n,
round(avg(benford_chi2), 4) as avg_chi2,
count(*) FILTER (WHERE benford_chi2 > 15.5) as suspicious
FROM skin_subs.provider_risk_profile
GROUP BY risk_tier ORDER BY avg_chi2 DESC
""").fetchall():
print(f" {r[0]:10s} n={r[1]:3d} avg_chi²={r[2]:8.4f} suspicious={r[3]}")
def main() -> None:
print("Building anomaly scoring and risk profiles ...")
con = duckdb.connect(str(DUCKDB_PATH))
build_anomaly_scores(con)
build_enforcement_correlation(con)
print("\n=== Final table count ===")
for r in con.execute("""
SELECT table_name FROM information_schema.tables
WHERE table_schema = 'skin_subs' ORDER BY table_name
""").fetchall():
cnt = con.execute(f"SELECT count(*) FROM skin_subs.{r[0]}").fetchone()[0]
print(f" skin_subs.{r[0]:30s}: {cnt:>6} rows")
con.close()
print("\nDone.")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,238 @@
"""Build geographic and setting analysis tables for skin substitutes.
Addresses #240 (geographic analysis) and #241 (setting analysis).
Usage:
uv run python dev/scripts/build_skin_subs_geo_setting.py
"""
from __future__ import annotations
from pathlib import Path
import duckdb
ROOT = Path(__file__).resolve().parents[2]
DUCKDB_PATH = ROOT / "data" / "aco.duckdb"
# MAC jurisdiction → states mapping (approximate, some states split)
MAC_JURISDICTIONS = {
"Novitas (JH/JL)": ["AR", "CO", "LA", "MS", "NM", "OK", "TX"],
"First Coast (JN)": ["FL"],
"Palmetto (JJ/JM)": ["AL", "GA", "NC", "SC", "TN", "VA", "WV"],
"CGS (J15)": ["KY", "OH"],
"WPS (J5/J8)": ["IA", "IN", "KS", "MI", "MO", "NE"],
"NGS (J6/JK)": ["CT", "IL", "MA", "ME", "MN", "NH", "NY", "RI", "VT", "WI"],
"Noridian (JE/JF)": ["AK", "AZ", "CA", "HI", "ID", "MT", "ND", "NV", "OR", "SD", "UT", "WA", "WY"],
}
# Invert: state → MAC
STATE_TO_MAC = {}
for mac, states in MAC_JURISDICTIONS.items():
for st in states:
STATE_TO_MAC[st] = mac
def build_geographic_summary(con: duckdb.DuckDBPyConnection) -> None:
"""#240 — State and MAC jurisdiction analysis."""
print("\n=== Geographic Analysis (#240) ===")
con.execute("DROP TABLE IF EXISTS skin_subs.geographic_summary")
con.execute("""
CREATE TABLE skin_subs.geographic_summary AS
SELECT
state,
count(*) AS claim_lines,
count(*) FILTER (WHERE claim_type = 'product') AS product_lines,
count(DISTINCT person_id) AS unique_benes,
count(DISTINCT rendering_npi) AS unique_providers,
round(sum(paid_amount), 2) AS total_paid,
round(avg(paid_amount), 2) AS avg_paid_per_line,
round(sum(paid_amount) / nullif(count(DISTINCT person_id), 0), 2)
AS paid_per_bene,
round(sum(paid_amount) / nullif(count(DISTINCT rendering_npi), 0), 2)
AS paid_per_provider,
count(DISTINCT hcpcs_code) AS product_variety,
-- Setting mix
round(count(*) FILTER (WHERE place_of_service = '11') * 100.0
/ count(*), 1) AS pct_office,
round(count(*) FILTER (WHERE place_of_service = '22') * 100.0
/ count(*), 1) AS pct_hopd,
round(count(*) FILTER (WHERE place_of_service = '24') * 100.0
/ count(*), 1) AS pct_asc,
round(count(*) FILTER (WHERE place_of_service IN ('31','32')) * 100.0
/ count(*), 1) AS pct_snf,
-- Specialty mix
round(count(*) FILTER (WHERE provider_specialty = 'podiatry') * 100.0
/ count(*), 1) AS pct_podiatry,
round(count(*) FILTER (WHERE provider_specialty = 'dermatology') * 100.0
/ count(*), 1) AS pct_dermatology
FROM skin_subs.claims_synthetic
GROUP BY state
ORDER BY total_paid DESC
""")
count = con.execute(
"SELECT count(*) FROM skin_subs.geographic_summary"
).fetchone()[0]
print(f" States: {count}")
# Add MAC jurisdiction column
mac_cases = " ".join(
f"WHEN state = '{st}' THEN '{mac}'"
for st, mac in STATE_TO_MAC.items()
)
con.execute(f"""
ALTER TABLE skin_subs.geographic_summary
ADD COLUMN mac_jurisdiction VARCHAR;
UPDATE skin_subs.geographic_summary
SET mac_jurisdiction = CASE {mac_cases} ELSE 'Other' END;
""")
# MAC-level rollup
con.execute("DROP TABLE IF EXISTS skin_subs.mac_summary")
con.execute("""
CREATE TABLE skin_subs.mac_summary AS
SELECT
mac_jurisdiction,
count(DISTINCT state) AS states,
sum(claim_lines) AS claim_lines,
sum(unique_benes) AS unique_benes,
sum(unique_providers) AS unique_providers,
round(sum(total_paid), 2) AS total_paid,
round(avg(paid_per_bene), 2) AS avg_paid_per_bene,
round(avg(pct_office), 1) AS avg_pct_office,
round(avg(pct_podiatry), 1) AS avg_pct_podiatry
FROM skin_subs.geographic_summary
GROUP BY mac_jurisdiction
ORDER BY total_paid DESC
""")
print("\n By MAC jurisdiction:")
for r in con.execute("SELECT * FROM skin_subs.mac_summary").fetchall():
print(f" {r[0]:25s} states={r[1]:2d} lines={r[2]:5d} "
f"paid=${r[5]:>10,.2f} per_bene=${r[6]:>8,.2f} "
f"office={r[7]}% podiatry={r[8]}%")
print("\n Top 5 states by spend per beneficiary:")
for r in con.execute("""
SELECT state, mac_jurisdiction, unique_benes, total_paid,
paid_per_bene, pct_office, pct_podiatry
FROM skin_subs.geographic_summary
ORDER BY paid_per_bene DESC LIMIT 5
""").fetchall():
print(f" {r[0]} {r[1]:25s} benes={r[2]:4d} "
f"per_bene=${r[4]:>8,.2f} office={r[5]}% podiatry={r[6]}%")
def build_setting_analysis(con: duckdb.DuckDBPyConnection) -> None:
"""#241 — Care setting analysis."""
print("\n=== Setting Analysis (#241) ===")
con.execute("DROP TABLE IF EXISTS skin_subs.setting_analysis")
con.execute("""
CREATE TABLE skin_subs.setting_analysis AS
WITH setting_product AS (
SELECT
place_of_service_description AS setting,
provider_specialty,
hcpcs_code,
count(*) AS claim_lines,
count(DISTINCT person_id) AS unique_benes,
count(DISTINCT rendering_npi) AS unique_providers,
round(sum(paid_amount), 2) AS total_paid,
round(avg(paid_amount), 2) AS avg_paid,
round(avg(units), 1) AS avg_units,
round(sum(paid_amount) / nullif(count(DISTINCT person_id), 0), 2)
AS paid_per_bene,
-- Provider concentration (HHI proxy)
round(sum(paid_amount) / nullif(
count(DISTINCT rendering_npi), 0), 2) AS paid_per_provider
FROM skin_subs.claims_synthetic
WHERE claim_type = 'product'
GROUP BY setting, provider_specialty, hcpcs_code
)
SELECT
setting,
provider_specialty,
sp.hcpcs_code,
h.product_name,
h.manufacturer,
h.category,
claim_lines,
unique_benes,
unique_providers,
total_paid,
avg_paid,
avg_units,
paid_per_bene,
paid_per_provider,
-- Flag: high per-provider concentration
CASE WHEN paid_per_provider > 10000 THEN true
ELSE false END AS high_concentration
FROM setting_product sp
LEFT JOIN skin_subs.hcpcs_universe h ON sp.hcpcs_code = h.hcpcs_code
ORDER BY total_paid DESC
""")
count = con.execute(
"SELECT count(*) FROM skin_subs.setting_analysis"
).fetchone()[0]
print(f" Setting×specialty×product combinations: {count}")
# Setting summary
print("\n By setting:")
for r in con.execute("""
SELECT setting, sum(claim_lines) as lines,
sum(unique_providers) as provs,
round(sum(total_paid), 2) as paid,
round(avg(avg_units), 1) as avg_units,
round(avg(paid_per_bene), 2) as per_bene
FROM skin_subs.setting_analysis
GROUP BY setting ORDER BY paid DESC
""").fetchall():
print(f" {r[0]:15s} lines={r[1]:5d} provs={r[2]:4d} "
f"paid=${r[3]:>12,.2f} units={r[4]:4.1f} per_bene=${r[5]:>8,.2f}")
# Specialty×setting cross-tab
print("\n Specialty × setting (total paid):")
for r in con.execute("""
SELECT provider_specialty, setting,
round(sum(total_paid), 2) as paid,
sum(claim_lines) as lines
FROM skin_subs.setting_analysis
GROUP BY provider_specialty, setting
ORDER BY paid DESC LIMIT 10
""").fetchall():
print(f" {r[0]:20s} {r[1]:10s} lines={r[3]:5d} paid=${r[2]:>10,.2f}")
# High-concentration flag
high_conc = con.execute("""
SELECT count(DISTINCT provider_specialty || setting || hcpcs_code)
FROM skin_subs.setting_analysis WHERE high_concentration
""").fetchone()[0]
print(f"\n High-concentration combos (>$10K/provider): {high_conc}")
def main() -> None:
print("Building geographic and setting analysis ...")
con = duckdb.connect(str(DUCKDB_PATH))
build_geographic_summary(con)
build_setting_analysis(con)
# Updated table inventory
print("\n=== skin_subs tables ===")
for r in con.execute("""
SELECT table_name FROM information_schema.tables
WHERE table_schema = 'skin_subs' ORDER BY table_name
""").fetchall():
cnt = con.execute(f"SELECT count(*) FROM skin_subs.{r[0]}").fetchone()[0]
print(f" skin_subs.{r[0]:30s}: {cnt:>6} rows")
con.close()
print("\nDone.")
if __name__ == "__main__":
main()