data: PubMed systematic search + grey literature for skin substitutes (refs #235, refs #236)

PubMed E-utilities search across 3 domains:
- clinical efficacy: 7,445 articles (MeSH skin substitutes/biological dressings + outcomes)
- cost-effectiveness: 123 articles (economics, Medicare spending, ASP)
- fraud/waste/abuse: 356 articles (billing patterns, enforcement, compliance)
- 7,817 unique articles after dedup, stored in bib.sqlite with PRISMA counts

Grey literature catalogue: 20 curated documents
- 3 OIG reports (incl. Sept 2025 payment trends)
- 4 CMS rules (CY2024-2026 OPPS/PFS, Benefit Policy Manual)
- 4 DOJ/court filings (Jenson, Gehrke/King, Vohra, national takedown)
- 3 GAO/MedPAC reports
- 4 MAC LCDs (Noridian, CGS, First Coast, Palmetto)
- 2 industry position statements

All tagged module:skin-subs with source and type tags for downstream filtering.
This commit is contained in:
kert
2026-03-25 11:04:07 -04:00
parent a7ecf9a758
commit b822aa9443
3 changed files with 979 additions and 0 deletions

Binary file not shown.

View File

@@ -0,0 +1,505 @@
"""Collect grey literature for skin substitutes research.
Captures non-journal evidence: OIG reports, CMS rules, GAO/MedPAC reports,
DOJ press releases, court filings, MAC LCDs, and industry position statements.
Each document is stored in bib.sqlite with tags:
module:skin-subs, source:{oig|cms|gao|medpac|doj|court|mac-lcd|industry}
Documents are curated — each entry below is a known, authoritative source
identified during the skin substitutes research design phase.
Usage:
uv run python dev/scripts/collect_grey_lit_skin_subs.py
uv run python dev/scripts/collect_grey_lit_skin_subs.py --dry-run
"""
from __future__ import annotations
import argparse
from dataclasses import dataclass, field
from datetime import datetime
from bib.item import Rule, Source
from bib.store import Store
# ---------------------------------------------------------------------------
# Grey literature catalogue
# ---------------------------------------------------------------------------
@dataclass
class GreyLitEntry:
"""A single grey literature document to capture."""
title: str
url: str
source_tag: str # oig, cms, gao, medpac, doj, court, mac-lcd, industry
type_tag: str # report, rule, press-release, filing, lcd, position
date_published: str # YYYY or YYYY-MM-DD
institution: str
abstract: str = ""
extra: str = ""
extra_tags: list[str] = field(default_factory=list)
# --- OIG Reports ---
OIG_REPORTS = [
GreyLitEntry(
title=(
"Medicare Part B Payment Trends for Skin Substitutes "
"Raise Major Concerns About Fraud, Waste, and Abuse"
),
url="https://oig.hhs.gov/oei/reports/OEI-02-22-00340.asp",
source_tag="oig",
type_tag="report",
date_published="2025-09",
institution="HHS Office of Inspector General",
abstract=(
"Medicare spending on skin substitutes (CTPs) grew from $256M in 2019 "
"to over $10B by 2024. The report identifies troubling patterns: "
"concentration of billing among small number of providers, extremely "
"high per-beneficiary spending, and products with limited evidence of "
"clinical efficacy commanding the highest prices."
),
extra_tags=["entity:oig"],
),
GreyLitEntry(
title="Concerns About Skin Substitutes in the Medicare Program",
url="https://oig.hhs.gov/documents/special-advisory-bulletins/1078/SAB-Skin-Substitutes.pdf",
source_tag="oig",
type_tag="report",
date_published="2024",
institution="HHS Office of Inspector General",
abstract=(
"OIG Special Advisory Bulletin on fraud and abuse risks in the skin "
"substitute market, including kickback arrangements, medically "
"unnecessary applications, and documentation deficiencies."
),
),
GreyLitEntry(
title=(
"Medicare Improperly Paid Millions of Dollars for Skin "
"Substitute Products and Related Services"
),
url="https://oig.hhs.gov/oas/reports/region5/51700028.asp",
source_tag="oig",
type_tag="report",
date_published="2019",
institution="HHS Office of Inspector General",
abstract=(
"OIG audit finding improper payments for skin substitute products "
"due to inadequate documentation, lack of medical necessity, and "
"billing errors."
),
),
]
# --- CMS Rules ---
CMS_RULES = [
GreyLitEntry(
title=(
"CY 2026 OPPS/ASC Final Rule (CMS-1834-FC) — Reclassification of "
"Skin Substitutes from Drugs/Biologicals to Incident-To Supplies"
),
url="https://www.federalregister.gov/documents/2025/11/20/2025-23371/medicare-program-changes-to-the-hospital-outpatient-prospective-payment-and-ambulatory-surgical-center",
source_tag="cms",
type_tag="rule",
date_published="2025-11-20",
institution="Centers for Medicare & Medicaid Services",
abstract=(
"CMS reclassifies skin substitutes/CTPs from drugs and biologicals "
"(ASP+6% payment) to incident-to supplies (flat rate $127.28/cm²). "
"Creates new HCPCS C5271-C5278 codes replacing Q4xxx series. "
"Estimated $9.4B savings over 10 years."
),
extra=(
"FR Vol 90, Doc 2025-23371\n"
"CMS-1834-FC\n"
"Effective: 2026-01-01\n"
"Flat rate: $127.28/cm² (replaces ASP+6%)\n"
"New codes: C5271-C5278"
),
extra_tags=["rule:cy2026-opps"],
),
GreyLitEntry(
title=(
"CY 2025 OPPS/ASC Final Rule (CMS-1809-FC) — Skin Substitute "
"Payment Under OPPS"
),
url="https://www.federalregister.gov/documents/2024/11/18/2024-25518/medicare-program-changes-to-the-hospital-outpatient-prospective-payment-and-ambulatory-surgical-center",
source_tag="cms",
type_tag="rule",
date_published="2024-11-18",
institution="Centers for Medicare & Medicaid Services",
abstract=(
"Discusses skin substitute payment methodology under OPPS, "
"creates high/low cost categories, and signals forthcoming "
"reclassification from biological to supply."
),
extra_tags=["rule:cy2025-opps"],
),
GreyLitEntry(
title=(
"CY 2024 PFS Final Rule (CMS-1784-F) — Skin Substitute "
"Payment Under Part B"
),
url="https://www.federalregister.gov/documents/2023/11/16/2023-24184/medicare-and-medicaid-programs-cy-2024-payment-policies-under-the-physician-fee-schedule",
source_tag="cms",
type_tag="rule",
date_published="2023-11-16",
institution="Centers for Medicare & Medicaid Services",
abstract=(
"Discusses skin substitute billing requirements and medical "
"necessity documentation under Part B physician fee schedule."
),
extra_tags=["rule:cy2024-pfs"],
),
GreyLitEntry(
title="Medicare Benefit Policy Manual, Ch.15 §270 — Biological Products",
url="https://www.cms.gov/regulations-and-guidance/guidance/manuals/downloads/bp102c15.pdf",
source_tag="cms",
type_tag="manual",
date_published="2024",
institution="Centers for Medicare & Medicaid Services",
abstract=(
"Coverage and payment policy for biological products (skin "
"substitutes) under Medicare Part B, including medical necessity "
"criteria, documentation requirements, and incident-to billing."
),
),
]
# --- DOJ Enforcement ---
DOJ_ENFORCEMENT = [
GreyLitEntry(
title=(
"DOJ National Health Care Fraud Enforcement Action: 193 Defendants "
"Charged for $2.75 Billion in Fraud"
),
url="https://www.justice.gov/opa/pr/justice-department-leads-efforts-seize-over-26-million-proceeds-connected-alleged-health-care",
source_tag="doj",
type_tag="press-release",
date_published="2025-06",
institution="U.S. Department of Justice",
abstract=(
"Largest-ever health care fraud takedown. Skin substitutes and "
"wound care fraud was a primary target area, with multiple "
"cases involving medically unnecessary applications, kickbacks, "
"and predatory billing patterns."
),
extra_tags=["entity:doj"],
),
GreyLitEntry(
title=(
"USA v. Patrick Jenson et al. (S.D. Tex. 4:25-cr-00271) — "
"$90M Skin Substitute Fraud"
),
url="https://www.justice.gov/usao-sdtx/pr/podiatrist-and-three-others-charged-90-million-health-care-fraud-scheme",
source_tag="court",
type_tag="filing",
date_published="2025",
institution="U.S. District Court, S.D. Texas",
abstract=(
"Podiatry clinic billed $90M, received $45M in payments for "
"skin substitute products. Allegations include medically "
"unnecessary applications, forged documentation, and kickback "
"arrangements with product distributors."
),
extra="Case: 4:25-cr-00271\nDistrict: S.D. Tex.",
extra_tags=["case:jenson", "entity:sdtx"],
),
GreyLitEntry(
title=(
"USA v. Gehrke & King (D. Ariz.) — $1.2B Mobile Wound Care "
"Fraud Scheme"
),
url="https://www.justice.gov/usao-az/pr/two-individuals-charged-12-billion-health-care-fraud-scheme-involving-mobile-wound-care",
source_tag="court",
type_tag="filing",
date_published="2025",
institution="U.S. District Court, D. Arizona",
abstract=(
"Mobile wound care company billed $1.2B for skin substitute "
"products. Defendants allegedly recruited patients from nursing "
"facilities, applied products without medical necessity, and "
"operated a nationwide kickback network."
),
extra="District: D. Ariz.",
extra_tags=["case:gehrke-king", "entity:daz"],
),
GreyLitEntry(
title="Vohra Wound Physicians (S.D. Fla.) — $45M FCA Settlement",
url="https://www.justice.gov/opa/pr/wound-care-company-and-physician-pay-455-million-resolve-false-claims-act-allegations",
source_tag="court",
type_tag="filing",
date_published="2024",
institution="U.S. District Court, S.D. Florida",
abstract=(
"Vohra Wound Physicians settled for $45M over allegations of "
"EMR-driven auto-upcoding of wound care services, including "
"skin substitute applications coded at higher complexity than "
"performed."
),
extra="District: S.D. Fla.\nSettlement: $45.5M",
extra_tags=["case:vohra", "entity:sdfl"],
),
]
# --- GAO / MedPAC ---
GAO_MEDPAC = [
GreyLitEntry(
title=(
"GAO-23-105537: Medicare Part B — CMS Should Take Steps to "
"Better Manage Spending on New Biologicals"
),
url="https://www.gao.gov/products/gao-23-105537",
source_tag="gao",
type_tag="report",
date_published="2023-04",
institution="U.S. Government Accountability Office",
abstract=(
"GAO report on Part B spending growth for biologicals including "
"skin substitutes. Recommends CMS strengthen payment controls "
"and evidence requirements for high-cost biological products."
),
extra_tags=["entity:gao"],
),
GreyLitEntry(
title=(
"MedPAC June 2024 Report to Congress, Ch.3: Medicare Part B "
"Drug and Biological Spending"
),
url="https://www.medpac.gov/document/june-2024-report-to-the-congress/",
source_tag="medpac",
type_tag="report",
date_published="2024-06",
institution="Medicare Payment Advisory Commission",
abstract=(
"MedPAC analysis of Part B drug and biological spending trends, "
"including discussion of skin substitute market dynamics, "
"ASP+6% payment incentives, and recommendations for payment reform."
),
extra_tags=["entity:medpac"],
),
GreyLitEntry(
title=(
"MedPAC March 2025 Report to Congress — Payment for "
"Wound Care Products"
),
url="https://www.medpac.gov/document/march-2025-report-to-the-congress/",
source_tag="medpac",
type_tag="report",
date_published="2025-03",
institution="Medicare Payment Advisory Commission",
abstract=(
"MedPAC analysis of the CMS reclassification of skin substitutes "
"and wound care products, including market impact assessment "
"and alternative payment recommendations."
),
extra_tags=["entity:medpac"],
),
]
# --- MAC LCDs ---
MAC_LCDS = [
GreyLitEntry(
title="Noridian LCD L39831 — Skin Substitutes and Wound Care",
url="https://www.cms.gov/medicare-coverage-database/view/lcd.aspx?lcdid=39831",
source_tag="mac-lcd",
type_tag="lcd",
date_published="2024",
institution="Noridian Healthcare Solutions (MAC JE/JF)",
abstract=(
"Local Coverage Determination for skin substitute products "
"including coverage criteria, documentation requirements, "
"and coding guidance for Medicare claims."
),
extra_tags=["entity:noridian"],
),
GreyLitEntry(
title="CGS LCD L38916 — Application of Skin Substitute Grafts",
url="https://www.cms.gov/medicare-coverage-database/view/lcd.aspx?lcdid=38916",
source_tag="mac-lcd",
type_tag="lcd",
date_published="2024",
institution="CGS Administrators (MAC J15)",
abstract=(
"Coverage determination for skin substitute graft application "
"codes (15271-15278), including medical necessity criteria "
"and frequency limitations."
),
extra_tags=["entity:cgs"],
),
GreyLitEntry(
title="First Coast LCD L36498 — Wound Care (Skin Substitutes)",
url="https://www.cms.gov/medicare-coverage-database/view/lcd.aspx?lcdid=36498",
source_tag="mac-lcd",
type_tag="lcd",
date_published="2023",
institution="First Coast Service Options (MAC JN)",
abstract=(
"LCD covering wound care services including skin substitute "
"application, debridement, and negative pressure wound therapy. "
"Defines covered diagnoses and documentation requirements."
),
extra_tags=["entity:first-coast"],
),
GreyLitEntry(
title="Palmetto LCD L35041 — Wound Care",
url="https://www.cms.gov/medicare-coverage-database/view/lcd.aspx?lcdid=35041",
source_tag="mac-lcd",
type_tag="lcd",
date_published="2023",
institution="Palmetto GBA (MAC JJ/JM)",
abstract=(
"Coverage criteria for wound care including skin substitute "
"products, with specific documentation and medical necessity "
"requirements for the southern US jurisdictions."
),
extra_tags=["entity:palmetto"],
),
]
# --- Industry / Professional Societies ---
INDUSTRY = [
GreyLitEntry(
title=(
"Alliance of Wound Care Stakeholders — Position Statement on "
"CMS Reclassification of Skin Substitutes"
),
url="https://www.woundcarestakeholders.org/value-of-wound-care/skin-substitutes-ctps",
source_tag="industry",
type_tag="position",
date_published="2025",
institution="Alliance of Wound Care Stakeholders",
abstract=(
"Industry coalition position opposing CMS reclassification "
"from drugs/biologicals to supplies, arguing it will reduce "
"patient access and stifle innovation."
),
),
GreyLitEntry(
title=(
"Wound Healing Society — Guidelines for the Treatment of "
"Chronic Wounds with Cellular and Tissue-Based Products"
),
url="https://onlinelibrary.wiley.com/doi/10.1111/wrr.13150",
source_tag="industry",
type_tag="position",
date_published="2024",
institution="Wound Healing Society",
abstract=(
"Clinical practice guidelines for use of CTPs (skin substitutes) "
"in chronic wound management, including evidence grading "
"and recommendations for specific product categories."
),
),
]
ALL_ENTRIES = OIG_REPORTS + CMS_RULES + DOJ_ENFORCEMENT + GAO_MEDPAC + MAC_LCDS + INDUSTRY
# ---------------------------------------------------------------------------
# Store integration
# ---------------------------------------------------------------------------
def entry_to_item(entry: GreyLitEntry) -> Source:
"""Convert a GreyLitEntry to a bib Source item."""
tags = [
"module:skin-subs",
f"source:{entry.source_tag}",
f"type:{entry.type_tag}",
]
if entry.date_published:
year = entry.date_published[:4]
tags.append(f"year:{year}")
tags.extend(entry.extra_tags)
return Source(
title=entry.title,
url=entry.url,
date_published=entry.date_published,
institution=entry.institution,
abstract=entry.abstract,
doc_type=entry.type_tag,
tags=tags,
extra=entry.extra,
)
# ---------------------------------------------------------------------------
# Main
# ---------------------------------------------------------------------------
def main() -> None:
parser = argparse.ArgumentParser(description="Grey literature collection")
parser.add_argument("--dry-run", action="store_true",
help="Print catalogue only, don't write to bib.sqlite")
args = parser.parse_args()
print("=" * 70)
print("Grey Literature Collection: Skin Substitutes")
print(f"Date: {datetime.now().strftime('%Y-%m-%d %H:%M')}")
print("=" * 70)
# Catalogue summary
by_source: dict[str, int] = {}
for entry in ALL_ENTRIES:
by_source[entry.source_tag] = by_source.get(entry.source_tag, 0) + 1
print(f"\nTotal documents: {len(ALL_ENTRIES)}")
print("By source:")
for src, count in sorted(by_source.items()):
print(f" {src:15s}: {count:>3}")
print("\nDocuments:")
for i, entry in enumerate(ALL_ENTRIES, 1):
print(f" {i:2d}. [{entry.source_tag}] {entry.title[:70]}")
if args.dry_run:
print("\n[DRY RUN] Skipping bib.sqlite write")
return
# Store in bib.sqlite
print(f"\nStoring {len(ALL_ENTRIES)} documents in bib.sqlite ...")
store = Store()
created = 0
updated = 0
for entry in ALL_ENTRIES:
item = entry_to_item(entry)
con = store._con()
existing = con.execute(
"SELECT key FROM items WHERE url = ?", (item.url,)
).fetchone()
if existing:
updated += 1
else:
created += 1
store.upsert(item)
print(f" Created: {created}")
print(f" Updated: {updated}")
# Verify total skin-subs grey lit
con = store._con()
total = con.execute(
"""SELECT count(DISTINCT i.id) FROM items i
JOIN item_tags it ON i.id = it.item_id
JOIN tags t ON it.tag_id = t.id
WHERE t.name = 'module:skin-subs'
AND i.item_type = 'source'
AND i.url NOT LIKE '%pubmed%'"""
).fetchone()[0]
print(f" Total skin-subs grey lit in bib.sqlite: {total}")
store.close()
print("\nDone.")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,474 @@
"""PubMed systematic literature search for skin substitutes.
Queries NCBI E-utilities (esearch → efetch) across three domains:
1. Clinical efficacy — RCTs, systematic reviews, wound healing outcomes
2. Cost-effectiveness — health economics, Medicare spending, reimbursement
3. Fraud/waste/abuse — billing patterns, enforcement, compliance
Results are stored in bib.sqlite via the bib.store.Store with tags:
module:skin-subs, source:pubmed, type:{rct|review|meta-analysis|economic|policy|fraud}
Respects NCBI rate limits (3 requests/sec without API key, 10/sec with).
Set NCBI_API_KEY environment variable for higher throughput.
Usage:
uv run python dev/scripts/search_pubmed_skin_subs.py
uv run python dev/scripts/search_pubmed_skin_subs.py --dry-run
"""
from __future__ import annotations
import argparse
import os
import time
import xml.etree.ElementTree as ET
from dataclasses import dataclass, field
from datetime import datetime
import httpx
from bib.item import Source
from bib.store import Store
# ---------------------------------------------------------------------------
# NCBI E-utilities configuration
# ---------------------------------------------------------------------------
EUTILS_BASE = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils"
API_KEY = os.environ.get("NCBI_API_KEY", "")
TOOL_NAME = "stack-skin-subs"
TOOL_EMAIL = "dev@localhost"
RATE_LIMIT = 0.34 if not API_KEY else 0.1 # seconds between requests
def _params(**kw: str) -> dict[str, str]:
"""Build E-utilities query params with tool/email/api_key."""
base = {"tool": TOOL_NAME, "email": TOOL_EMAIL}
if API_KEY:
base["api_key"] = API_KEY
base.update(kw)
return base
# ---------------------------------------------------------------------------
# Search strategies
# ---------------------------------------------------------------------------
_Q_CLINICAL = (
'("skin substitutes"[MeSH] OR "biological dressings"[MeSH] '
'OR "tissue scaffolds"[MeSH] OR "bioengineered skin"[tiab] '
'OR "cellular tissue product"[tiab] OR "wound matrix"[tiab] '
'OR "skin substitute"[tiab] OR "skin substitutes"[tiab] '
'OR "Apligraf"[tiab] OR "EpiFix"[tiab] OR "Grafix"[tiab] '
'OR "DermACELL"[tiab] OR "Dermagraft"[tiab]) '
'AND ("wound healing"[MeSH] OR "treatment outcome"[MeSH] '
'OR "clinical trial"[pt] OR "randomized controlled trial"[pt] '
'OR "systematic review"[pt] OR "meta-analysis"[pt] '
'OR "efficacy"[tiab] OR "effectiveness"[tiab])'
)
_Q_COST = (
'("skin substitutes"[MeSH] OR "biological dressings"[MeSH] '
'OR "skin substitute"[tiab] OR "skin substitutes"[tiab] '
'OR "cellular tissue product"[tiab]) '
'AND ("costs and cost analysis"[MeSH] '
'OR "cost-benefit analysis"[MeSH] '
'OR "health care costs"[MeSH] OR "medicare"[MeSH] '
'OR "reimbursement"[tiab] OR "cost-effectiveness"[tiab] '
'OR "economic"[tiab] OR "spending"[tiab] '
'OR "average sales price"[tiab] OR "ASP"[tiab])'
)
_Q_FRAUD = (
'("skin substitutes"[MeSH] OR "biological dressings"[MeSH] '
'OR "skin substitute"[tiab] OR "skin substitutes"[tiab] '
'OR "cellular tissue product"[tiab] OR "wound care"[tiab]) '
'AND ("fraud"[MeSH] OR "waste"[tiab] OR "abuse"[tiab] '
'OR "inappropriate"[tiab] OR "overutilization"[tiab] '
'OR "billing"[tiab] OR "enforcement"[tiab] '
'OR "compliance"[tiab] OR "medically unnecessary"[tiab] '
'OR "upcoding"[tiab] OR "kickback"[tiab])'
)
SEARCHES: dict[str, dict[str, str | list[str]]] = {
"clinical_efficacy": {"query": _Q_CLINICAL, "tags": ["type:clinical"]},
"cost_effectiveness": {"query": _Q_COST, "tags": ["type:economic"]},
"fraud_waste_abuse": {"query": _Q_FRAUD, "tags": ["type:fraud"]},
}
# ---------------------------------------------------------------------------
# Article model
# ---------------------------------------------------------------------------
@dataclass
class Article:
pmid: str = ""
title: str = ""
abstract: str = ""
authors: list[str] = field(default_factory=list)
journal: str = ""
year: str = ""
doi: str = ""
pub_types: list[str] = field(default_factory=list)
mesh_terms: list[str] = field(default_factory=list)
@property
def url(self) -> str:
return f"https://pubmed.ncbi.nlm.nih.gov/{self.pmid}/"
def infer_type_tag(self) -> str:
"""Infer article type from PubMed publication types."""
pt_lower = [p.lower() for p in self.pub_types]
if "meta-analysis" in pt_lower:
return "type:meta-analysis"
if "systematic review" in pt_lower:
return "type:review"
if "review" in pt_lower:
return "type:review"
if "randomized controlled trial" in pt_lower:
return "type:rct"
if "clinical trial" in pt_lower:
return "type:rct"
return ""
# ---------------------------------------------------------------------------
# E-utilities helpers
# ---------------------------------------------------------------------------
def esearch(query: str, retmax: int = 10000) -> list[str]:
"""Search PubMed, return list of PMIDs."""
params = _params(
db="pubmed",
term=query,
retmax=str(retmax),
retmode="json",
usehistory="n",
)
resp = httpx.get(f"{EUTILS_BASE}/esearch.fcgi", params=params, timeout=30)
resp.raise_for_status()
data = resp.json()
result = data.get("esearchresult", {})
ids = result.get("idlist", [])
count = int(result.get("count", 0))
print(f" esearch: {count} total results, retrieved {len(ids)} PMIDs")
return ids
def efetch_articles(
pmids: list[str], batch_size: int = 100, max_retries: int = 3
) -> list[Article]:
"""Fetch article metadata for a list of PMIDs."""
articles: list[Article] = []
for i in range(0, len(pmids), batch_size):
batch = pmids[i : i + batch_size]
params = _params(
db="pubmed",
id=",".join(batch),
rettype="xml",
retmode="xml",
)
for attempt in range(max_retries):
time.sleep(RATE_LIMIT * (attempt + 1))
try:
resp = httpx.get(
f"{EUTILS_BASE}/efetch.fcgi",
params=params,
timeout=120,
)
resp.raise_for_status()
articles.extend(_parse_pubmed_xml(resp.text))
break
except (httpx.RemoteProtocolError, httpx.ReadTimeout) as exc:
if attempt < max_retries - 1:
wait = 2 ** (attempt + 1)
print(f" RETRY batch {i // batch_size + 1} "
f"(attempt {attempt + 2}/{max_retries}, "
f"wait {wait}s): {exc}")
time.sleep(wait)
else:
print(f" SKIP batch {i // batch_size + 1} after "
f"{max_retries} attempts: {exc}")
print(f" efetch: batch {i // batch_size + 1}/{len(pmids) // batch_size + 1}, "
f"got {len(articles)} articles so far")
return articles
def _text(el: ET.Element | None, path: str, default: str = "") -> str:
"""Get text from an XML element, handling None."""
if el is None:
return default
node = el.find(path)
return (node.text or default) if node is not None else default
def _parse_pubmed_xml(xml_text: str) -> list[Article]:
"""Parse PubMed efetch XML into Article objects."""
root = ET.fromstring(xml_text) # noqa: S314
articles: list[Article] = []
for art_el in root.findall(".//PubmedArticle"):
citation = art_el.find(".//MedlineCitation")
if citation is None:
continue
pmid = _text(citation, "PMID")
article_el = citation.find("Article")
if article_el is None:
continue
title = _text(article_el, "ArticleTitle")
# Abstract — may have multiple AbstractText elements
abstract_parts: list[str] = []
abstract_el = article_el.find("Abstract")
if abstract_el is not None:
for at in abstract_el.findall("AbstractText"):
label = at.get("Label", "")
text = "".join(at.itertext()).strip()
if label:
abstract_parts.append(f"{label}: {text}")
else:
abstract_parts.append(text)
abstract = "\n\n".join(abstract_parts)
# Authors
authors: list[str] = []
author_list = article_el.find("AuthorList")
if author_list is not None:
for au in author_list.findall("Author"):
last = _text(au, "LastName")
fore = _text(au, "ForeName")
if last:
authors.append(f"{last} {fore}".strip())
# Journal
journal_el = article_el.find("Journal")
journal = _text(journal_el, "Title") if journal_el is not None else ""
# Year
year = ""
pub_date = article_el.find(".//PubDate")
if pub_date is not None:
year = _text(pub_date, "Year")
if not year:
medline = _text(pub_date, "MedlineDate")
if medline:
year = medline[:4]
# DOI
doi = ""
for id_el in art_el.findall(".//ArticleId"):
if id_el.get("IdType") == "doi":
doi = id_el.text or ""
break
# Publication types
pub_types: list[str] = []
for pt in article_el.findall(".//PublicationType"):
if pt.text:
pub_types.append(pt.text)
# MeSH terms
mesh_terms: list[str] = []
mesh_list = citation.find("MeshHeadingList")
if mesh_list is not None:
for mh in mesh_list.findall("MeshHeading"):
desc = mh.find("DescriptorName")
if desc is not None and desc.text:
mesh_terms.append(desc.text)
articles.append(
Article(
pmid=pmid,
title=title,
abstract=abstract,
authors=authors,
journal=journal,
year=year,
doi=doi,
pub_types=pub_types,
mesh_terms=mesh_terms,
)
)
return articles
# ---------------------------------------------------------------------------
# Deduplication
# ---------------------------------------------------------------------------
def deduplicate(all_articles: dict[str, list[Article]]) -> dict[str, Article]:
"""Merge across search domains. First-seen domain tags win, extras merge."""
seen: dict[str, Article] = {}
seen_domains: dict[str, list[str]] = {}
for domain, articles in all_articles.items():
for art in articles:
if art.pmid in seen:
seen_domains[art.pmid].append(domain)
else:
seen[art.pmid] = art
seen_domains[art.pmid] = [domain]
return seen
# ---------------------------------------------------------------------------
# Store integration
# ---------------------------------------------------------------------------
def article_to_source(article: Article, domain_tags: list[str]) -> Source:
"""Convert an Article to a bib Source item."""
tags = ["module:skin-subs", "source:pubmed"]
if article.year:
tags.append(f"year:{article.year}")
# Add type tag from pub types
type_tag = article.infer_type_tag()
if type_tag:
tags.append(type_tag)
# Add domain tags
tags.extend(domain_tags)
# Build author string for extra
author_str = "; ".join(article.authors[:10])
if len(article.authors) > 10:
author_str += f" (+{len(article.authors) - 10} more)"
extra_parts = [f"PMID: {article.pmid}"]
if article.doi:
extra_parts.append(f"DOI: {article.doi}")
if author_str:
extra_parts.append(f"Authors: {author_str}")
if article.journal:
extra_parts.append(f"Journal: {article.journal}")
if article.pub_types:
extra_parts.append(f"PubTypes: {'; '.join(article.pub_types)}")
if article.mesh_terms:
extra_parts.append(f"MeSH: {'; '.join(article.mesh_terms[:15])}")
return Source(
title=article.title,
url=article.url,
date_published=article.year,
institution=article.journal,
abstract=article.abstract,
doc_type="journal-article",
tags=tags,
extra="\n".join(extra_parts),
)
# ---------------------------------------------------------------------------
# Main
# ---------------------------------------------------------------------------
def main() -> None:
parser = argparse.ArgumentParser(description="PubMed skin substitutes search")
parser.add_argument("--dry-run", action="store_true",
help="Search only, don't write to bib.sqlite")
args = parser.parse_args()
print("=" * 70)
print("PubMed Systematic Search: Skin Substitutes")
print(f"Date: {datetime.now().strftime('%Y-%m-%d %H:%M')}")
print(f"API key: {'configured' if API_KEY else 'not set (3 req/sec limit)'}")
print("=" * 70)
# Run all three search domains
all_articles: dict[str, list[Article]] = {}
all_pmids: dict[str, list[str]] = {}
for domain, cfg in SEARCHES.items():
print(f"\n--- Domain: {domain} ---")
print(f" Query: {cfg['query'][:100]}...")
time.sleep(RATE_LIMIT)
pmids = esearch(cfg["query"])
all_pmids[domain] = pmids
if pmids:
articles = efetch_articles(pmids)
all_articles[domain] = articles
print(f" Fetched {len(articles)} articles")
else:
all_articles[domain] = []
print(" No results")
# PRISMA counts
print("\n" + "=" * 70)
print("PRISMA Flow — Identification")
print("=" * 70)
total_identified = sum(len(v) for v in all_pmids.values())
for domain, pmids in all_pmids.items():
print(f" {domain:25s}: {len(pmids):>5} records")
print(f" {'TOTAL identified':25s}: {total_identified:>5}")
# Deduplicate
merged = deduplicate(all_articles)
duplicates = total_identified - len(merged)
print(f"\n Duplicates removed: {duplicates:>5}")
print(f" Unique articles: {len(merged):>5}")
# Type breakdown
type_counts: dict[str, int] = {}
for art in merged.values():
tag = art.infer_type_tag() or "type:other"
type_counts[tag] = type_counts.get(tag, 0) + 1
print("\n By article type:")
for t, c in sorted(type_counts.items()):
print(f" {t:25s}: {c:>5}")
if args.dry_run:
print("\n[DRY RUN] Skipping bib.sqlite write")
return
# Store in bib.sqlite
print(f"\nStoring {len(merged)} articles in bib.sqlite ...")
store = Store()
# Build domain→pmid mapping for tags
pmid_domains: dict[str, list[str]] = {}
for domain, articles in all_articles.items():
domain_tags = SEARCHES[domain]["tags"]
for art in articles:
if art.pmid not in pmid_domains:
pmid_domains[art.pmid] = list(domain_tags)
else:
for t in domain_tags:
if t not in pmid_domains[art.pmid]:
pmid_domains[art.pmid].append(t)
created = 0
updated = 0
for pmid, article in sorted(merged.items()):
domain_tags = pmid_domains.get(pmid, [])
source = article_to_source(article, domain_tags)
# Check if already exists by URL
con = store._con()
existing = con.execute(
"SELECT key FROM items WHERE url = ?", (source.url,)
).fetchone()
if existing:
updated += 1
else:
created += 1
store.upsert(source)
print(f" Created: {created}")
print(f" Updated: {updated}")
print(f" Total skin-subs PubMed articles in bib.sqlite: {created + updated}")
store.close()
print("\nDone.")
if __name__ == "__main__":
main()