merge: #723 — prisma fetch tiers: every Unpaywall location, Semantic Scholar openAccessPdf, idconv PMCID resolution (refs #723)
Some checks failed
CI / lint (push) Successful in 29s
CI / notebooks-smoke (push) Successful in 1m35s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
CI / test (push) Successful in 2m16s
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Failing after 36s
Infra CI / notebooks (push) Successful in 53s
Infra CI / api (push) Successful in 1m13s
Infra CI / docs (push) Successful in 1m49s
Infra CI / mc (push) Successful in 18s
Infra CI / llm (push) Successful in 43s
Deploy / report (push) Successful in 13s
Some checks failed
CI / lint (push) Successful in 29s
CI / notebooks-smoke (push) Successful in 1m35s
Deploy / notebooks (push) Has been skipped
Deploy / zotero (push) Has been skipped
Deploy / docs (push) Has been skipped
CI / test (push) Successful in 2m16s
Deploy / api (push) Has been skipped
Deploy / llm (push) Has been skipped
Deploy / mc (push) Has been skipped
Infra CI / zotero (push) Failing after 36s
Infra CI / notebooks (push) Successful in 53s
Infra CI / api (push) Successful in 1m13s
Infra CI / docs (push) Successful in 1m49s
Infra CI / mc (push) Successful in 18s
Infra CI / llm (push) Successful in 43s
Deploy / report (push) Successful in 13s
This commit is contained in:
@@ -12,7 +12,10 @@ Usage: stack bib fetch-pfs-comments [OPTIONS]
|
|||||||
docket from its stored watermark.
|
docket from its stored watermark.
|
||||||
|
|
||||||
Sealed dockets (comment period closed + quiet period + an empty pull)
|
Sealed dockets (comment period closed + quiet period + an empty pull)
|
||||||
cost nothing: no rule-metadata fetch, no resolve call, no walk.
|
cost nothing: no rule-metadata fetch, no resolve call, no walk. The
|
||||||
|
same is true for a docket already known via `--docket` that resolves
|
||||||
|
to a different id. A CMS id with no `dockets` row at all is not free
|
||||||
|
even under `--docket`: see that option's help.
|
||||||
|
|
||||||
╭─ Options ────────────────────────────────────────────────────────────────────╮
|
╭─ Options ────────────────────────────────────────────────────────────────────╮
|
||||||
│ --since TEXT [default: 2017-01-01] │
|
│ --since TEXT [default: 2017-01-01] │
|
||||||
@@ -23,8 +26,15 @@ Usage: stack bib fetch-pfs-comments [OPTIONS]
|
|||||||
│ [default: 0] │
|
│ [default: 0] │
|
||||||
│ --sleep FLOAT [default: 1.3] │
|
│ --sleep FLOAT [default: 1.3] │
|
||||||
│ --docket TEXT Only pull this reg.gov docket (e.g. │
|
│ --docket TEXT Only pull this reg.gov docket (e.g. │
|
||||||
│ CMS-2026-2377); other dockets are skipped │
|
│ CMS-2026-2377). Filtered before any API │
|
||||||
│ before any API call once known. │
|
│ call ONLY once the docket is known (a │
|
||||||
|
│ `dockets` row already exists for its CMS │
|
||||||
|
│ id) — a CMS id with no `dockets` row yet │
|
||||||
|
│ still costs one Federal Register │
|
||||||
|
│ rule-metadata fetch plus one │
|
||||||
|
│ resolve_docket call, because the reg.gov │
|
||||||
|
│ docket id isn't known until those calls │
|
||||||
|
│ resolve it. │
|
||||||
│ --force Walk every docket from page 1, sealed or │
|
│ --force Walk every docket from page 1, sealed or │
|
||||||
│ not. Never unseals. │
|
│ not. Never unseals. │
|
||||||
│ --help Show this message and exit. │
|
│ --help Show this message and exit. │
|
||||||
|
|||||||
@@ -14,9 +14,8 @@ Usage: stack pfs elements [OPTIONS]
|
|||||||
│ --code TEXT HCPCS/CPT code (repeatable). │
|
│ --code TEXT HCPCS/CPT code (repeatable). │
|
||||||
│ --family TEXT Expand a registered family (CCM, APCM, …); │
|
│ --family TEXT Expand a registered family (CCM, APCM, …); │
|
||||||
│ repeatable. │
|
│ repeatable. │
|
||||||
│ --all-payable Every A/R/T code in the newest RVU year │
|
│ --all-payable Every A/R/T code in the newest RVU year, one │
|
||||||
│ (experimental: ~20 min of SQL before any model │
|
│ inverted pass over fr_anchors. │
|
||||||
│ call). │
|
|
||||||
│ --no-llm Skip the local-model classifier (unknown lines go │
|
│ --no-llm Skip the local-model classifier (unknown lines go │
|
||||||
│ to review). │
|
│ to review). │
|
||||||
│ --dry-run Extract and report; write nothing. │
|
│ --dry-run Extract and report; write nothing. │
|
||||||
|
|||||||
@@ -10,7 +10,9 @@ Usage: stack prisma fetch [OPTIONS] NAME
|
|||||||
|
|
||||||
Fetch PDFs for every non-excluded item lacking an attachment.
|
Fetch PDFs for every non-excluded item lacking an attachment.
|
||||||
|
|
||||||
Source cascade: Unpaywall → PMC → fallback (via the VPN droplet).
|
Source cascade: Unpaywall (every OA location) → Semantic Scholar → PMC
|
||||||
|
→ fallback (via the VPN droplet). PMIDs lacking a PMCID are resolved
|
||||||
|
through NCBI idconv before the PMC tiers run.
|
||||||
Items excluded at stage 2 are untouched — the queue filter only
|
Items excluded at stage 2 are untouched — the queue filter only
|
||||||
passes `screen:include` and `screen:uncertain` through.
|
passes `screen:include` and `screen:uncertain` through.
|
||||||
|
|
||||||
|
|||||||
@@ -342,7 +342,9 @@ def fetch(
|
|||||||
) -> None:
|
) -> None:
|
||||||
"""Fetch PDFs for every non-excluded item lacking an attachment.
|
"""Fetch PDFs for every non-excluded item lacking an attachment.
|
||||||
|
|
||||||
Source cascade: Unpaywall → PMC → fallback (via the VPN droplet).
|
Source cascade: Unpaywall (every OA location) → Semantic Scholar → PMC
|
||||||
|
→ fallback (via the VPN droplet). PMIDs lacking a PMCID are resolved
|
||||||
|
through NCBI idconv before the PMC tiers run.
|
||||||
Items excluded at stage 2 are untouched — the queue filter only
|
Items excluded at stage 2 are untouched — the queue filter only
|
||||||
passes `screen:include` and `screen:uncertain` through.
|
passes `screen:include` and `screen:uncertain` through.
|
||||||
|
|
||||||
|
|||||||
@@ -1,11 +1,15 @@
|
|||||||
"""PDF retrieval for PRISMA screening — Unpaywall → PMC → fallback proxy.
|
"""PDF retrieval for PRISMA screening — Unpaywall → Semantic Scholar → PMC
|
||||||
|
→ fallback proxy.
|
||||||
|
|
||||||
Call graph::
|
Call graph::
|
||||||
|
|
||||||
pending_queue(db, project)
|
pending_queue(db, project, client=...)
|
||||||
→ list of (zot_id, doi, pmid, pmcid) needing a PDF
|
→ list of (zot_id, doi, pmid, pmcid) needing a PDF; with a client,
|
||||||
|
PMIDs lacking a PMCID are resolved through NCBI idconv (#723)
|
||||||
fetch_one(item, *, proxy=None)
|
fetch_one(item, *, proxy=None)
|
||||||
→ tries each source in order, returns (source, path_on_disk) | None
|
→ tries each source in order, returns (source, path_on_disk) | None
|
||||||
|
(every Unpaywall oa_location, then Semantic Scholar's
|
||||||
|
openAccessPdf — which also lends a PMCID to the PMC tiers)
|
||||||
attach_pdf(db, zot_parent_id, path)
|
attach_pdf(db, zot_parent_id, path)
|
||||||
→ creates a Zotero attachment row and copies the file into
|
→ creates a Zotero attachment row and copies the file into
|
||||||
the Zotero storage directory
|
the Zotero storage directory
|
||||||
@@ -71,7 +75,14 @@ class PendingItem:
|
|||||||
# ── Queue ────────────────────────────────────────────────────────
|
# ── Queue ────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
def pending_queue(db: Db, project: str, limit: int | None = None) -> list[PendingItem]:
|
def pending_queue(
|
||||||
|
db: Db,
|
||||||
|
project: str,
|
||||||
|
limit: int | None = None,
|
||||||
|
*,
|
||||||
|
client: httpx.Client | None = None,
|
||||||
|
email: str = "",
|
||||||
|
) -> list[PendingItem]:
|
||||||
"""Items in *project* that passed or deferred stage-2 screening and
|
"""Items in *project* that passed or deferred stage-2 screening and
|
||||||
don't yet have a PDF attached.
|
don't yet have a PDF attached.
|
||||||
|
|
||||||
@@ -80,6 +91,12 @@ def pending_queue(db: Db, project: str, limit: int | None = None) -> list[Pendin
|
|||||||
- tagged ``screen:include`` OR ``screen:uncertain``
|
- tagged ``screen:include`` OR ``screen:uncertain``
|
||||||
- NOT tagged ``module:prisma`` (anchor guard)
|
- NOT tagged ``module:prisma`` (anchor guard)
|
||||||
- no child attachment with an existing PDF path
|
- no child attachment with an existing PDF path
|
||||||
|
|
||||||
|
With *client*, items that carry a PMID but no PMCID are resolved in
|
||||||
|
one NCBI idconv batch and the answer is written back to ``extra`` as
|
||||||
|
a ``PMCID:`` line, so the PMC tiers get their chance and the next
|
||||||
|
run asks no question twice (#723). Without a client the queue is
|
||||||
|
built from the stored fields only — no network.
|
||||||
"""
|
"""
|
||||||
rows = db.con.execute(
|
rows = db.con.execute(
|
||||||
"""
|
"""
|
||||||
@@ -120,16 +137,87 @@ def pending_queue(db: Db, project: str, limit: int | None = None) -> list[Pendin
|
|||||||
title=_field(db, zot_id, "title"),
|
title=_field(db, zot_id, "title"),
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
if client is not None:
|
||||||
|
_resolve_queue_pmcids(db, out, client, email)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _resolve_queue_pmcids(
|
||||||
|
db: Db, items: list[PendingItem], client: httpx.Client, email: str
|
||||||
|
) -> None:
|
||||||
|
missing = [it for it in items if it.pmid and not it.pmcid]
|
||||||
|
if not missing:
|
||||||
|
return
|
||||||
|
found = resolve_pmcids(client, [it.pmid for it in missing], email)
|
||||||
|
if not found:
|
||||||
|
return
|
||||||
|
for it in missing:
|
||||||
|
pmcid = found.get(it.pmid.strip())
|
||||||
|
if not pmcid:
|
||||||
|
continue
|
||||||
|
it.pmcid = pmcid
|
||||||
|
extra = _field(db, it.zot_id, "extra")
|
||||||
|
db.set_fields(
|
||||||
|
it.zot_id,
|
||||||
|
{"extra": f"{extra}\nPMCID: {pmcid}" if extra else f"PMCID: {pmcid}"},
|
||||||
|
)
|
||||||
|
db.commit()
|
||||||
|
|
||||||
|
|
||||||
|
_IDCONV = "https://pmc.ncbi.nlm.nih.gov/tools/idconv/api/v1/articles/"
|
||||||
|
_IDCONV_BATCH = 200
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_pmcids(
|
||||||
|
client: httpx.Client, pmids: list[str], email: str
|
||||||
|
) -> dict[str, str]:
|
||||||
|
"""PMID → PMCID via NCBI's idconv service, 200 ids per call (#723).
|
||||||
|
|
||||||
|
Only records idconv actually found are returned; ``status: error``
|
||||||
|
rows ("Identifier not found in PMC") and transport failures drop out.
|
||||||
|
"""
|
||||||
|
ids = [p for p in dict.fromkeys((p or "").strip() for p in pmids) if p.isdigit()]
|
||||||
|
out: dict[str, str] = {}
|
||||||
|
for i in range(0, len(ids), _IDCONV_BATCH):
|
||||||
|
chunk = ids[i : i + _IDCONV_BATCH]
|
||||||
|
try:
|
||||||
|
r = client.get(
|
||||||
|
_IDCONV,
|
||||||
|
params={
|
||||||
|
"ids": ",".join(chunk),
|
||||||
|
"idtype": "pmid",
|
||||||
|
"format": "json",
|
||||||
|
"tool": "stack-prisma",
|
||||||
|
"email": email,
|
||||||
|
},
|
||||||
|
timeout=30,
|
||||||
|
)
|
||||||
|
if r.status_code != 200:
|
||||||
|
continue
|
||||||
|
records = r.json().get("records") or []
|
||||||
|
except (httpx.HTTPError, ValueError):
|
||||||
|
continue
|
||||||
|
for rec in records:
|
||||||
|
pmcid = rec.get("pmcid")
|
||||||
|
if not pmcid or rec.get("status") == "error":
|
||||||
|
continue
|
||||||
|
out[str(rec.get("requested-id") or rec.get("pmid"))] = pmcid
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
# ── Source fetchers ─────────────────────────────────────────────
|
# ── Source fetchers ─────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
def fetch_unpaywall(client: httpx.Client, doi: str, email: str) -> str | None:
|
def unpaywall_urls(client: httpx.Client, doi: str, email: str) -> list[str]:
|
||||||
"""Unpaywall → PDF URL (or None)."""
|
"""Every OA URL Unpaywall knows for *doi*, best location first.
|
||||||
|
|
||||||
|
``best_oa_location`` is a ranking, not a guarantee — on the #650
|
||||||
|
pass it pointed at dead publisher links while a later
|
||||||
|
``oa_locations`` entry served the PDF (#723). Direct PDF URLs come
|
||||||
|
before landing pages across all locations; duplicates dropped.
|
||||||
|
"""
|
||||||
if not doi:
|
if not doi:
|
||||||
return None
|
return []
|
||||||
try:
|
try:
|
||||||
r = client.get(
|
r = client.get(
|
||||||
f"https://api.unpaywall.org/v2/{doi}",
|
f"https://api.unpaywall.org/v2/{doi}",
|
||||||
@@ -137,11 +225,78 @@ def fetch_unpaywall(client: httpx.Client, doi: str, email: str) -> str | None:
|
|||||||
timeout=15,
|
timeout=15,
|
||||||
)
|
)
|
||||||
if r.status_code != 200:
|
if r.status_code != 200:
|
||||||
return None
|
return []
|
||||||
best = r.json().get("best_oa_location") or {}
|
data = r.json()
|
||||||
return best.get("url_for_pdf") or best.get("url")
|
except (httpx.HTTPError, ValueError):
|
||||||
except httpx.HTTPError:
|
return []
|
||||||
|
locs = [data.get("best_oa_location") or {}, *(data.get("oa_locations") or [])]
|
||||||
|
out: list[str] = []
|
||||||
|
for key in ("url_for_pdf", "url"):
|
||||||
|
for loc in locs:
|
||||||
|
u = loc.get(key)
|
||||||
|
if u and u not in out:
|
||||||
|
out.append(u)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def fetch_unpaywall(client: httpx.Client, doi: str, email: str) -> str | None:
|
||||||
|
"""Unpaywall → best PDF URL (or None). See :func:`unpaywall_urls`."""
|
||||||
|
urls = unpaywall_urls(client, doi, email)
|
||||||
|
return urls[0] if urls else None
|
||||||
|
|
||||||
|
|
||||||
|
_S2_BASE = "https://api.semanticscholar.org/graph/v1/paper"
|
||||||
|
_S2_MIN_INTERVAL = 1.0 # unauthenticated graph API: ~1 req/s shared pool
|
||||||
|
_S2_LAST = [0.0]
|
||||||
|
_s2_lock = threading.Lock()
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class S2Hit:
|
||||||
|
"""What Semantic Scholar knows that helps the cascade."""
|
||||||
|
|
||||||
|
pdf_url: str
|
||||||
|
pmcid: str
|
||||||
|
|
||||||
|
|
||||||
|
def fetch_semantic_scholar(client: httpx.Client, doi: str) -> S2Hit | None:
|
||||||
|
"""Semantic Scholar graph API → ``openAccessPdf`` URL and/or a PMCID.
|
||||||
|
|
||||||
|
Found 4 of 30 items the rest of the cascade missed on the #650 pass
|
||||||
|
(#723). Its ``externalIds.PubMedCentral`` also fills in a PMCID the
|
||||||
|
Zotero record lacks, so the PMC tiers run for items Unpaywall only
|
||||||
|
knows by landing page. Rate-limited to ``_S2_MIN_INTERVAL`` across
|
||||||
|
threads; no API key needed.
|
||||||
|
"""
|
||||||
|
if not doi:
|
||||||
return None
|
return None
|
||||||
|
with _s2_lock:
|
||||||
|
wait = _S2_MIN_INTERVAL - (time.monotonic() - _S2_LAST[0])
|
||||||
|
if wait > 0:
|
||||||
|
time.sleep(wait)
|
||||||
|
try:
|
||||||
|
r = client.get(
|
||||||
|
f"{_S2_BASE}/DOI:{doi}",
|
||||||
|
params={"fields": "openAccessPdf,externalIds"},
|
||||||
|
timeout=15,
|
||||||
|
)
|
||||||
|
except httpx.HTTPError:
|
||||||
|
return None
|
||||||
|
finally:
|
||||||
|
_S2_LAST[0] = time.monotonic()
|
||||||
|
if r.status_code != 200:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
data = r.json()
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
pdf = (data.get("openAccessPdf") or {}).get("url") or ""
|
||||||
|
pmc = str((data.get("externalIds") or {}).get("PubMedCentral") or "")
|
||||||
|
if pmc and not pmc.startswith("PMC"):
|
||||||
|
pmc = f"PMC{pmc}"
|
||||||
|
if not pdf and not pmc:
|
||||||
|
return None
|
||||||
|
return S2Hit(pdf_url=pdf, pmcid=pmc)
|
||||||
|
|
||||||
|
|
||||||
def fetch_pmc(client: httpx.Client, pmcid: str) -> str | None:
|
def fetch_pmc(client: httpx.Client, pmcid: str) -> str | None:
|
||||||
@@ -469,34 +624,47 @@ def fetch_one(
|
|||||||
) -> tuple[str, Path] | None:
|
) -> tuple[str, Path] | None:
|
||||||
"""Run one item through the source cascade.
|
"""Run one item through the source cascade.
|
||||||
|
|
||||||
``client`` is the unproxied httpx client (for Unpaywall + PMC).
|
``client`` is the unproxied httpx client (Unpaywall, Semantic
|
||||||
``client_proxied`` (optional) routes the fallback tier through the
|
Scholar, PMC). ``client_proxied`` (optional) routes the fallback
|
||||||
VPN; if None, the fallback is skipped.
|
tier through the VPN; if None, the fallback is skipped.
|
||||||
|
|
||||||
|
Order: every Unpaywall location → Semantic Scholar openAccessPdf →
|
||||||
|
PMC S3 → PMC OA service → Europe PMC → fallback. A PMCID the item
|
||||||
|
lacks may arrive from Semantic Scholar before the PMC tiers run.
|
||||||
"""
|
"""
|
||||||
filename = scratch / f"{item.zot_id}-{_hash(item.doi or item.title)}.pdf"
|
filename = scratch / f"{item.zot_id}-{_hash(item.doi or item.title)}.pdf"
|
||||||
doi = item.doi
|
doi = item.doi
|
||||||
if not doi and item.title:
|
if not doi and item.title:
|
||||||
doi = resolve_doi_from_title(client, item.title) or ""
|
doi = resolve_doi_from_title(client, item.title) or ""
|
||||||
|
|
||||||
url = fetch_unpaywall(client, doi, email)
|
for url in unpaywall_urls(client, doi, email):
|
||||||
if url:
|
|
||||||
path = download(client, url, filename)
|
path = download(client, url, filename)
|
||||||
if path:
|
if path:
|
||||||
return "unpaywall", path
|
return "unpaywall", path
|
||||||
|
|
||||||
url = fetch_pmc_s3(client, item.pmcid)
|
pmcid = item.pmcid
|
||||||
|
s2 = fetch_semantic_scholar(client, doi)
|
||||||
|
if s2:
|
||||||
|
if s2.pdf_url:
|
||||||
|
path = download(client, s2.pdf_url, filename)
|
||||||
|
if path:
|
||||||
|
return "s2", path
|
||||||
|
if not pmcid:
|
||||||
|
pmcid = s2.pmcid
|
||||||
|
|
||||||
|
url = fetch_pmc_s3(client, pmcid)
|
||||||
if url:
|
if url:
|
||||||
path = download(client, url, filename)
|
path = download(client, url, filename)
|
||||||
if path:
|
if path:
|
||||||
return "pmc", path
|
return "pmc", path
|
||||||
|
|
||||||
url = fetch_pmc(client, item.pmcid)
|
url = fetch_pmc(client, pmcid)
|
||||||
if url:
|
if url:
|
||||||
path = download(client, url, filename)
|
path = download(client, url, filename)
|
||||||
if path:
|
if path:
|
||||||
return "pmc", path
|
return "pmc", path
|
||||||
|
|
||||||
path = fetch_europepmc_download(client, item.pmcid, filename)
|
path = fetch_europepmc_download(client, pmcid, filename)
|
||||||
if path:
|
if path:
|
||||||
return "pmc", path
|
return "pmc", path
|
||||||
|
|
||||||
@@ -563,13 +731,13 @@ def run(
|
|||||||
as `screen:reason:unavailable-fulltext`.
|
as `screen:reason:unavailable-fulltext`.
|
||||||
"""
|
"""
|
||||||
scratch_dir.mkdir(parents=True, exist_ok=True)
|
scratch_dir.mkdir(parents=True, exist_ok=True)
|
||||||
pending = pending_queue(db, project, limit=limit)
|
stats = {"unpaywall": 0, "s2": 0, "pmc": 0, "fallback": 0, "missed": 0, "errors": 0}
|
||||||
stats = {"unpaywall": 0, "pmc": 0, "fallback": 0, "missed": 0, "errors": 0}
|
|
||||||
|
|
||||||
client = httpx.Client(
|
client = httpx.Client(
|
||||||
headers={"User-Agent": USER_AGENT.format(email)},
|
headers={"User-Agent": USER_AGENT.format(email)},
|
||||||
follow_redirects=True,
|
follow_redirects=True,
|
||||||
)
|
)
|
||||||
|
pending = pending_queue(db, project, limit=limit, client=client, email=email)
|
||||||
client_proxied: httpx.Client | None = None
|
client_proxied: httpx.Client | None = None
|
||||||
if fetch_proxy:
|
if fetch_proxy:
|
||||||
# Mirrors behind altcha treat our generic UA as a bot; a real
|
# Mirrors behind altcha treat our generic UA as a bot; a real
|
||||||
@@ -626,8 +794,9 @@ def run(
|
|||||||
if progress and done % 25 == 0:
|
if progress and done % 25 == 0:
|
||||||
print(
|
print(
|
||||||
f" fetched {done}/{len(pending)} "
|
f" fetched {done}/{len(pending)} "
|
||||||
f"(up={stats['unpaywall']} pmc={stats['pmc']} "
|
f"(up={stats['unpaywall']} s2={stats['s2']} "
|
||||||
f"fb={stats['fallback']} miss={stats['missed']})"
|
f"pmc={stats['pmc']} fb={stats['fallback']} "
|
||||||
|
f"miss={stats['missed']})"
|
||||||
)
|
)
|
||||||
finally:
|
finally:
|
||||||
client.close()
|
client.close()
|
||||||
|
|||||||
@@ -55,7 +55,8 @@ class TestFallbackSerialized:
|
|||||||
return None # miss → cascade ends, no download
|
return None # miss → cascade ends, no download
|
||||||
|
|
||||||
# direct tiers all miss so every call reaches the fallback tier
|
# direct tiers all miss so every call reaches the fallback tier
|
||||||
monkeypatch.setattr(fetch_mod, "fetch_unpaywall", lambda c, d, e: None)
|
monkeypatch.setattr(fetch_mod, "unpaywall_urls", lambda c, d, e: [])
|
||||||
|
monkeypatch.setattr(fetch_mod, "fetch_semantic_scholar", lambda c, d: None)
|
||||||
monkeypatch.setattr(fetch_mod, "fetch_pmc", lambda c, p: None)
|
monkeypatch.setattr(fetch_mod, "fetch_pmc", lambda c, p: None)
|
||||||
monkeypatch.setattr(fetch_mod, "fetch_fallback", fake_fallback)
|
monkeypatch.setattr(fetch_mod, "fetch_fallback", fake_fallback)
|
||||||
|
|
||||||
|
|||||||
@@ -184,7 +184,7 @@ class TestFetchOne:
|
|||||||
)
|
)
|
||||||
with (
|
with (
|
||||||
patch(
|
patch(
|
||||||
"prisma.fetch.fetch_unpaywall", return_value="https://ex.com/paper.pdf"
|
"prisma.fetch.unpaywall_urls", return_value=["https://ex.com/paper.pdf"]
|
||||||
),
|
),
|
||||||
patch("prisma.fetch.download", return_value=tmp_path / "paper.pdf"),
|
patch("prisma.fetch.download", return_value=tmp_path / "paper.pdf"),
|
||||||
):
|
):
|
||||||
@@ -196,7 +196,8 @@ class TestFetchOne:
|
|||||||
def test_pmc_hit(self, tmp_path):
|
def test_pmc_hit(self, tmp_path):
|
||||||
item = PendingItem(zot_id=1, doi="", title="Paper", pmcid="PMC123", pmid="")
|
item = PendingItem(zot_id=1, doi="", title="Paper", pmcid="PMC123", pmid="")
|
||||||
with (
|
with (
|
||||||
patch("prisma.fetch.fetch_unpaywall", return_value=None),
|
patch("prisma.fetch.unpaywall_urls", return_value=[]),
|
||||||
|
patch("prisma.fetch.fetch_semantic_scholar", return_value=None),
|
||||||
patch("prisma.fetch.fetch_pmc", return_value="https://pmc/pdf.pdf"),
|
patch("prisma.fetch.fetch_pmc", return_value="https://pmc/pdf.pdf"),
|
||||||
patch("prisma.fetch.download", return_value=tmp_path / "paper.pdf"),
|
patch("prisma.fetch.download", return_value=tmp_path / "paper.pdf"),
|
||||||
):
|
):
|
||||||
@@ -211,7 +212,8 @@ class TestFetchOne:
|
|||||||
)
|
)
|
||||||
proxied = MagicMock()
|
proxied = MagicMock()
|
||||||
with (
|
with (
|
||||||
patch("prisma.fetch.fetch_unpaywall", return_value=None),
|
patch("prisma.fetch.unpaywall_urls", return_value=[]),
|
||||||
|
patch("prisma.fetch.fetch_semantic_scholar", return_value=None),
|
||||||
patch("prisma.fetch.fetch_pmc", return_value=None),
|
patch("prisma.fetch.fetch_pmc", return_value=None),
|
||||||
patch(
|
patch(
|
||||||
"prisma.fetch.fetch_fallback", return_value="https://mirror/paper.pdf"
|
"prisma.fetch.fetch_fallback", return_value="https://mirror/paper.pdf"
|
||||||
@@ -232,7 +234,8 @@ class TestFetchOne:
|
|||||||
zot_id=1, doi="10.1234/test", title="Paper", pmcid="", pmid=""
|
zot_id=1, doi="10.1234/test", title="Paper", pmcid="", pmid=""
|
||||||
)
|
)
|
||||||
with (
|
with (
|
||||||
patch("prisma.fetch.fetch_unpaywall", return_value=None),
|
patch("prisma.fetch.unpaywall_urls", return_value=[]),
|
||||||
|
patch("prisma.fetch.fetch_semantic_scholar", return_value=None),
|
||||||
patch("prisma.fetch.fetch_pmc", return_value=None),
|
patch("prisma.fetch.fetch_pmc", return_value=None),
|
||||||
):
|
):
|
||||||
result = fetch_one(
|
result = fetch_one(
|
||||||
@@ -246,7 +249,7 @@ class TestFetchOne:
|
|||||||
patch(
|
patch(
|
||||||
"prisma.fetch.resolve_doi_from_title", return_value="10.1234/resolved"
|
"prisma.fetch.resolve_doi_from_title", return_value="10.1234/resolved"
|
||||||
),
|
),
|
||||||
patch("prisma.fetch.fetch_unpaywall", return_value="https://ex.com/p.pdf"),
|
patch("prisma.fetch.unpaywall_urls", return_value=["https://ex.com/p.pdf"]),
|
||||||
patch("prisma.fetch.download", return_value=tmp_path / "p.pdf"),
|
patch("prisma.fetch.download", return_value=tmp_path / "p.pdf"),
|
||||||
):
|
):
|
||||||
result = fetch_one(
|
result = fetch_one(
|
||||||
|
|||||||
312
tests/prisma/test_fetch_tiers.py
Normal file
312
tests/prisma/test_fetch_tiers.py
Normal file
@@ -0,0 +1,312 @@
|
|||||||
|
"""Tests for the #723 fetch tiers — Unpaywall all-locations, Semantic
|
||||||
|
Scholar openAccessPdf, and idconv PMCID resolution in the queue."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from unittest.mock import MagicMock
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from prisma import fetch as fetch_mod
|
||||||
|
from prisma.fetch import (
|
||||||
|
PendingItem,
|
||||||
|
fetch_one,
|
||||||
|
fetch_semantic_scholar,
|
||||||
|
pending_queue,
|
||||||
|
resolve_pmcids,
|
||||||
|
unpaywall_urls,
|
||||||
|
)
|
||||||
|
from zot.db import TYPE_MAP, Db
|
||||||
|
from zot.schema import create_db
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(autouse=True)
|
||||||
|
def _no_s2_throttle(monkeypatch):
|
||||||
|
monkeypatch.setattr(fetch_mod, "_S2_MIN_INTERVAL", 0.0)
|
||||||
|
|
||||||
|
|
||||||
|
def _json_client(routes: dict[str, object], seen: list[str] | None = None):
|
||||||
|
"""MockTransport client: substring of URL → JSON body (or status int)."""
|
||||||
|
|
||||||
|
def handler(request):
|
||||||
|
url = str(request.url)
|
||||||
|
if seen is not None:
|
||||||
|
seen.append(url)
|
||||||
|
for needle, body in routes.items():
|
||||||
|
if needle in url:
|
||||||
|
if isinstance(body, int):
|
||||||
|
return httpx.Response(body)
|
||||||
|
return httpx.Response(200, json=body)
|
||||||
|
return httpx.Response(404)
|
||||||
|
|
||||||
|
return httpx.Client(transport=httpx.MockTransport(handler))
|
||||||
|
|
||||||
|
|
||||||
|
# ── unpaywall_urls ────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestUnpaywallUrls:
|
||||||
|
def test_all_locations_pdf_first(self):
|
||||||
|
body = {
|
||||||
|
"best_oa_location": {"url_for_pdf": None, "url": "https://a/landing"},
|
||||||
|
"oa_locations": [
|
||||||
|
{"url_for_pdf": None, "url": "https://a/landing"},
|
||||||
|
{"url_for_pdf": "https://b/paper.pdf", "url": "https://b/landing"},
|
||||||
|
{"url_for_pdf": "https://c/paper.pdf", "url": None},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
urls = unpaywall_urls(_json_client({"unpaywall": body}), "10.1/x", "e@x")
|
||||||
|
# every PDF URL before any landing page; no duplicates
|
||||||
|
assert urls == [
|
||||||
|
"https://b/paper.pdf",
|
||||||
|
"https://c/paper.pdf",
|
||||||
|
"https://a/landing",
|
||||||
|
"https://b/landing",
|
||||||
|
]
|
||||||
|
|
||||||
|
def test_best_location_only_legacy_shape(self):
|
||||||
|
body = {"best_oa_location": {"url_for_pdf": "https://oa/p.pdf"}}
|
||||||
|
assert unpaywall_urls(_json_client({"unpaywall": body}), "10.1/x", "e@x") == [
|
||||||
|
"https://oa/p.pdf"
|
||||||
|
]
|
||||||
|
|
||||||
|
def test_no_doi(self):
|
||||||
|
assert unpaywall_urls(MagicMock(), "", "e@x") == []
|
||||||
|
|
||||||
|
def test_404(self):
|
||||||
|
assert unpaywall_urls(_json_client({"unpaywall": 404}), "10.1/x", "e@x") == []
|
||||||
|
|
||||||
|
def test_no_locations(self):
|
||||||
|
body = {"best_oa_location": None, "oa_locations": []}
|
||||||
|
assert unpaywall_urls(_json_client({"unpaywall": body}), "10.1/x", "e@x") == []
|
||||||
|
|
||||||
|
|
||||||
|
# ── fetch_semantic_scholar ────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestFetchSemanticScholar:
|
||||||
|
def test_pdf_and_pmcid(self):
|
||||||
|
seen: list[str] = []
|
||||||
|
body = {
|
||||||
|
"openAccessPdf": {"url": "https://s2/p.pdf", "status": "GREEN"},
|
||||||
|
"externalIds": {"PubMedCentral": "5226373", "DOI": "10.1/x"},
|
||||||
|
}
|
||||||
|
hit = fetch_semantic_scholar(
|
||||||
|
_json_client({"semanticscholar": body}, seen), "10.1/x"
|
||||||
|
)
|
||||||
|
assert hit is not None
|
||||||
|
assert hit.pdf_url == "https://s2/p.pdf"
|
||||||
|
assert hit.pmcid == "PMC5226373"
|
||||||
|
assert "paper/DOI:10.1/x" in seen[0] and "openAccessPdf" in seen[0]
|
||||||
|
|
||||||
|
def test_pmcid_only(self):
|
||||||
|
body = {"openAccessPdf": None, "externalIds": {"PubMedCentral": "PMC42"}}
|
||||||
|
hit = fetch_semantic_scholar(_json_client({"semanticscholar": body}), "10.1/x")
|
||||||
|
assert hit is not None
|
||||||
|
assert hit.pdf_url == ""
|
||||||
|
assert hit.pmcid == "PMC42"
|
||||||
|
|
||||||
|
def test_nothing_useful(self):
|
||||||
|
body = {"openAccessPdf": None, "externalIds": {"DOI": "10.1/x"}}
|
||||||
|
assert (
|
||||||
|
fetch_semantic_scholar(_json_client({"semanticscholar": body}), "10.1/x")
|
||||||
|
is None
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_404(self):
|
||||||
|
assert (
|
||||||
|
fetch_semantic_scholar(_json_client({"semanticscholar": 404}), "10.1/x")
|
||||||
|
is None
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_no_doi(self):
|
||||||
|
assert fetch_semantic_scholar(MagicMock(), "") is None
|
||||||
|
|
||||||
|
def test_throttled(self, monkeypatch):
|
||||||
|
"""Calls are spaced by _S2_MIN_INTERVAL even from a cold start."""
|
||||||
|
monkeypatch.setattr(fetch_mod, "_S2_MIN_INTERVAL", 0.2)
|
||||||
|
monkeypatch.setattr(fetch_mod, "_S2_LAST", [0.0])
|
||||||
|
client = _json_client({"semanticscholar": 404})
|
||||||
|
import time
|
||||||
|
|
||||||
|
t0 = time.monotonic()
|
||||||
|
fetch_semantic_scholar(client, "10.1/a")
|
||||||
|
fetch_semantic_scholar(client, "10.1/b")
|
||||||
|
assert time.monotonic() - t0 >= 0.2
|
||||||
|
|
||||||
|
|
||||||
|
# ── resolve_pmcids ────────────────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestResolvePmcids:
|
||||||
|
_RECORDS = {
|
||||||
|
"status": "ok",
|
||||||
|
"records": [
|
||||||
|
{
|
||||||
|
"doi": "10.1/x",
|
||||||
|
"pmcid": "PMC5226373",
|
||||||
|
"pmid": 27893131,
|
||||||
|
"requested-id": "27893131",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"pmid": 28679817,
|
||||||
|
"requested-id": "28679817",
|
||||||
|
"status": "error",
|
||||||
|
"errmsg": "Identifier not found in PMC",
|
||||||
|
},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
|
||||||
|
def test_maps_found_only(self):
|
||||||
|
seen: list[str] = []
|
||||||
|
out = resolve_pmcids(
|
||||||
|
_json_client({"idconv": self._RECORDS}, seen),
|
||||||
|
["27893131", "28679817"],
|
||||||
|
"e@x",
|
||||||
|
)
|
||||||
|
assert out == {"27893131": "PMC5226373"}
|
||||||
|
assert "idtype=pmid" in seen[0] and "ids=27893131%2C28679817" in seen[0]
|
||||||
|
assert "email=e%40x" in seen[0]
|
||||||
|
|
||||||
|
def test_batches_of_200(self):
|
||||||
|
seen: list[str] = []
|
||||||
|
client = _json_client({"idconv": {"status": "ok", "records": []}}, seen)
|
||||||
|
resolve_pmcids(client, [str(i) for i in range(1, 402)], "e@x")
|
||||||
|
assert len(seen) == 3
|
||||||
|
|
||||||
|
def test_skips_non_numeric_and_dupes(self):
|
||||||
|
seen: list[str] = []
|
||||||
|
client = _json_client({"idconv": {"status": "ok", "records": []}}, seen)
|
||||||
|
resolve_pmcids(client, ["", "abc", "12", "12"], "e@x")
|
||||||
|
assert len(seen) == 1 and "ids=12&" in seen[0]
|
||||||
|
|
||||||
|
def test_empty(self):
|
||||||
|
assert resolve_pmcids(MagicMock(), [], "e@x") == {}
|
||||||
|
|
||||||
|
def test_http_error(self):
|
||||||
|
assert resolve_pmcids(_json_client({"idconv": 500}), ["1"], "e@x") == {}
|
||||||
|
|
||||||
|
|
||||||
|
# ── pending_queue PMCID resolution ────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def _db(tmp_path):
|
||||||
|
path = str(tmp_path / "z.sqlite")
|
||||||
|
create_db(path).close()
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def _item(db, project, extra, doi="10.1/x"):
|
||||||
|
iid = db.create_item(TYPE_MAP["journalArticle"])
|
||||||
|
db.set_fields(iid, {"title": "Test Item", "DOI": doi, "extra": extra})
|
||||||
|
db.sync_tags(iid, [f"project:{project}", "screen:include"])
|
||||||
|
db.commit()
|
||||||
|
return iid
|
||||||
|
|
||||||
|
|
||||||
|
class TestPendingQueueIdconv:
|
||||||
|
def test_resolves_and_persists(self, tmp_path):
|
||||||
|
path = _db(tmp_path)
|
||||||
|
recs = {
|
||||||
|
"status": "ok",
|
||||||
|
"records": [
|
||||||
|
{"pmcid": "PMC777", "pmid": 111, "requested-id": "111"},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
seen: list[str] = []
|
||||||
|
client = _json_client({"idconv": recs}, seen)
|
||||||
|
with Db(path) as db:
|
||||||
|
a = _item(db, "p", "PMID: 111")
|
||||||
|
b = _item(db, "p", "PMID: 222\nPMCID: PMC222", doi="10.1/y")
|
||||||
|
c = _item(db, "p", "", doi="10.1/z")
|
||||||
|
items = pending_queue(db, "p", client=client, email="e@x")
|
||||||
|
by_id = {it.zot_id: it for it in items}
|
||||||
|
assert by_id[a].pmcid == "PMC777"
|
||||||
|
assert by_id[b].pmcid == "PMC222"
|
||||||
|
assert by_id[c].pmcid == ""
|
||||||
|
# only the one PMID lacking a PMCID was sent
|
||||||
|
assert len(seen) == 1 and "ids=111&" in seen[0]
|
||||||
|
# persisted into extra, existing lines kept
|
||||||
|
assert fetch_mod._field(db, a, "extra") == "PMID: 111\nPMCID: PMC777"
|
||||||
|
# second call: nothing left to resolve
|
||||||
|
pending_queue(db, "p", client=client, email="e@x")
|
||||||
|
assert len(seen) == 1
|
||||||
|
|
||||||
|
def test_no_client_no_network(self, tmp_path):
|
||||||
|
path = _db(tmp_path)
|
||||||
|
with Db(path) as db:
|
||||||
|
a = _item(db, "p", "PMID: 111")
|
||||||
|
items = pending_queue(db, "p")
|
||||||
|
assert items[0].zot_id == a and items[0].pmcid == ""
|
||||||
|
|
||||||
|
|
||||||
|
# ── fetch_one cascade order ───────────────────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_client(routes_json: dict[str, object], pdf_urls: set[str], seen: list[str]):
|
||||||
|
def handler(request):
|
||||||
|
url = str(request.url)
|
||||||
|
seen.append(url)
|
||||||
|
if url in pdf_urls:
|
||||||
|
return httpx.Response(
|
||||||
|
200,
|
||||||
|
content=b"%PDF-1.7\n" + b"x" * 2000,
|
||||||
|
headers={"content-type": "application/pdf"},
|
||||||
|
)
|
||||||
|
for needle, body in routes_json.items():
|
||||||
|
if needle in url:
|
||||||
|
return (
|
||||||
|
httpx.Response(200, json=body)
|
||||||
|
if not isinstance(body, int)
|
||||||
|
else httpx.Response(body)
|
||||||
|
)
|
||||||
|
return httpx.Response(404)
|
||||||
|
|
||||||
|
return httpx.Client(transport=httpx.MockTransport(handler))
|
||||||
|
|
||||||
|
|
||||||
|
class TestFetchOneTiers:
|
||||||
|
def test_second_unpaywall_location_wins(self, tmp_path):
|
||||||
|
seen: list[str] = []
|
||||||
|
up = {
|
||||||
|
"best_oa_location": {"url_for_pdf": "https://dead/p.pdf"},
|
||||||
|
"oa_locations": [
|
||||||
|
{"url_for_pdf": "https://dead/p.pdf"},
|
||||||
|
{"url_for_pdf": "https://alive/p.pdf"},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
client = _pdf_client({"unpaywall": up}, {"https://alive/p.pdf"}, seen)
|
||||||
|
item = PendingItem(zot_id=1, doi="10.1/x", pmid="", pmcid="", title="T")
|
||||||
|
res = fetch_one(item, email="e@x", scratch=tmp_path, client=client)
|
||||||
|
assert res is not None and res[0] == "unpaywall"
|
||||||
|
assert "https://dead/p.pdf" in seen and "https://alive/p.pdf" in seen
|
||||||
|
assert not any("semanticscholar" in u for u in seen)
|
||||||
|
|
||||||
|
def test_s2_pdf_after_unpaywall_miss(self, tmp_path):
|
||||||
|
seen: list[str] = []
|
||||||
|
s2 = {"openAccessPdf": {"url": "https://s2/p.pdf"}, "externalIds": {}}
|
||||||
|
client = _pdf_client(
|
||||||
|
{"unpaywall": 404, "semanticscholar": s2}, {"https://s2/p.pdf"}, seen
|
||||||
|
)
|
||||||
|
item = PendingItem(zot_id=1, doi="10.1/x", pmid="", pmcid="", title="T")
|
||||||
|
res = fetch_one(item, email="e@x", scratch=tmp_path, client=client)
|
||||||
|
assert res is not None and res[0] == "s2"
|
||||||
|
|
||||||
|
def test_s2_pmcid_feeds_pmc_tiers(self, tmp_path):
|
||||||
|
seen: list[str] = []
|
||||||
|
s2 = {"openAccessPdf": None, "externalIds": {"PubMedCentral": "99"}}
|
||||||
|
client = _pdf_client({"unpaywall": 404, "semanticscholar": s2}, set(), seen)
|
||||||
|
item = PendingItem(zot_id=1, doi="10.1/x", pmid="", pmcid="", title="T")
|
||||||
|
assert fetch_one(item, email="e@x", scratch=tmp_path, client=client) is None
|
||||||
|
assert any("prefix=PMC99." in u for u in seen), seen # S3 listing tried
|
||||||
|
assert any("id=PMC99" in u for u in seen), seen # OA service tried
|
||||||
|
|
||||||
|
def test_item_pmcid_not_overridden(self, tmp_path):
|
||||||
|
seen: list[str] = []
|
||||||
|
s2 = {"openAccessPdf": None, "externalIds": {"PubMedCentral": "99"}}
|
||||||
|
client = _pdf_client({"unpaywall": 404, "semanticscholar": s2}, set(), seen)
|
||||||
|
item = PendingItem(zot_id=1, doi="10.1/x", pmid="", pmcid="PMC1", title="T")
|
||||||
|
fetch_one(item, email="e@x", scratch=tmp_path, client=client)
|
||||||
|
assert not any("PMC99" in u for u in seen)
|
||||||
|
assert any("prefix=PMC1." in u for u in seen)
|
||||||
Reference in New Issue
Block a user