feat: Cloudflare Worker PDF proxy + headless batch retrieval (fixes #273)
Some checks failed
CI / skinny-install (aco) (push) Successful in 2m19s
CI / lint-test (push) Failing after 3m13s
CI / skinny-install (api) (push) Successful in 1m45s
CI / skinny-install (bcda) (push) Successful in 1m41s
CI / skinny-install (bib) (push) Successful in 2m2s
CI / skinny-install (bls) (push) Successful in 2m1s
CI / skinny-install (ccw) (push) Successful in 1m37s
CI / skinny-install (cli) (push) Successful in 2m7s
CI / skinny-install (cms) (push) Successful in 1m40s
CI / skinny-install (conf) (push) Successful in 1m57s
CI / skinny-install (perf) (push) Successful in 1m41s
CI / skinny-install (rex) (push) Successful in 1m46s
CI / skinny-install (pfs) (push) Successful in 2m37s
Deploy / build-scan-report (push) Failing after 2m34s

Cloudflare Worker (pdf-proxy) deployed to bypass ISP-level SciHub blocking:
- Worker fetches PDFs from SciHub mirrors on Cloudflare's edge network
- Authenticated via X-Proxy-Key header
- Routes: /pdf?doi=, /fetch?url=, /health
- Deployed at pdf-proxy.lite7889.workers.dev

fetch_pdfs.py updated with 5-phase waterfall:
1. Unpaywall (legal OA, ~4% for this corpus)
2. PMC (free, PMCID-based)
3. Semantic Scholar (batch 500)
4. CF Worker → SciHub (72% hit rate on first 50, running full batch)
5. SciHub direct (fallback with proxy)

Also: cloudflared tunnel (stack-proxy) with WARP routing in compose.yml,
CoreDNS sci-hub domain routing via unfiltered DNS.

Test: 36/50 PDFs (72%) retrieved in first batch via Worker.
This commit is contained in:
kert
2026-03-25 21:54:44 -04:00
parent de93598c2f
commit 88901d0d17
6 changed files with 241 additions and 7 deletions

View File

@@ -624,6 +624,16 @@ services:
- no-new-privileges:true - no-new-privileges:true
restart: unless-stopped restart: unless-stopped
cloudflared:
image: cloudflare/cloudflared:latest
container_name: cloudflared
command: tunnel --config /home/nonroot/.cloudflared/config.yml run
networks:
- gateway
volumes:
- ./infra/cloudflared:/home/nonroot/.cloudflared:ro
restart: unless-stopped
volumes: volumes:
postgres_data: postgres_data:
rustfs_data: rustfs_data:

View File

@@ -155,11 +155,47 @@ SCIHUB_MIRRORS = [
# Proxy for SciHub (set SCIHUB_PROXY env var, e.g. socks5://localhost:1080) # Proxy for SciHub (set SCIHUB_PROXY env var, e.g. socks5://localhost:1080)
SCIHUB_PROXY = os.environ.get("SCIHUB_PROXY", "") SCIHUB_PROXY = os.environ.get("SCIHUB_PROXY", "")
# Cloudflare Worker proxy (preferred — bypasses ISP blocking via CF edge)
CF_WORKER_URL = os.environ.get(
"PDF_PROXY_URL", "https://pdf-proxy.lite7889.workers.dev"
)
CF_WORKER_SECRET = os.environ.get("PDF_PROXY_SECRET", "")
def fetch_via_worker(doi: str, dest: Path) -> bool:
"""Fetch PDF via Cloudflare Worker proxy. Returns True on success.
The Worker fetches from SciHub on Cloudflare's edge network,
bypassing ISP-level DNS/IP blocking.
"""
if not CF_WORKER_SECRET:
return False
try:
resp = httpx.get(
f"{CF_WORKER_URL}/pdf",
params={"doi": doi},
headers={
"X-Proxy-Key": CF_WORKER_SECRET,
"User-Agent": USER_AGENT,
},
timeout=45,
)
if resp.status_code != 200:
return False
if b"%PDF" not in resp.content[:1024]:
return False
dest.parent.mkdir(parents=True, exist_ok=True)
dest.write_bytes(resp.content)
return True
except Exception:
return False
def fetch_scihub(doi: str) -> str | None: def fetch_scihub(doi: str) -> str | None:
"""Get PDF URL from SciHub. Returns direct PDF URL or None. """Get PDF URL from SciHub directly. Returns direct PDF URL or None.
Set SCIHUB_PROXY env var for networks where SciHub is blocked. Set SCIHUB_PROXY env var for networks where SciHub is blocked.
Prefer fetch_via_worker() which uses Cloudflare edge.
""" """
transport = httpx.HTTPTransport(proxy=SCIHUB_PROXY) if SCIHUB_PROXY else None transport = httpx.HTTPTransport(proxy=SCIHUB_PROXY) if SCIHUB_PROXY else None
client_kwargs = {"transport": transport} if transport else {} client_kwargs = {"transport": transport} if transport else {}
@@ -297,7 +333,7 @@ def get_items_needing_pdfs(con: sqlite3.Connection) -> list[dict]:
def main() -> None: def main() -> None:
parser = argparse.ArgumentParser(description="Headless PDF retrieval") parser = argparse.ArgumentParser(description="Headless PDF retrieval")
parser.add_argument("--limit", type=int, default=0, help="Max items to process (0=all)") parser.add_argument("--limit", type=int, default=0, help="Max items to process (0=all)")
parser.add_argument("--source", choices=["all", "unpaywall", "pmc", "s2", "scihub"], parser.add_argument("--source", choices=["all", "unpaywall", "pmc", "s2", "worker", "scihub"],
default="all", help="Which source to use") default="all", help="Which source to use")
parser.add_argument("--proxy", help="SOCKS/HTTP proxy for SciHub (e.g. socks5://localhost:1080)") parser.add_argument("--proxy", help="SOCKS/HTTP proxy for SciHub (e.g. socks5://localhost:1080)")
args = parser.parse_args() args = parser.parse_args()
@@ -388,10 +424,33 @@ def main() -> None:
print(f" Semantic Scholar: {stats.get('s2', 0)} PDFs") print(f" Semantic Scholar: {stats.get('s2', 0)} PDFs")
# Phase 4: SciHub (fallback, may hit CAPTCHAs or be blocked) # Phase 4: Cloudflare Worker → SciHub (bypasses ISP blocking)
remaining = [it for it in items if not it.get("_done")] remaining = [it for it in items if not it.get("_done")]
if args.source in ("all", "scihub") and remaining: if args.source in ("all", "worker", "scihub") and remaining and CF_WORKER_SECRET:
print(f"\n--- Phase 4: SciHub ({len(remaining)} remaining) ---") print(f"\n--- Phase 4: CF Worker → SciHub ({len(remaining)} remaining) ---")
consecutive_failures = 0
for i, item in enumerate(remaining):
if consecutive_failures >= 20:
print(f" Stopping: {consecutive_failures} consecutive failures")
break
dest = tmp_dir / f"{item['item_id']}.pdf"
if fetch_via_worker(item["doi"], dest):
attach_pdf_to_item(con, item["item_id"], dest, item["doi"])
stats["worker"] = stats.get("worker", 0) + 1
item["_done"] = True
consecutive_failures = 0
else:
consecutive_failures += 1
if (i + 1) % 25 == 0:
print(f" {i + 1}/{len(remaining)} worker={stats.get('worker', 0)} "
f"fails={consecutive_failures}")
time.sleep(1.5) # Be polite to Worker + SciHub
print(f" CF Worker: {stats.get('worker', 0)} PDFs")
# Phase 5: SciHub direct (fallback if Worker unavailable)
remaining = [it for it in items if not it.get("_done")]
if args.source in ("scihub",) and remaining and not CF_WORKER_SECRET:
print(f"\n--- Phase 5: SciHub direct ({len(remaining)} remaining) ---")
if not SCIHUB_PROXY: if not SCIHUB_PROXY:
print(" WARNING: No proxy set. SciHub may be blocked from this network.") print(" WARNING: No proxy set. SciHub may be blocked from this network.")
print(" Set SCIHUB_PROXY or use --proxy socks5://host:port") print(" Set SCIHUB_PROXY or use --proxy socks5://host:port")
@@ -422,14 +481,16 @@ def main() -> None:
print(f" SciHub: {stats['scihub']} PDFs") print(f" SciHub: {stats['scihub']} PDFs")
stats["failed"] = len([it for it in items if not it.get("_done")]) stats["failed"] = len([it for it in items if not it.get("_done")])
total_found = stats["unpaywall"] + stats["pmc"] + stats.get("s2", 0) + stats["scihub"] total_found = (stats["unpaywall"] + stats["pmc"] + stats.get("s2", 0)
+ stats.get("worker", 0) + stats.get("scihub", 0))
print(f"\n{'=' * 60}") print(f"\n{'=' * 60}")
print(f"Results: {total_found} PDFs downloaded") print(f"Results: {total_found} PDFs downloaded")
print(f" Unpaywall: {stats['unpaywall']}") print(f" Unpaywall: {stats['unpaywall']}")
print(f" PMC: {stats['pmc']}") print(f" PMC: {stats['pmc']}")
print(f" S2: {stats.get('s2', 0)}") print(f" S2: {stats.get('s2', 0)}")
print(f" SciHub: {stats['scihub']}") print(f" CF Worker: {stats.get('worker', 0)}")
print(f" SciHub: {stats.get('scihub', 0)}")
print(f" Failed: {stats['failed']}") print(f" Failed: {stats['failed']}")
print(f" Coverage: {total_found * 100 // max(len(items), 1)}%") print(f" Coverage: {total_found * 100 // max(len(items), 1)}%")

View File

@@ -0,0 +1,10 @@
tunnel: 1389035e-d3ba-4a4f-969d-a369c07ee057
credentials-file: /home/nonroot/.cloudflared/1389035e-d3ba-4a4f-969d-a369c07ee057.json
warp-routing:
enabled: true
ingress:
- service: socks-proxy
originRequest:
connectTimeout: 30s

View File

@@ -0,0 +1,6 @@
{
"account": {
"id": "89f36257ec24ba34152e7c82a66335a3",
"name": "Lite@fhirworx.io's Account"
}
}

View File

@@ -0,0 +1,139 @@
/**
* PDF Fetch Proxy — Cloudflare Worker
*
* Relays HTTP requests through Cloudflare's edge network,
* bypassing ISP-level DNS/IP blocking of academic sources.
*
* Usage:
* GET https://<worker>/fetch?url=https://sci-hub.st/10.1234/example
* GET https://<worker>/pdf?doi=10.1234/example
*
* Security: requires X-Proxy-Key header matching the SECRET binding.
*/
const SCIHUB_MIRRORS = [
"https://sci-hub.st",
"https://sci-hub.ru",
"https://sci-hub.se",
"https://sci-hub.ren",
"https://sci-hub.ee",
];
export default {
async fetch(request, env) {
// Auth check
const key = request.headers.get("X-Proxy-Key");
if (!key || key !== env.SECRET) {
return new Response("Unauthorized", { status: 401 });
}
const url = new URL(request.url);
// Route: /fetch?url=<encoded_url> — generic proxy
if (url.pathname === "/fetch") {
const targetUrl = url.searchParams.get("url");
if (!targetUrl) {
return new Response("Missing url parameter", { status: 400 });
}
return proxyFetch(targetUrl);
}
// Route: /pdf?doi=<doi> — SciHub PDF resolver
if (url.pathname === "/pdf") {
const doi = url.searchParams.get("doi");
if (!doi) {
return new Response("Missing doi parameter", { status: 400 });
}
return fetchPdfFromScihub(doi);
}
// Route: /health
if (url.pathname === "/health") {
return new Response(JSON.stringify({ status: "ok", ts: Date.now() }), {
headers: { "Content-Type": "application/json" },
});
}
return new Response("Not found. Use /fetch?url=, /pdf?doi=, or /health", {
status: 404,
});
},
};
async function proxyFetch(targetUrl) {
try {
const resp = await fetch(targetUrl, {
headers: { "User-Agent": "Mozilla/5.0 (compatible; stack-proxy/1.0)" },
redirect: "follow",
});
// Stream the response back with original headers
const headers = new Headers(resp.headers);
headers.set("X-Proxy-Source", "cloudflare-worker");
return new Response(resp.body, {
status: resp.status,
headers,
});
} catch (e) {
return new Response(JSON.stringify({ error: e.message }), {
status: 502,
headers: { "Content-Type": "application/json" },
});
}
}
async function fetchPdfFromScihub(doi) {
for (const mirror of SCIHUB_MIRRORS) {
try {
const pageUrl = `${mirror}/${doi}`;
const resp = await fetch(pageUrl, {
headers: { "User-Agent": "Mozilla/5.0 (compatible; stack-proxy/1.0)" },
redirect: "follow",
});
if (resp.status !== 200) continue;
const html = await resp.text();
// Extract PDF URL from SciHub page
let pdfUrl = null;
const patterns = [
/id="pdf"[^>]*src="([^"]+)"/,
/<iframe[^>]*src="([^"]*\.pdf[^"]*)"/,
/<embed[^>]*src="([^"]*\.pdf[^"]*)"/,
];
for (const pat of patterns) {
const m = html.match(pat);
if (m) {
pdfUrl = m[1];
break;
}
}
if (!pdfUrl) continue;
// Normalize URL
if (pdfUrl.startsWith("//")) pdfUrl = "https:" + pdfUrl;
else if (pdfUrl.startsWith("/")) pdfUrl = mirror + pdfUrl;
// Fetch the actual PDF
const pdfResp = await fetch(pdfUrl, {
headers: { "User-Agent": "Mozilla/5.0" },
redirect: "follow",
});
if (pdfResp.status === 200) {
const headers = new Headers(pdfResp.headers);
headers.set("Content-Type", "application/pdf");
headers.set("X-Proxy-Source", "cloudflare-worker");
headers.set("X-Proxy-Mirror", mirror);
return new Response(pdfResp.body, { status: 200, headers });
}
} catch (e) {
continue;
}
}
return new Response(
JSON.stringify({ error: "PDF not found on any mirror", doi }),
{ status: 404, headers: { "Content-Type": "application/json" } }
);
}

View File

@@ -0,0 +1,8 @@
name = "pdf-proxy"
main = "worker.js"
compatibility_date = "2026-03-25"
[vars]
# SECRET is set via `wrangler secret put SECRET`
workers_dev = true