feat: Cloudflare Worker PDF proxy + headless batch retrieval (fixes #273)
Some checks failed
CI / skinny-install (aco) (push) Successful in 2m19s
CI / lint-test (push) Failing after 3m13s
CI / skinny-install (api) (push) Successful in 1m45s
CI / skinny-install (bcda) (push) Successful in 1m41s
CI / skinny-install (bib) (push) Successful in 2m2s
CI / skinny-install (bls) (push) Successful in 2m1s
CI / skinny-install (ccw) (push) Successful in 1m37s
CI / skinny-install (cli) (push) Successful in 2m7s
CI / skinny-install (cms) (push) Successful in 1m40s
CI / skinny-install (conf) (push) Successful in 1m57s
CI / skinny-install (perf) (push) Successful in 1m41s
CI / skinny-install (rex) (push) Successful in 1m46s
CI / skinny-install (pfs) (push) Successful in 2m37s
Deploy / build-scan-report (push) Failing after 2m34s
Some checks failed
CI / skinny-install (aco) (push) Successful in 2m19s
CI / lint-test (push) Failing after 3m13s
CI / skinny-install (api) (push) Successful in 1m45s
CI / skinny-install (bcda) (push) Successful in 1m41s
CI / skinny-install (bib) (push) Successful in 2m2s
CI / skinny-install (bls) (push) Successful in 2m1s
CI / skinny-install (ccw) (push) Successful in 1m37s
CI / skinny-install (cli) (push) Successful in 2m7s
CI / skinny-install (cms) (push) Successful in 1m40s
CI / skinny-install (conf) (push) Successful in 1m57s
CI / skinny-install (perf) (push) Successful in 1m41s
CI / skinny-install (rex) (push) Successful in 1m46s
CI / skinny-install (pfs) (push) Successful in 2m37s
Deploy / build-scan-report (push) Failing after 2m34s
Cloudflare Worker (pdf-proxy) deployed to bypass ISP-level SciHub blocking: - Worker fetches PDFs from SciHub mirrors on Cloudflare's edge network - Authenticated via X-Proxy-Key header - Routes: /pdf?doi=, /fetch?url=, /health - Deployed at pdf-proxy.lite7889.workers.dev fetch_pdfs.py updated with 5-phase waterfall: 1. Unpaywall (legal OA, ~4% for this corpus) 2. PMC (free, PMCID-based) 3. Semantic Scholar (batch 500) 4. CF Worker → SciHub (72% hit rate on first 50, running full batch) 5. SciHub direct (fallback with proxy) Also: cloudflared tunnel (stack-proxy) with WARP routing in compose.yml, CoreDNS sci-hub domain routing via unfiltered DNS. Test: 36/50 PDFs (72%) retrieved in first batch via Worker.
This commit is contained in:
10
compose.yml
10
compose.yml
@@ -624,6 +624,16 @@ services:
|
||||
- no-new-privileges:true
|
||||
restart: unless-stopped
|
||||
|
||||
cloudflared:
|
||||
image: cloudflare/cloudflared:latest
|
||||
container_name: cloudflared
|
||||
command: tunnel --config /home/nonroot/.cloudflared/config.yml run
|
||||
networks:
|
||||
- gateway
|
||||
volumes:
|
||||
- ./infra/cloudflared:/home/nonroot/.cloudflared:ro
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
postgres_data:
|
||||
rustfs_data:
|
||||
|
||||
@@ -155,11 +155,47 @@ SCIHUB_MIRRORS = [
|
||||
# Proxy for SciHub (set SCIHUB_PROXY env var, e.g. socks5://localhost:1080)
|
||||
SCIHUB_PROXY = os.environ.get("SCIHUB_PROXY", "")
|
||||
|
||||
# Cloudflare Worker proxy (preferred — bypasses ISP blocking via CF edge)
|
||||
CF_WORKER_URL = os.environ.get(
|
||||
"PDF_PROXY_URL", "https://pdf-proxy.lite7889.workers.dev"
|
||||
)
|
||||
CF_WORKER_SECRET = os.environ.get("PDF_PROXY_SECRET", "")
|
||||
|
||||
|
||||
def fetch_via_worker(doi: str, dest: Path) -> bool:
|
||||
"""Fetch PDF via Cloudflare Worker proxy. Returns True on success.
|
||||
|
||||
The Worker fetches from SciHub on Cloudflare's edge network,
|
||||
bypassing ISP-level DNS/IP blocking.
|
||||
"""
|
||||
if not CF_WORKER_SECRET:
|
||||
return False
|
||||
try:
|
||||
resp = httpx.get(
|
||||
f"{CF_WORKER_URL}/pdf",
|
||||
params={"doi": doi},
|
||||
headers={
|
||||
"X-Proxy-Key": CF_WORKER_SECRET,
|
||||
"User-Agent": USER_AGENT,
|
||||
},
|
||||
timeout=45,
|
||||
)
|
||||
if resp.status_code != 200:
|
||||
return False
|
||||
if b"%PDF" not in resp.content[:1024]:
|
||||
return False
|
||||
dest.parent.mkdir(parents=True, exist_ok=True)
|
||||
dest.write_bytes(resp.content)
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def fetch_scihub(doi: str) -> str | None:
|
||||
"""Get PDF URL from SciHub. Returns direct PDF URL or None.
|
||||
"""Get PDF URL from SciHub directly. Returns direct PDF URL or None.
|
||||
|
||||
Set SCIHUB_PROXY env var for networks where SciHub is blocked.
|
||||
Prefer fetch_via_worker() which uses Cloudflare edge.
|
||||
"""
|
||||
transport = httpx.HTTPTransport(proxy=SCIHUB_PROXY) if SCIHUB_PROXY else None
|
||||
client_kwargs = {"transport": transport} if transport else {}
|
||||
@@ -297,7 +333,7 @@ def get_items_needing_pdfs(con: sqlite3.Connection) -> list[dict]:
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Headless PDF retrieval")
|
||||
parser.add_argument("--limit", type=int, default=0, help="Max items to process (0=all)")
|
||||
parser.add_argument("--source", choices=["all", "unpaywall", "pmc", "s2", "scihub"],
|
||||
parser.add_argument("--source", choices=["all", "unpaywall", "pmc", "s2", "worker", "scihub"],
|
||||
default="all", help="Which source to use")
|
||||
parser.add_argument("--proxy", help="SOCKS/HTTP proxy for SciHub (e.g. socks5://localhost:1080)")
|
||||
args = parser.parse_args()
|
||||
@@ -388,10 +424,33 @@ def main() -> None:
|
||||
|
||||
print(f" Semantic Scholar: {stats.get('s2', 0)} PDFs")
|
||||
|
||||
# Phase 4: SciHub (fallback, may hit CAPTCHAs or be blocked)
|
||||
# Phase 4: Cloudflare Worker → SciHub (bypasses ISP blocking)
|
||||
remaining = [it for it in items if not it.get("_done")]
|
||||
if args.source in ("all", "scihub") and remaining:
|
||||
print(f"\n--- Phase 4: SciHub ({len(remaining)} remaining) ---")
|
||||
if args.source in ("all", "worker", "scihub") and remaining and CF_WORKER_SECRET:
|
||||
print(f"\n--- Phase 4: CF Worker → SciHub ({len(remaining)} remaining) ---")
|
||||
consecutive_failures = 0
|
||||
for i, item in enumerate(remaining):
|
||||
if consecutive_failures >= 20:
|
||||
print(f" Stopping: {consecutive_failures} consecutive failures")
|
||||
break
|
||||
dest = tmp_dir / f"{item['item_id']}.pdf"
|
||||
if fetch_via_worker(item["doi"], dest):
|
||||
attach_pdf_to_item(con, item["item_id"], dest, item["doi"])
|
||||
stats["worker"] = stats.get("worker", 0) + 1
|
||||
item["_done"] = True
|
||||
consecutive_failures = 0
|
||||
else:
|
||||
consecutive_failures += 1
|
||||
if (i + 1) % 25 == 0:
|
||||
print(f" {i + 1}/{len(remaining)} worker={stats.get('worker', 0)} "
|
||||
f"fails={consecutive_failures}")
|
||||
time.sleep(1.5) # Be polite to Worker + SciHub
|
||||
print(f" CF Worker: {stats.get('worker', 0)} PDFs")
|
||||
|
||||
# Phase 5: SciHub direct (fallback if Worker unavailable)
|
||||
remaining = [it for it in items if not it.get("_done")]
|
||||
if args.source in ("scihub",) and remaining and not CF_WORKER_SECRET:
|
||||
print(f"\n--- Phase 5: SciHub direct ({len(remaining)} remaining) ---")
|
||||
if not SCIHUB_PROXY:
|
||||
print(" WARNING: No proxy set. SciHub may be blocked from this network.")
|
||||
print(" Set SCIHUB_PROXY or use --proxy socks5://host:port")
|
||||
@@ -422,14 +481,16 @@ def main() -> None:
|
||||
print(f" SciHub: {stats['scihub']} PDFs")
|
||||
|
||||
stats["failed"] = len([it for it in items if not it.get("_done")])
|
||||
total_found = stats["unpaywall"] + stats["pmc"] + stats.get("s2", 0) + stats["scihub"]
|
||||
total_found = (stats["unpaywall"] + stats["pmc"] + stats.get("s2", 0)
|
||||
+ stats.get("worker", 0) + stats.get("scihub", 0))
|
||||
|
||||
print(f"\n{'=' * 60}")
|
||||
print(f"Results: {total_found} PDFs downloaded")
|
||||
print(f" Unpaywall: {stats['unpaywall']}")
|
||||
print(f" PMC: {stats['pmc']}")
|
||||
print(f" S2: {stats.get('s2', 0)}")
|
||||
print(f" SciHub: {stats['scihub']}")
|
||||
print(f" CF Worker: {stats.get('worker', 0)}")
|
||||
print(f" SciHub: {stats.get('scihub', 0)}")
|
||||
print(f" Failed: {stats['failed']}")
|
||||
print(f" Coverage: {total_found * 100 // max(len(items), 1)}%")
|
||||
|
||||
|
||||
10
infra/cloudflared/config.yml
Normal file
10
infra/cloudflared/config.yml
Normal file
@@ -0,0 +1,10 @@
|
||||
tunnel: 1389035e-d3ba-4a4f-969d-a369c07ee057
|
||||
credentials-file: /home/nonroot/.cloudflared/1389035e-d3ba-4a4f-969d-a369c07ee057.json
|
||||
|
||||
warp-routing:
|
||||
enabled: true
|
||||
|
||||
ingress:
|
||||
- service: socks-proxy
|
||||
originRequest:
|
||||
connectTimeout: 30s
|
||||
6
infra/cloudflared/worker/.wrangler/cache/wrangler-account.json
vendored
Normal file
6
infra/cloudflared/worker/.wrangler/cache/wrangler-account.json
vendored
Normal file
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"account": {
|
||||
"id": "89f36257ec24ba34152e7c82a66335a3",
|
||||
"name": "Lite@fhirworx.io's Account"
|
||||
}
|
||||
}
|
||||
139
infra/cloudflared/worker/worker.js
Normal file
139
infra/cloudflared/worker/worker.js
Normal file
@@ -0,0 +1,139 @@
|
||||
/**
|
||||
* PDF Fetch Proxy — Cloudflare Worker
|
||||
*
|
||||
* Relays HTTP requests through Cloudflare's edge network,
|
||||
* bypassing ISP-level DNS/IP blocking of academic sources.
|
||||
*
|
||||
* Usage:
|
||||
* GET https://<worker>/fetch?url=https://sci-hub.st/10.1234/example
|
||||
* GET https://<worker>/pdf?doi=10.1234/example
|
||||
*
|
||||
* Security: requires X-Proxy-Key header matching the SECRET binding.
|
||||
*/
|
||||
|
||||
const SCIHUB_MIRRORS = [
|
||||
"https://sci-hub.st",
|
||||
"https://sci-hub.ru",
|
||||
"https://sci-hub.se",
|
||||
"https://sci-hub.ren",
|
||||
"https://sci-hub.ee",
|
||||
];
|
||||
|
||||
export default {
|
||||
async fetch(request, env) {
|
||||
// Auth check
|
||||
const key = request.headers.get("X-Proxy-Key");
|
||||
if (!key || key !== env.SECRET) {
|
||||
return new Response("Unauthorized", { status: 401 });
|
||||
}
|
||||
|
||||
const url = new URL(request.url);
|
||||
|
||||
// Route: /fetch?url=<encoded_url> — generic proxy
|
||||
if (url.pathname === "/fetch") {
|
||||
const targetUrl = url.searchParams.get("url");
|
||||
if (!targetUrl) {
|
||||
return new Response("Missing url parameter", { status: 400 });
|
||||
}
|
||||
return proxyFetch(targetUrl);
|
||||
}
|
||||
|
||||
// Route: /pdf?doi=<doi> — SciHub PDF resolver
|
||||
if (url.pathname === "/pdf") {
|
||||
const doi = url.searchParams.get("doi");
|
||||
if (!doi) {
|
||||
return new Response("Missing doi parameter", { status: 400 });
|
||||
}
|
||||
return fetchPdfFromScihub(doi);
|
||||
}
|
||||
|
||||
// Route: /health
|
||||
if (url.pathname === "/health") {
|
||||
return new Response(JSON.stringify({ status: "ok", ts: Date.now() }), {
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
|
||||
return new Response("Not found. Use /fetch?url=, /pdf?doi=, or /health", {
|
||||
status: 404,
|
||||
});
|
||||
},
|
||||
};
|
||||
|
||||
async function proxyFetch(targetUrl) {
|
||||
try {
|
||||
const resp = await fetch(targetUrl, {
|
||||
headers: { "User-Agent": "Mozilla/5.0 (compatible; stack-proxy/1.0)" },
|
||||
redirect: "follow",
|
||||
});
|
||||
// Stream the response back with original headers
|
||||
const headers = new Headers(resp.headers);
|
||||
headers.set("X-Proxy-Source", "cloudflare-worker");
|
||||
return new Response(resp.body, {
|
||||
status: resp.status,
|
||||
headers,
|
||||
});
|
||||
} catch (e) {
|
||||
return new Response(JSON.stringify({ error: e.message }), {
|
||||
status: 502,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
async function fetchPdfFromScihub(doi) {
|
||||
for (const mirror of SCIHUB_MIRRORS) {
|
||||
try {
|
||||
const pageUrl = `${mirror}/${doi}`;
|
||||
const resp = await fetch(pageUrl, {
|
||||
headers: { "User-Agent": "Mozilla/5.0 (compatible; stack-proxy/1.0)" },
|
||||
redirect: "follow",
|
||||
});
|
||||
if (resp.status !== 200) continue;
|
||||
|
||||
const html = await resp.text();
|
||||
|
||||
// Extract PDF URL from SciHub page
|
||||
let pdfUrl = null;
|
||||
const patterns = [
|
||||
/id="pdf"[^>]*src="([^"]+)"/,
|
||||
/<iframe[^>]*src="([^"]*\.pdf[^"]*)"/,
|
||||
/<embed[^>]*src="([^"]*\.pdf[^"]*)"/,
|
||||
];
|
||||
for (const pat of patterns) {
|
||||
const m = html.match(pat);
|
||||
if (m) {
|
||||
pdfUrl = m[1];
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (!pdfUrl) continue;
|
||||
|
||||
// Normalize URL
|
||||
if (pdfUrl.startsWith("//")) pdfUrl = "https:" + pdfUrl;
|
||||
else if (pdfUrl.startsWith("/")) pdfUrl = mirror + pdfUrl;
|
||||
|
||||
// Fetch the actual PDF
|
||||
const pdfResp = await fetch(pdfUrl, {
|
||||
headers: { "User-Agent": "Mozilla/5.0" },
|
||||
redirect: "follow",
|
||||
});
|
||||
|
||||
if (pdfResp.status === 200) {
|
||||
const headers = new Headers(pdfResp.headers);
|
||||
headers.set("Content-Type", "application/pdf");
|
||||
headers.set("X-Proxy-Source", "cloudflare-worker");
|
||||
headers.set("X-Proxy-Mirror", mirror);
|
||||
return new Response(pdfResp.body, { status: 200, headers });
|
||||
}
|
||||
} catch (e) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
return new Response(
|
||||
JSON.stringify({ error: "PDF not found on any mirror", doi }),
|
||||
{ status: 404, headers: { "Content-Type": "application/json" } }
|
||||
);
|
||||
}
|
||||
8
infra/cloudflared/worker/wrangler.toml
Normal file
8
infra/cloudflared/worker/wrangler.toml
Normal file
@@ -0,0 +1,8 @@
|
||||
name = "pdf-proxy"
|
||||
main = "worker.js"
|
||||
compatibility_date = "2026-03-25"
|
||||
|
||||
[vars]
|
||||
# SECRET is set via `wrangler secret put SECRET`
|
||||
|
||||
workers_dev = true
|
||||
Reference in New Issue
Block a user