feat: Cloudflare Worker PDF proxy + headless batch retrieval (fixes #273)
Some checks failed
CI / skinny-install (aco) (push) Successful in 2m19s
CI / lint-test (push) Failing after 3m13s
CI / skinny-install (api) (push) Successful in 1m45s
CI / skinny-install (bcda) (push) Successful in 1m41s
CI / skinny-install (bib) (push) Successful in 2m2s
CI / skinny-install (bls) (push) Successful in 2m1s
CI / skinny-install (ccw) (push) Successful in 1m37s
CI / skinny-install (cli) (push) Successful in 2m7s
CI / skinny-install (cms) (push) Successful in 1m40s
CI / skinny-install (conf) (push) Successful in 1m57s
CI / skinny-install (perf) (push) Successful in 1m41s
CI / skinny-install (rex) (push) Successful in 1m46s
CI / skinny-install (pfs) (push) Successful in 2m37s
Deploy / build-scan-report (push) Failing after 2m34s
Some checks failed
CI / skinny-install (aco) (push) Successful in 2m19s
CI / lint-test (push) Failing after 3m13s
CI / skinny-install (api) (push) Successful in 1m45s
CI / skinny-install (bcda) (push) Successful in 1m41s
CI / skinny-install (bib) (push) Successful in 2m2s
CI / skinny-install (bls) (push) Successful in 2m1s
CI / skinny-install (ccw) (push) Successful in 1m37s
CI / skinny-install (cli) (push) Successful in 2m7s
CI / skinny-install (cms) (push) Successful in 1m40s
CI / skinny-install (conf) (push) Successful in 1m57s
CI / skinny-install (perf) (push) Successful in 1m41s
CI / skinny-install (rex) (push) Successful in 1m46s
CI / skinny-install (pfs) (push) Successful in 2m37s
Deploy / build-scan-report (push) Failing after 2m34s
Cloudflare Worker (pdf-proxy) deployed to bypass ISP-level SciHub blocking: - Worker fetches PDFs from SciHub mirrors on Cloudflare's edge network - Authenticated via X-Proxy-Key header - Routes: /pdf?doi=, /fetch?url=, /health - Deployed at pdf-proxy.lite7889.workers.dev fetch_pdfs.py updated with 5-phase waterfall: 1. Unpaywall (legal OA, ~4% for this corpus) 2. PMC (free, PMCID-based) 3. Semantic Scholar (batch 500) 4. CF Worker → SciHub (72% hit rate on first 50, running full batch) 5. SciHub direct (fallback with proxy) Also: cloudflared tunnel (stack-proxy) with WARP routing in compose.yml, CoreDNS sci-hub domain routing via unfiltered DNS. Test: 36/50 PDFs (72%) retrieved in first batch via Worker.
This commit is contained in:
10
compose.yml
10
compose.yml
@@ -624,6 +624,16 @@ services:
|
|||||||
- no-new-privileges:true
|
- no-new-privileges:true
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
|
|
||||||
|
cloudflared:
|
||||||
|
image: cloudflare/cloudflared:latest
|
||||||
|
container_name: cloudflared
|
||||||
|
command: tunnel --config /home/nonroot/.cloudflared/config.yml run
|
||||||
|
networks:
|
||||||
|
- gateway
|
||||||
|
volumes:
|
||||||
|
- ./infra/cloudflared:/home/nonroot/.cloudflared:ro
|
||||||
|
restart: unless-stopped
|
||||||
|
|
||||||
volumes:
|
volumes:
|
||||||
postgres_data:
|
postgres_data:
|
||||||
rustfs_data:
|
rustfs_data:
|
||||||
|
|||||||
@@ -155,11 +155,47 @@ SCIHUB_MIRRORS = [
|
|||||||
# Proxy for SciHub (set SCIHUB_PROXY env var, e.g. socks5://localhost:1080)
|
# Proxy for SciHub (set SCIHUB_PROXY env var, e.g. socks5://localhost:1080)
|
||||||
SCIHUB_PROXY = os.environ.get("SCIHUB_PROXY", "")
|
SCIHUB_PROXY = os.environ.get("SCIHUB_PROXY", "")
|
||||||
|
|
||||||
|
# Cloudflare Worker proxy (preferred — bypasses ISP blocking via CF edge)
|
||||||
|
CF_WORKER_URL = os.environ.get(
|
||||||
|
"PDF_PROXY_URL", "https://pdf-proxy.lite7889.workers.dev"
|
||||||
|
)
|
||||||
|
CF_WORKER_SECRET = os.environ.get("PDF_PROXY_SECRET", "")
|
||||||
|
|
||||||
|
|
||||||
|
def fetch_via_worker(doi: str, dest: Path) -> bool:
|
||||||
|
"""Fetch PDF via Cloudflare Worker proxy. Returns True on success.
|
||||||
|
|
||||||
|
The Worker fetches from SciHub on Cloudflare's edge network,
|
||||||
|
bypassing ISP-level DNS/IP blocking.
|
||||||
|
"""
|
||||||
|
if not CF_WORKER_SECRET:
|
||||||
|
return False
|
||||||
|
try:
|
||||||
|
resp = httpx.get(
|
||||||
|
f"{CF_WORKER_URL}/pdf",
|
||||||
|
params={"doi": doi},
|
||||||
|
headers={
|
||||||
|
"X-Proxy-Key": CF_WORKER_SECRET,
|
||||||
|
"User-Agent": USER_AGENT,
|
||||||
|
},
|
||||||
|
timeout=45,
|
||||||
|
)
|
||||||
|
if resp.status_code != 200:
|
||||||
|
return False
|
||||||
|
if b"%PDF" not in resp.content[:1024]:
|
||||||
|
return False
|
||||||
|
dest.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
dest.write_bytes(resp.content)
|
||||||
|
return True
|
||||||
|
except Exception:
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
def fetch_scihub(doi: str) -> str | None:
|
def fetch_scihub(doi: str) -> str | None:
|
||||||
"""Get PDF URL from SciHub. Returns direct PDF URL or None.
|
"""Get PDF URL from SciHub directly. Returns direct PDF URL or None.
|
||||||
|
|
||||||
Set SCIHUB_PROXY env var for networks where SciHub is blocked.
|
Set SCIHUB_PROXY env var for networks where SciHub is blocked.
|
||||||
|
Prefer fetch_via_worker() which uses Cloudflare edge.
|
||||||
"""
|
"""
|
||||||
transport = httpx.HTTPTransport(proxy=SCIHUB_PROXY) if SCIHUB_PROXY else None
|
transport = httpx.HTTPTransport(proxy=SCIHUB_PROXY) if SCIHUB_PROXY else None
|
||||||
client_kwargs = {"transport": transport} if transport else {}
|
client_kwargs = {"transport": transport} if transport else {}
|
||||||
@@ -297,7 +333,7 @@ def get_items_needing_pdfs(con: sqlite3.Connection) -> list[dict]:
|
|||||||
def main() -> None:
|
def main() -> None:
|
||||||
parser = argparse.ArgumentParser(description="Headless PDF retrieval")
|
parser = argparse.ArgumentParser(description="Headless PDF retrieval")
|
||||||
parser.add_argument("--limit", type=int, default=0, help="Max items to process (0=all)")
|
parser.add_argument("--limit", type=int, default=0, help="Max items to process (0=all)")
|
||||||
parser.add_argument("--source", choices=["all", "unpaywall", "pmc", "s2", "scihub"],
|
parser.add_argument("--source", choices=["all", "unpaywall", "pmc", "s2", "worker", "scihub"],
|
||||||
default="all", help="Which source to use")
|
default="all", help="Which source to use")
|
||||||
parser.add_argument("--proxy", help="SOCKS/HTTP proxy for SciHub (e.g. socks5://localhost:1080)")
|
parser.add_argument("--proxy", help="SOCKS/HTTP proxy for SciHub (e.g. socks5://localhost:1080)")
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
@@ -388,10 +424,33 @@ def main() -> None:
|
|||||||
|
|
||||||
print(f" Semantic Scholar: {stats.get('s2', 0)} PDFs")
|
print(f" Semantic Scholar: {stats.get('s2', 0)} PDFs")
|
||||||
|
|
||||||
# Phase 4: SciHub (fallback, may hit CAPTCHAs or be blocked)
|
# Phase 4: Cloudflare Worker → SciHub (bypasses ISP blocking)
|
||||||
remaining = [it for it in items if not it.get("_done")]
|
remaining = [it for it in items if not it.get("_done")]
|
||||||
if args.source in ("all", "scihub") and remaining:
|
if args.source in ("all", "worker", "scihub") and remaining and CF_WORKER_SECRET:
|
||||||
print(f"\n--- Phase 4: SciHub ({len(remaining)} remaining) ---")
|
print(f"\n--- Phase 4: CF Worker → SciHub ({len(remaining)} remaining) ---")
|
||||||
|
consecutive_failures = 0
|
||||||
|
for i, item in enumerate(remaining):
|
||||||
|
if consecutive_failures >= 20:
|
||||||
|
print(f" Stopping: {consecutive_failures} consecutive failures")
|
||||||
|
break
|
||||||
|
dest = tmp_dir / f"{item['item_id']}.pdf"
|
||||||
|
if fetch_via_worker(item["doi"], dest):
|
||||||
|
attach_pdf_to_item(con, item["item_id"], dest, item["doi"])
|
||||||
|
stats["worker"] = stats.get("worker", 0) + 1
|
||||||
|
item["_done"] = True
|
||||||
|
consecutive_failures = 0
|
||||||
|
else:
|
||||||
|
consecutive_failures += 1
|
||||||
|
if (i + 1) % 25 == 0:
|
||||||
|
print(f" {i + 1}/{len(remaining)} worker={stats.get('worker', 0)} "
|
||||||
|
f"fails={consecutive_failures}")
|
||||||
|
time.sleep(1.5) # Be polite to Worker + SciHub
|
||||||
|
print(f" CF Worker: {stats.get('worker', 0)} PDFs")
|
||||||
|
|
||||||
|
# Phase 5: SciHub direct (fallback if Worker unavailable)
|
||||||
|
remaining = [it for it in items if not it.get("_done")]
|
||||||
|
if args.source in ("scihub",) and remaining and not CF_WORKER_SECRET:
|
||||||
|
print(f"\n--- Phase 5: SciHub direct ({len(remaining)} remaining) ---")
|
||||||
if not SCIHUB_PROXY:
|
if not SCIHUB_PROXY:
|
||||||
print(" WARNING: No proxy set. SciHub may be blocked from this network.")
|
print(" WARNING: No proxy set. SciHub may be blocked from this network.")
|
||||||
print(" Set SCIHUB_PROXY or use --proxy socks5://host:port")
|
print(" Set SCIHUB_PROXY or use --proxy socks5://host:port")
|
||||||
@@ -422,14 +481,16 @@ def main() -> None:
|
|||||||
print(f" SciHub: {stats['scihub']} PDFs")
|
print(f" SciHub: {stats['scihub']} PDFs")
|
||||||
|
|
||||||
stats["failed"] = len([it for it in items if not it.get("_done")])
|
stats["failed"] = len([it for it in items if not it.get("_done")])
|
||||||
total_found = stats["unpaywall"] + stats["pmc"] + stats.get("s2", 0) + stats["scihub"]
|
total_found = (stats["unpaywall"] + stats["pmc"] + stats.get("s2", 0)
|
||||||
|
+ stats.get("worker", 0) + stats.get("scihub", 0))
|
||||||
|
|
||||||
print(f"\n{'=' * 60}")
|
print(f"\n{'=' * 60}")
|
||||||
print(f"Results: {total_found} PDFs downloaded")
|
print(f"Results: {total_found} PDFs downloaded")
|
||||||
print(f" Unpaywall: {stats['unpaywall']}")
|
print(f" Unpaywall: {stats['unpaywall']}")
|
||||||
print(f" PMC: {stats['pmc']}")
|
print(f" PMC: {stats['pmc']}")
|
||||||
print(f" S2: {stats.get('s2', 0)}")
|
print(f" S2: {stats.get('s2', 0)}")
|
||||||
print(f" SciHub: {stats['scihub']}")
|
print(f" CF Worker: {stats.get('worker', 0)}")
|
||||||
|
print(f" SciHub: {stats.get('scihub', 0)}")
|
||||||
print(f" Failed: {stats['failed']}")
|
print(f" Failed: {stats['failed']}")
|
||||||
print(f" Coverage: {total_found * 100 // max(len(items), 1)}%")
|
print(f" Coverage: {total_found * 100 // max(len(items), 1)}%")
|
||||||
|
|
||||||
|
|||||||
10
infra/cloudflared/config.yml
Normal file
10
infra/cloudflared/config.yml
Normal file
@@ -0,0 +1,10 @@
|
|||||||
|
tunnel: 1389035e-d3ba-4a4f-969d-a369c07ee057
|
||||||
|
credentials-file: /home/nonroot/.cloudflared/1389035e-d3ba-4a4f-969d-a369c07ee057.json
|
||||||
|
|
||||||
|
warp-routing:
|
||||||
|
enabled: true
|
||||||
|
|
||||||
|
ingress:
|
||||||
|
- service: socks-proxy
|
||||||
|
originRequest:
|
||||||
|
connectTimeout: 30s
|
||||||
6
infra/cloudflared/worker/.wrangler/cache/wrangler-account.json
vendored
Normal file
6
infra/cloudflared/worker/.wrangler/cache/wrangler-account.json
vendored
Normal file
@@ -0,0 +1,6 @@
|
|||||||
|
{
|
||||||
|
"account": {
|
||||||
|
"id": "89f36257ec24ba34152e7c82a66335a3",
|
||||||
|
"name": "Lite@fhirworx.io's Account"
|
||||||
|
}
|
||||||
|
}
|
||||||
139
infra/cloudflared/worker/worker.js
Normal file
139
infra/cloudflared/worker/worker.js
Normal file
@@ -0,0 +1,139 @@
|
|||||||
|
/**
|
||||||
|
* PDF Fetch Proxy — Cloudflare Worker
|
||||||
|
*
|
||||||
|
* Relays HTTP requests through Cloudflare's edge network,
|
||||||
|
* bypassing ISP-level DNS/IP blocking of academic sources.
|
||||||
|
*
|
||||||
|
* Usage:
|
||||||
|
* GET https://<worker>/fetch?url=https://sci-hub.st/10.1234/example
|
||||||
|
* GET https://<worker>/pdf?doi=10.1234/example
|
||||||
|
*
|
||||||
|
* Security: requires X-Proxy-Key header matching the SECRET binding.
|
||||||
|
*/
|
||||||
|
|
||||||
|
const SCIHUB_MIRRORS = [
|
||||||
|
"https://sci-hub.st",
|
||||||
|
"https://sci-hub.ru",
|
||||||
|
"https://sci-hub.se",
|
||||||
|
"https://sci-hub.ren",
|
||||||
|
"https://sci-hub.ee",
|
||||||
|
];
|
||||||
|
|
||||||
|
export default {
|
||||||
|
async fetch(request, env) {
|
||||||
|
// Auth check
|
||||||
|
const key = request.headers.get("X-Proxy-Key");
|
||||||
|
if (!key || key !== env.SECRET) {
|
||||||
|
return new Response("Unauthorized", { status: 401 });
|
||||||
|
}
|
||||||
|
|
||||||
|
const url = new URL(request.url);
|
||||||
|
|
||||||
|
// Route: /fetch?url=<encoded_url> — generic proxy
|
||||||
|
if (url.pathname === "/fetch") {
|
||||||
|
const targetUrl = url.searchParams.get("url");
|
||||||
|
if (!targetUrl) {
|
||||||
|
return new Response("Missing url parameter", { status: 400 });
|
||||||
|
}
|
||||||
|
return proxyFetch(targetUrl);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Route: /pdf?doi=<doi> — SciHub PDF resolver
|
||||||
|
if (url.pathname === "/pdf") {
|
||||||
|
const doi = url.searchParams.get("doi");
|
||||||
|
if (!doi) {
|
||||||
|
return new Response("Missing doi parameter", { status: 400 });
|
||||||
|
}
|
||||||
|
return fetchPdfFromScihub(doi);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Route: /health
|
||||||
|
if (url.pathname === "/health") {
|
||||||
|
return new Response(JSON.stringify({ status: "ok", ts: Date.now() }), {
|
||||||
|
headers: { "Content-Type": "application/json" },
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
return new Response("Not found. Use /fetch?url=, /pdf?doi=, or /health", {
|
||||||
|
status: 404,
|
||||||
|
});
|
||||||
|
},
|
||||||
|
};
|
||||||
|
|
||||||
|
async function proxyFetch(targetUrl) {
|
||||||
|
try {
|
||||||
|
const resp = await fetch(targetUrl, {
|
||||||
|
headers: { "User-Agent": "Mozilla/5.0 (compatible; stack-proxy/1.0)" },
|
||||||
|
redirect: "follow",
|
||||||
|
});
|
||||||
|
// Stream the response back with original headers
|
||||||
|
const headers = new Headers(resp.headers);
|
||||||
|
headers.set("X-Proxy-Source", "cloudflare-worker");
|
||||||
|
return new Response(resp.body, {
|
||||||
|
status: resp.status,
|
||||||
|
headers,
|
||||||
|
});
|
||||||
|
} catch (e) {
|
||||||
|
return new Response(JSON.stringify({ error: e.message }), {
|
||||||
|
status: 502,
|
||||||
|
headers: { "Content-Type": "application/json" },
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async function fetchPdfFromScihub(doi) {
|
||||||
|
for (const mirror of SCIHUB_MIRRORS) {
|
||||||
|
try {
|
||||||
|
const pageUrl = `${mirror}/${doi}`;
|
||||||
|
const resp = await fetch(pageUrl, {
|
||||||
|
headers: { "User-Agent": "Mozilla/5.0 (compatible; stack-proxy/1.0)" },
|
||||||
|
redirect: "follow",
|
||||||
|
});
|
||||||
|
if (resp.status !== 200) continue;
|
||||||
|
|
||||||
|
const html = await resp.text();
|
||||||
|
|
||||||
|
// Extract PDF URL from SciHub page
|
||||||
|
let pdfUrl = null;
|
||||||
|
const patterns = [
|
||||||
|
/id="pdf"[^>]*src="([^"]+)"/,
|
||||||
|
/<iframe[^>]*src="([^"]*\.pdf[^"]*)"/,
|
||||||
|
/<embed[^>]*src="([^"]*\.pdf[^"]*)"/,
|
||||||
|
];
|
||||||
|
for (const pat of patterns) {
|
||||||
|
const m = html.match(pat);
|
||||||
|
if (m) {
|
||||||
|
pdfUrl = m[1];
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!pdfUrl) continue;
|
||||||
|
|
||||||
|
// Normalize URL
|
||||||
|
if (pdfUrl.startsWith("//")) pdfUrl = "https:" + pdfUrl;
|
||||||
|
else if (pdfUrl.startsWith("/")) pdfUrl = mirror + pdfUrl;
|
||||||
|
|
||||||
|
// Fetch the actual PDF
|
||||||
|
const pdfResp = await fetch(pdfUrl, {
|
||||||
|
headers: { "User-Agent": "Mozilla/5.0" },
|
||||||
|
redirect: "follow",
|
||||||
|
});
|
||||||
|
|
||||||
|
if (pdfResp.status === 200) {
|
||||||
|
const headers = new Headers(pdfResp.headers);
|
||||||
|
headers.set("Content-Type", "application/pdf");
|
||||||
|
headers.set("X-Proxy-Source", "cloudflare-worker");
|
||||||
|
headers.set("X-Proxy-Mirror", mirror);
|
||||||
|
return new Response(pdfResp.body, { status: 200, headers });
|
||||||
|
}
|
||||||
|
} catch (e) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return new Response(
|
||||||
|
JSON.stringify({ error: "PDF not found on any mirror", doi }),
|
||||||
|
{ status: 404, headers: { "Content-Type": "application/json" } }
|
||||||
|
);
|
||||||
|
}
|
||||||
8
infra/cloudflared/worker/wrangler.toml
Normal file
8
infra/cloudflared/worker/wrangler.toml
Normal file
@@ -0,0 +1,8 @@
|
|||||||
|
name = "pdf-proxy"
|
||||||
|
main = "worker.js"
|
||||||
|
compatibility_date = "2026-03-25"
|
||||||
|
|
||||||
|
[vars]
|
||||||
|
# SECRET is set via `wrangler secret put SECRET`
|
||||||
|
|
||||||
|
workers_dev = true
|
||||||
Reference in New Issue
Block a user