Files
n8n-flows/flows/paywall-bypass.json
T
fkrebs 48f7c2588a feat: 📰 state for free articles + crawl4ai-first ordering
- crawl4ai direct first (detect if article actually needs paywall bypass)
- Only hit archive.ph when paywall detected (saves rate-limit budget)
- New 📰 icon for free articles enriched with full text
- 📖 reserved for actual paywall bypass via archive.ph

Final state set:
- 🔓 OA via Unpaywall
- 📰 Free article enriched
- 📖 Paywall bypassed via archive.ph
-  Paywalled, retrying
- 🚫 Paywalled, gave up after 3h
-  Fetch error
2026-05-28 07:21:47 +02:00

152 lines
13 KiB
JSON

{
"updatedAt": "2026-05-27T23:47:30.991Z",
"createdAt": "2026-05-27T22:09:54.846Z",
"id": "JqX7ejCYX1oAHvQ2",
"name": "paywall-bypass",
"description": null,
"active": true,
"isArchived": false,
"nodes": [
{
"id": "86fa097b-c230-484d-806d-45b421635c05",
"name": "Every 5 Minutes",
"type": "n8n-nodes-base.scheduleTrigger",
"position": [
-800,
0
],
"parameters": {
"rule": {
"interval": [
{
"field": "minutes",
"minutesInterval": 5
}
]
}
},
"typeVersion": 1.2
},
{
"id": "eb644570-cca3-4464-9b99-02084f199078",
"name": "Get Miniflux Entries",
"type": "n8n-nodes-base.httpRequest",
"position": [
-360,
0
],
"parameters": {
"url": "http://192.168.1.40:17002/v1/entries",
"method": "GET",
"options": {},
"sendQuery": true,
"sendHeaders": true,
"queryParameters": {
"parameters": [
{
"name": "status",
"value": "unread"
},
{
"name": "limit",
"value": "100"
},
{
"name": "order",
"value": "published_at"
},
{
"name": "direction",
"value": "desc"
}
]
},
"headerParameters": {
"parameters": [
{
"name": "X-Auth-Token",
"value": "3f439f2db55967ed5acd83398214d2aa76124927cde07115d945b9282597fecb"
}
]
}
},
"typeVersion": 4.2
},
{
"id": "f8024262-3163-4cd9-8233-3c91bfddcb01",
"name": "Process Entries",
"type": "n8n-nodes-base.code",
"position": [
-120,
0
],
"parameters": {
"mode": "runOnceForAllItems",
"jsCode": "const MINIFLUX_TOKEN = '3f439f2db55967ed5acd83398214d2aa76124927cde07115d945b9282597fecb';\nconst MINIFLUX_URL = 'http://192.168.1.40:17002';\n\nconst PAYWALL_NEWS = [\n 'spiegel.de','zeit.de','faz.net','sueddeutsche.de','handelsblatt.com',\n 'nytimes.com','wsj.com','ft.com','washingtonpost.com','thetimes.co.uk',\n 'wired.com','theatlantic.com','economist.com','bloomberg.com',\n 'hbr.org','foreignpolicy.com','newyorker.com','taz.de',\n 'telegraph.co.uk','businessinsider.com'\n];\nconst PAYWALL_SCIENCE = [\n 'nature.com','science.org','cell.com','nejm.org','thelancet.com',\n 'link.springer.com','springer.com','wiley.com','onlinelibrary.wiley.com',\n 'sciencedirect.com','pubs.acs.org','rsc.org','pubs.rsc.org',\n 'bmj.com','annualreviews.org','pnas.org',\n 'jamanetwork.com','ahajournals.org','tandfonline.com','academic.oup.com'\n];\n\nconst CONTENT_MIN = 1500;\nconst RETRY_LIMIT_MS = 3 * 60 * 60 * 1000;\n\nconst ALREADY_PROCESSED = /[\\u{1F513}\\u{1F4D6}\\u{1F4F0}\\u{1F6AB}\\u{274C}]$/u;\nconst ANY_STATUS = /\\s[\\u{1F513}\\u{1F4D6}\\u{1F4F0}\\u{1F6AB}\\u{274C}⏳]$/u;\n\nconst PAYWALL_MARKERS = [\n 'Sie können den Artikel leider nicht mehr aufrufen',\n 'Diesen Artikel weiterlesen mit',\n 'SPIEGEL plus',\n 'Jetzt abonnieren',\n 'Bereits Abonnent',\n 'Subscribe to read',\n 'Become a subscriber',\n 'To continue reading',\n 'Get full access',\n 'um diesen Artikel zu lesen',\n 'Mit Digital-Abo',\n];\n\nfunction hostname(url) {\n const m = String(url || '').match(/^[a-z]+:\\/\\/([^/?#:]+)/i);\n if (!m) return '';\n return m[1].toLowerCase().replace(/^www\\./, '');\n}\n\nfunction isPaywalled(text) {\n if (!text) return true;\n const clean = text.replace(/<[^>]+>/g, '').trim();\n if (clean.length < CONTENT_MIN) return true;\n return PAYWALL_MARKERS.some(m => text.includes(m));\n}\n\nfunction extractDoi(url) {\n let m = url.match(/doi\\.org\\/(10\\.\\d{4,}\\/[^\\s?#]+)/);\n if (m) return decodeURIComponent(m[1]);\n m = url.match(/\\/doi\\/(10\\.\\d{4,}\\/[^\\s?#]+)/);\n if (m) return decodeURIComponent(m[1].replace(/[?#].*$/, ''));\n return null;\n}\n\nfunction badge(source, type, status) {\n if (status === 'paywalled') return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #9E9E9E;padding:3px 8px;margin:0 0 1em\">⏳ Paywall detected — will retry archive.ph (latest fetch shown below)</p>';\n if (status === 'giveup') return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #757575;padding:3px 8px;margin:0 0 1em\">🚫 Paywall — gave up after 3h (latest attempt below)</p>';\n if (source === 'unpaywall') return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #4CAF50;padding:3px 8px;margin:0 0 1em\">🔓 Open access via Unpaywall</p>';\n if (source === 'free') return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #4CAF50;padding:3px 8px;margin:0 0 1em\">📰 Free article — full text via crawl4ai</p>';\n if (source === 'archive') return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #2196F3;padding:3px 8px;margin:0 0 1em\">📖 Full text via archive.ph</p>';\n if (type === 'science') return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #2196F3;padding:3px 8px;margin:0 0 1em\">📖 Full text via crawl4ai (not OA)</p>';\n return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #FF9800;padding:3px 8px;margin:0 0 1em\">📖 Paywall bypass via crawl4ai</p>';\n}\n\nfunction mdToHtml(md) {\n if (!md) return '';\n const lines = md.split('\\n').filter(l => {\n const t = l.trim();\n if (!t) return true;\n if (/Zur Merkliste hinzufügen|Artikel anhören|Bild vergrößern|Bild schließen|Link kopieren|Weitere Optionen zum Teilen|Mehr lesen über|Verwandte Artikel/.test(t)) return false;\n if (/^\\s*\\*\\s*\\[?\\s*(X\\.com|Facebook|Messenger|WhatsApp|E-Mail|Threads|Mastodon|Telegram)\\b/i.test(t)) return false;\n return true;\n });\n let s = lines.join('\\n');\n s = s.replace(/!\\[([^\\]]*)\\]\\(([^)\\s]+)[^)]*\\)/g, '<img alt=\"$1\" src=\"$2\" style=\"max-width:100%;height:auto\">');\n s = s.replace(/\\[([^\\]]+)\\]\\(([^)\\s]+)[^)]*\\)/g, '<a href=\"$2\">$1</a>');\n s = s.replace(/\\*\\*([^*\\n]+)\\*\\*/g, '<strong>$1</strong>');\n s = s.replace(/`([^`\\n]+)`/g, '<code>$1</code>');\n const out = [];\n let inList = false, inPara = false;\n for (const raw of s.split('\\n')) {\n const line = raw.trim();\n if (!line) {\n if (inList) { out.push('</ul>'); inList = false; }\n if (inPara) { out.push('</p>'); inPara = false; }\n continue;\n }\n const h = line.match(/^(#{1,6})\\s+(.+)$/);\n if (h) {\n if (inList) { out.push('</ul>'); inList = false; }\n if (inPara) { out.push('</p>'); inPara = false; }\n const n = h[1].length;\n out.push(`<h${n}>${h[2]}</h${n}>`);\n continue;\n }\n const li = line.match(/^[*\\-]\\s+(.+)$/);\n if (li) {\n if (inPara) { out.push('</p>'); inPara = false; }\n if (!inList) { out.push('<ul>'); inList = true; }\n out.push(`<li>${li[1]}</li>`);\n continue;\n }\n if (inList) { out.push('</ul>'); inList = false; }\n if (!inPara) { out.push('<p>'); inPara = true; }\n out.push(line);\n }\n if (inList) out.push('</ul>');\n if (inPara) out.push('</p>');\n return out.join('\\n');\n}\n\nconst self = this;\nasync function fetchCrawl4ai(url) {\n try {\n const cr = await self.helpers.httpRequest({\n method: 'POST',\n url: 'http://mcp-crawl4ai:11235/md',\n headers: { 'Content-Type': 'application/json' },\n body: { url, filter: 'fit' },\n json: true,\n timeout: 60000,\n });\n if (cr && cr.success && cr.markdown) return cr.markdown;\n } catch (_) {}\n return '';\n}\n\nasync function fetchUnpaywall(doi) {\n try {\n const up = await self.helpers.httpRequest({\n method: 'GET',\n url: `https://api.unpaywall.org/v2/${encodeURIComponent(doi)}?email=fkrebs@nucli.de`,\n json: true,\n timeout: 15000,\n });\n if (up.is_oa && up.best_oa_location) {\n return up.best_oa_location.url || up.best_oa_location.url_for_pdf || '';\n }\n } catch (_) {}\n return '';\n}\n\nasync function getArchiveSnapshotUrl(url) {\n try {\n const tm = await self.helpers.httpRequest({\n method: 'GET',\n url: 'https://archive.ph/timemap/' + url,\n headers: { 'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 Chrome/120.0' },\n timeout: 30000,\n });\n const txt = typeof tm === 'string' ? tm : (tm && tm.body) || '';\n const matches = [...txt.matchAll(/<([^>]+)>;\\s*rel=\"[^\"]*memento[^\"]*\"/g)];\n if (matches.length > 0) return matches[matches.length - 1][1];\n } catch (_) {}\n return '';\n}\n\nconst entries = $('Get Miniflux Entries').first().json.entries || [];\nconst results = [];\nconst stats = { skipped_done: 0, skipped_nopaywall: 0, attempted: 0, success: 0, retry: 0, giveup: 0, error: 0 };\n\nfor (const e of entries) {\n if (ALREADY_PROCESSED.test(e.title)) { stats.skipped_done++; continue; }\n\n const host = hostname(e.url);\n const isScience = PAYWALL_SCIENCE.some(d => host === d || host.endsWith('.' + d));\n const isNews = PAYWALL_NEWS.some(d => host === d || host.endsWith('.' + d));\n const isShort = (e.content || '').replace(/<[^>]+>/g, '').trim().length < 800;\n\n if (!isScience && !isNews && !isShort) { stats.skipped_nopaywall++; continue; }\n stats.attempted++;\n\n const type = isScience ? 'science' : isNews ? 'news' : 'short';\n const publishedMs = new Date(e.published_at).getTime();\n const ageMs = Date.now() - publishedMs;\n const exhausted = ageMs >= RETRY_LIMIT_MS;\n\n let markdown = '';\n let source = '';\n\n if (isScience) {\n const doi = extractDoi(e.url);\n if (doi) {\n const oaUrl = await fetchUnpaywall(doi);\n if (oaUrl) {\n const md = await fetchCrawl4ai(oaUrl);\n if (md && !isPaywalled(md)) { markdown = md; source = 'unpaywall'; }\n }\n }\n }\n\n // Try crawl4ai direct first to find out if it's actually paywalled\n if (!source) {\n const md = await fetchCrawl4ai(e.url);\n if (md) {\n if (!isPaywalled(md)) {\n markdown = md;\n source = 'free';\n } else {\n markdown = md; // keep for retry/giveup display\n }\n }\n }\n\n // If paywalled (or no content), try archive.ph for bypass\n if (!source) {\n const snapUrl = await getArchiveSnapshotUrl(e.url);\n if (snapUrl) {\n const md = await fetchCrawl4ai(snapUrl);\n if (md && !isPaywalled(md)) { markdown = md; source = 'archive'; }\n }\n }\n\n const cleanTitle = e.title.replace(ANY_STATUS, '');\n let newTitle, content, status;\n\n if (source) {\n content = badge(source, type, 'ok') + mdToHtml(markdown);\n const emoji = source === 'unpaywall' ? '🔓' : source === 'free' ? '📰' : '📖';\n newTitle = cleanTitle + ' ' + emoji;\n stats.success++;\n status = 'updated';\n } else if (markdown && exhausted) {\n content = badge('', type, 'giveup') + mdToHtml(markdown);\n newTitle = cleanTitle + ' 🚫';\n stats.giveup++;\n status = 'gaveup';\n } else if (markdown) {\n content = badge('', type, 'paywalled') + mdToHtml(markdown);\n newTitle = cleanTitle + ' ⏳';\n stats.retry++;\n status = 'retry';\n } else {\n content = null;\n newTitle = cleanTitle + ' ❌';\n stats.error++;\n status = 'error';\n }\n\n const body = content !== null ? { content, title: newTitle } : { title: newTitle };\n try {\n await self.helpers.httpRequest({\n method: 'PUT',\n url: `${MINIFLUX_URL}/v1/entries/${e.id}`,\n headers: { 'X-Auth-Token': MINIFLUX_TOKEN, 'Content-Type': 'application/json' },\n body,\n json: true,\n timeout: 15000,\n });\n results.push({ json: { entry_id: e.id, title: newTitle.slice(0, 80), status, source, type, host } });\n } catch (err) {\n results.push({ json: { entry_id: e.id, title: e.title.slice(0, 80), status: 'put_failed', error: String(err).slice(0, 200) } });\n }\n}\n\nresults.push({ json: { stats } });\nreturn results;"
},
"typeVersion": 2
}
],
"connections": {
"Every 5 Minutes": {
"main": [
[
{
"node": "Get Miniflux Entries",
"type": "main",
"index": 0
}
]
]
},
"Get Miniflux Entries": {
"main": [
[
{
"node": "Process Entries",
"type": "main",
"index": 0
}
]
]
}
},
"settings": {
"executionOrder": "v1"
},
"staticData": {
"node:Every 5 Minutes": {
"recurrenceRules": []
}
},
"meta": null,
"pinData": null,
"versionId": "0a3f5003-653c-4494-96e9-a779815a1b9c",
"activeVersionId": "0a3f5003-653c-4494-96e9-a779815a1b9c",
"versionCounter": 17,
"triggerCount": 1,
"tags": [],
"shared": [
{
"updatedAt": "2026-05-27T22:09:54.846Z",
"createdAt": "2026-05-27T22:09:54.846Z",
"role": "workflow:owner",
"workflowId": "JqX7ejCYX1oAHvQ2",
"projectId": "hkIaTyp1CKVS8ywT",
"project": {
"updatedAt": "2026-03-09T07:21:36.510Z",
"createdAt": "2026-03-09T07:20:59.502Z",
"id": "hkIaTyp1CKVS8ywT",
"name": "Florian Krebs <fkrebs@nucli.de>",
"type": "personal",
"icon": null,
"description": null,
"creatorId": "8af5c813-f89c-4b83-bcf8-5d59864a038a"
}
}
],
"versionMetadata": {
"name": "paywall-bypass",
"description": null
}
}