9a4b2788e4
User can click "Check archive.ph manually →" from inside Miniflux to see if a snapshot exists or to trigger one. Useful when the automated archive.ph lookup hit rate limits or no snapshot was ready yet.
152 lines
14 KiB
JSON
152 lines
14 KiB
JSON
{
|
|
"updatedAt": "2026-05-27T23:47:30.991Z",
|
|
"createdAt": "2026-05-27T22:09:54.846Z",
|
|
"id": "JqX7ejCYX1oAHvQ2",
|
|
"name": "paywall-bypass",
|
|
"description": null,
|
|
"active": true,
|
|
"isArchived": false,
|
|
"nodes": [
|
|
{
|
|
"id": "86fa097b-c230-484d-806d-45b421635c05",
|
|
"name": "Every 5 Minutes",
|
|
"type": "n8n-nodes-base.scheduleTrigger",
|
|
"position": [
|
|
-800,
|
|
0
|
|
],
|
|
"parameters": {
|
|
"rule": {
|
|
"interval": [
|
|
{
|
|
"field": "minutes",
|
|
"minutesInterval": 5
|
|
}
|
|
]
|
|
}
|
|
},
|
|
"typeVersion": 1.2
|
|
},
|
|
{
|
|
"id": "eb644570-cca3-4464-9b99-02084f199078",
|
|
"name": "Get Miniflux Entries",
|
|
"type": "n8n-nodes-base.httpRequest",
|
|
"position": [
|
|
-360,
|
|
0
|
|
],
|
|
"parameters": {
|
|
"url": "http://192.168.1.40:17002/v1/entries",
|
|
"method": "GET",
|
|
"options": {},
|
|
"sendQuery": true,
|
|
"sendHeaders": true,
|
|
"queryParameters": {
|
|
"parameters": [
|
|
{
|
|
"name": "status",
|
|
"value": "unread"
|
|
},
|
|
{
|
|
"name": "limit",
|
|
"value": "100"
|
|
},
|
|
{
|
|
"name": "order",
|
|
"value": "published_at"
|
|
},
|
|
{
|
|
"name": "direction",
|
|
"value": "desc"
|
|
}
|
|
]
|
|
},
|
|
"headerParameters": {
|
|
"parameters": [
|
|
{
|
|
"name": "X-Auth-Token",
|
|
"value": "3f439f2db55967ed5acd83398214d2aa76124927cde07115d945b9282597fecb"
|
|
}
|
|
]
|
|
}
|
|
},
|
|
"typeVersion": 4.2
|
|
},
|
|
{
|
|
"id": "f8024262-3163-4cd9-8233-3c91bfddcb01",
|
|
"name": "Process Entries",
|
|
"type": "n8n-nodes-base.code",
|
|
"position": [
|
|
-120,
|
|
0
|
|
],
|
|
"parameters": {
|
|
"mode": "runOnceForAllItems",
|
|
"jsCode": "const MINIFLUX_TOKEN = '3f439f2db55967ed5acd83398214d2aa76124927cde07115d945b9282597fecb';\nconst MINIFLUX_URL = 'http://192.168.1.40:17002';\n\nconst PAYWALL_NEWS = [\n 'spiegel.de','zeit.de','faz.net','sueddeutsche.de','handelsblatt.com',\n 'nytimes.com','wsj.com','ft.com','washingtonpost.com','thetimes.co.uk',\n 'wired.com','theatlantic.com','economist.com','bloomberg.com',\n 'hbr.org','foreignpolicy.com','newyorker.com','taz.de',\n 'telegraph.co.uk','businessinsider.com'\n];\nconst PAYWALL_SCIENCE = [\n 'nature.com','science.org','cell.com','nejm.org','thelancet.com',\n 'link.springer.com','springer.com','wiley.com','onlinelibrary.wiley.com',\n 'sciencedirect.com','pubs.acs.org','rsc.org','pubs.rsc.org',\n 'bmj.com','annualreviews.org','pnas.org',\n 'jamanetwork.com','ahajournals.org','tandfonline.com','academic.oup.com'\n];\n\nconst CONTENT_MIN = 1500;\nconst RETRY_LIMIT_MS = 3 * 60 * 60 * 1000;\n\nconst ALREADY_PROCESSED = /[\\u{1F513}\\u{1F4D6}\\u{1F4F0}\\u{1F6AB}\\u{274C}]$/u;\nconst ANY_STATUS = /\\s[\\u{1F513}\\u{1F4D6}\\u{1F4F0}\\u{1F6AB}\\u{274C}⏳]$/u;\n\nconst PAYWALL_MARKERS = [\n 'Sie können den Artikel leider nicht mehr aufrufen',\n 'Diesen Artikel weiterlesen mit',\n 'SPIEGEL plus',\n 'Jetzt abonnieren',\n 'Bereits Abonnent',\n 'Subscribe to read',\n 'Become a subscriber',\n 'To continue reading',\n 'Get full access',\n 'um diesen Artikel zu lesen',\n 'Mit Digital-Abo',\n];\n\nfunction hostname(url) {\n const m = String(url || '').match(/^[a-z]+:\\/\\/([^/?#:]+)/i);\n if (!m) return '';\n return m[1].toLowerCase().replace(/^www\\./, '');\n}\n\nfunction isPaywalled(text) {\n if (!text) return true;\n const clean = text.replace(/<[^>]+>/g, '').trim();\n if (clean.length < CONTENT_MIN) return true;\n return PAYWALL_MARKERS.some(m => text.includes(m));\n}\n\nfunction extractDoi(url) {\n let m = url.match(/doi\\.org\\/(10\\.\\d{4,}\\/[^\\s?#]+)/);\n if (m) return decodeURIComponent(m[1]);\n m = url.match(/\\/doi\\/(10\\.\\d{4,}\\/[^\\s?#]+)/);\n if (m) return decodeURIComponent(m[1].replace(/[?#].*$/, ''));\n return null;\n}\n\nfunction badge(source, type, status, url) {\n const archiveLink = url ? '<a href=\"https://archive.ph/newest/' + encodeURI(url) + '\" target=\"_blank\" rel=\"noopener\">Check archive.ph manually →</a>' : '';\n if (status === 'paywalled') return '<p style=\"font-size:0.85em;color:#555;border-left:3px solid #9E9E9E;padding:6px 10px;margin:0 0 1em;background:#fafafa\">⏳ Paywall detected — will retry archive.ph<br>' + archiveLink + '</p>';\n if (status === 'giveup') return '<p style=\"font-size:0.85em;color:#555;border-left:3px solid #757575;padding:6px 10px;margin:0 0 1em;background:#fafafa\">🚫 Paywall — gave up after 3h<br>' + archiveLink + '</p>';\n if (source === 'unpaywall') return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #4CAF50;padding:3px 8px;margin:0 0 1em\">🔓 Open access via Unpaywall</p>';\n if (source === 'free') return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #4CAF50;padding:3px 8px;margin:0 0 1em\">📰 Free article — full text via crawl4ai</p>';\n if (source === 'archive') return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #2196F3;padding:3px 8px;margin:0 0 1em\">📖 Full text via archive.ph</p>';\n if (type === 'science') return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #2196F3;padding:3px 8px;margin:0 0 1em\">📖 Full text via crawl4ai (not OA)</p>';\n return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #FF9800;padding:3px 8px;margin:0 0 1em\">📖 Paywall bypass via crawl4ai</p>';\n}\n\nfunction mdToHtml(md) {\n if (!md) return '';\n const lines = md.split('\\n').filter(l => {\n const t = l.trim();\n if (!t) return true;\n if (/Zur Merkliste hinzufügen|Artikel anhören|Bild vergrößern|Bild schließen|Link kopieren|Weitere Optionen zum Teilen|Mehr lesen über|Verwandte Artikel/.test(t)) return false;\n if (/^\\s*\\*\\s*\\[?\\s*(X\\.com|Facebook|Messenger|WhatsApp|E-Mail|Threads|Mastodon|Telegram)\\b/i.test(t)) return false;\n return true;\n });\n let s = lines.join('\\n');\n s = s.replace(/!\\[([^\\]]*)\\]\\(([^)\\s]+)[^)]*\\)/g, '<img alt=\"$1\" src=\"$2\" style=\"max-width:100%;height:auto\">');\n s = s.replace(/\\[([^\\]]+)\\]\\(([^)\\s]+)[^)]*\\)/g, '<a href=\"$2\">$1</a>');\n s = s.replace(/\\*\\*([^*\\n]+)\\*\\*/g, '<strong>$1</strong>');\n s = s.replace(/`([^`\\n]+)`/g, '<code>$1</code>');\n const out = [];\n let inList = false, inPara = false;\n for (const raw of s.split('\\n')) {\n const line = raw.trim();\n if (!line) {\n if (inList) { out.push('</ul>'); inList = false; }\n if (inPara) { out.push('</p>'); inPara = false; }\n continue;\n }\n const h = line.match(/^(#{1,6})\\s+(.+)$/);\n if (h) {\n if (inList) { out.push('</ul>'); inList = false; }\n if (inPara) { out.push('</p>'); inPara = false; }\n const n = h[1].length;\n out.push(`<h${n}>${h[2]}</h${n}>`);\n continue;\n }\n const li = line.match(/^[*\\-]\\s+(.+)$/);\n if (li) {\n if (inPara) { out.push('</p>'); inPara = false; }\n if (!inList) { out.push('<ul>'); inList = true; }\n out.push(`<li>${li[1]}</li>`);\n continue;\n }\n if (inList) { out.push('</ul>'); inList = false; }\n if (!inPara) { out.push('<p>'); inPara = true; }\n out.push(line);\n }\n if (inList) out.push('</ul>');\n if (inPara) out.push('</p>');\n return out.join('\\n');\n}\n\nconst self = this;\nasync function fetchCrawl4ai(url) {\n try {\n const cr = await self.helpers.httpRequest({\n method: 'POST',\n url: 'http://mcp-crawl4ai:11235/md',\n headers: { 'Content-Type': 'application/json' },\n body: { url, filter: 'fit' },\n json: true,\n timeout: 60000,\n });\n if (cr && cr.success && cr.markdown) return cr.markdown;\n } catch (_) {}\n return '';\n}\n\nasync function fetchUnpaywall(doi) {\n try {\n const up = await self.helpers.httpRequest({\n method: 'GET',\n url: `https://api.unpaywall.org/v2/${encodeURIComponent(doi)}?email=fkrebs@nucli.de`,\n json: true,\n timeout: 15000,\n });\n if (up.is_oa && up.best_oa_location) {\n return up.best_oa_location.url || up.best_oa_location.url_for_pdf || '';\n }\n } catch (_) {}\n return '';\n}\n\nasync function getArchiveSnapshotUrl(url) {\n try {\n const tm = await self.helpers.httpRequest({\n method: 'GET',\n url: 'https://archive.ph/timemap/' + url,\n headers: { 'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 Chrome/120.0' },\n timeout: 30000,\n });\n const txt = typeof tm === 'string' ? tm : (tm && tm.body) || '';\n const matches = [...txt.matchAll(/<([^>]+)>;\\s*rel=\"[^\"]*memento[^\"]*\"/g)];\n if (matches.length > 0) return matches[matches.length - 1][1];\n } catch (_) {}\n return '';\n}\n\nconst entries = $('Get Miniflux Entries').first().json.entries || [];\nconst results = [];\nconst stats = { skipped_done: 0, skipped_nopaywall: 0, attempted: 0, success: 0, retry: 0, giveup: 0, error: 0 };\n\nfor (const e of entries) {\n if (ALREADY_PROCESSED.test(e.title)) { stats.skipped_done++; continue; }\n\n const host = hostname(e.url);\n const isScience = PAYWALL_SCIENCE.some(d => host === d || host.endsWith('.' + d));\n const isNews = PAYWALL_NEWS.some(d => host === d || host.endsWith('.' + d));\n const isShort = (e.content || '').replace(/<[^>]+>/g, '').trim().length < 800;\n\n if (!isScience && !isNews && !isShort) { stats.skipped_nopaywall++; continue; }\n stats.attempted++;\n\n const type = isScience ? 'science' : isNews ? 'news' : 'short';\n const publishedMs = new Date(e.published_at).getTime();\n const ageMs = Date.now() - publishedMs;\n const exhausted = ageMs >= RETRY_LIMIT_MS;\n\n let markdown = '';\n let source = '';\n\n if (isScience) {\n const doi = extractDoi(e.url);\n if (doi) {\n const oaUrl = await fetchUnpaywall(doi);\n if (oaUrl) {\n const md = await fetchCrawl4ai(oaUrl);\n if (md && !isPaywalled(md)) { markdown = md; source = 'unpaywall'; }\n }\n }\n }\n\n // Try crawl4ai direct first to find out if it's actually paywalled\n if (!source) {\n const md = await fetchCrawl4ai(e.url);\n if (md) {\n if (!isPaywalled(md)) {\n markdown = md;\n source = 'free';\n } else {\n markdown = md; // keep for retry/giveup display\n }\n }\n }\n\n // If paywalled (or no content), try archive.ph for bypass\n if (!source) {\n const snapUrl = await getArchiveSnapshotUrl(e.url);\n if (snapUrl) {\n const md = await fetchCrawl4ai(snapUrl);\n if (md && !isPaywalled(md)) { markdown = md; source = 'archive'; }\n }\n }\n\n const cleanTitle = e.title.replace(ANY_STATUS, '');\n let newTitle, content, status;\n\n if (source) {\n content = badge(source, type, 'ok') + mdToHtml(markdown);\n const emoji = source === 'unpaywall' ? '🔓' : source === 'free' ? '📰' : '📖';\n newTitle = cleanTitle + ' ' + emoji;\n stats.success++;\n status = 'updated';\n } else if (markdown && exhausted) {\n content = badge('', type, 'giveup', e.url) + mdToHtml(markdown);\n newTitle = cleanTitle + ' 🚫';\n stats.giveup++;\n status = 'gaveup';\n } else if (markdown) {\n content = badge('', type, 'paywalled', e.url) + mdToHtml(markdown);\n newTitle = cleanTitle + ' ⏳';\n stats.retry++;\n status = 'retry';\n } else {\n content = '<p style=\"font-size:0.85em;color:#555;border-left:3px solid #c62828;padding:6px 10px;margin:0 0 1em;background:#fafafa\">❌ Fetch error<br><a href=\"https://archive.ph/newest/' + encodeURI(e.url) + '\" target=\"_blank\" rel=\"noopener\">Check archive.ph manually →</a></p>';\n newTitle = cleanTitle + ' ❌';\n stats.error++;\n status = 'error';\n }\n\n const body = content !== null ? { content, title: newTitle } : { title: newTitle };\n try {\n await self.helpers.httpRequest({\n method: 'PUT',\n url: `${MINIFLUX_URL}/v1/entries/${e.id}`,\n headers: { 'X-Auth-Token': MINIFLUX_TOKEN, 'Content-Type': 'application/json' },\n body,\n json: true,\n timeout: 15000,\n });\n results.push({ json: { entry_id: e.id, title: newTitle.slice(0, 80), status, source, type, host } });\n } catch (err) {\n results.push({ json: { entry_id: e.id, title: e.title.slice(0, 80), status: 'put_failed', error: String(err).slice(0, 200) } });\n }\n}\n\nresults.push({ json: { stats } });\nreturn results;"
|
|
},
|
|
"typeVersion": 2
|
|
}
|
|
],
|
|
"connections": {
|
|
"Every 5 Minutes": {
|
|
"main": [
|
|
[
|
|
{
|
|
"node": "Get Miniflux Entries",
|
|
"type": "main",
|
|
"index": 0
|
|
}
|
|
]
|
|
]
|
|
},
|
|
"Get Miniflux Entries": {
|
|
"main": [
|
|
[
|
|
{
|
|
"node": "Process Entries",
|
|
"type": "main",
|
|
"index": 0
|
|
}
|
|
]
|
|
]
|
|
}
|
|
},
|
|
"settings": {
|
|
"executionOrder": "v1"
|
|
},
|
|
"staticData": {
|
|
"node:Every 5 Minutes": {
|
|
"recurrenceRules": []
|
|
}
|
|
},
|
|
"meta": null,
|
|
"pinData": null,
|
|
"versionId": "0a3f5003-653c-4494-96e9-a779815a1b9c",
|
|
"activeVersionId": "0a3f5003-653c-4494-96e9-a779815a1b9c",
|
|
"versionCounter": 17,
|
|
"triggerCount": 1,
|
|
"tags": [],
|
|
"shared": [
|
|
{
|
|
"updatedAt": "2026-05-27T22:09:54.846Z",
|
|
"createdAt": "2026-05-27T22:09:54.846Z",
|
|
"role": "workflow:owner",
|
|
"workflowId": "JqX7ejCYX1oAHvQ2",
|
|
"projectId": "hkIaTyp1CKVS8ywT",
|
|
"project": {
|
|
"updatedAt": "2026-03-09T07:21:36.510Z",
|
|
"createdAt": "2026-03-09T07:20:59.502Z",
|
|
"id": "hkIaTyp1CKVS8ywT",
|
|
"name": "Florian Krebs <fkrebs@nucli.de>",
|
|
"type": "personal",
|
|
"icon": null,
|
|
"description": null,
|
|
"creatorId": "8af5c813-f89c-4b83-bcf8-5d59864a038a"
|
|
}
|
|
}
|
|
],
|
|
"versionMetadata": {
|
|
"name": "paywall-bypass",
|
|
"description": null
|
|
}
|
|
} |