6004183d4b
Replace 6-min published_at cutoff with emoji-suffix guard so the scheduled flow processes all unread paywalled articles regardless of age. Fixes badge text: "Wallabag" -> "crawl4ai".
103 lines
7.2 KiB
JSON
103 lines
7.2 KiB
JSON
{
|
|
"name": "paywall-bypass-backfill",
|
|
"nodes": [
|
|
{
|
|
"id": "2c9974bf-181e-4c58-a1f2-5223a662379a",
|
|
"name": "Run Manually",
|
|
"type": "n8n-nodes-base.manualTrigger",
|
|
"typeVersion": 1,
|
|
"position": [
|
|
-800,
|
|
0
|
|
],
|
|
"parameters": {}
|
|
},
|
|
{
|
|
"id": "1dda2695-81a1-4fab-a7cf-095f441862eb",
|
|
"name": "Get Miniflux Entries",
|
|
"type": "n8n-nodes-base.httpRequest",
|
|
"typeVersion": 4.2,
|
|
"position": [
|
|
-360,
|
|
0
|
|
],
|
|
"parameters": {
|
|
"method": "GET",
|
|
"url": "http://192.168.1.40:17002/v1/entries",
|
|
"sendHeaders": true,
|
|
"headerParameters": {
|
|
"parameters": [
|
|
{
|
|
"name": "X-Auth-Token",
|
|
"value": "3f439f2db55967ed5acd83398214d2aa76124927cde07115d945b9282597fecb"
|
|
}
|
|
]
|
|
},
|
|
"sendQuery": true,
|
|
"queryParameters": {
|
|
"parameters": [
|
|
{
|
|
"name": "status",
|
|
"value": "unread"
|
|
},
|
|
{
|
|
"name": "limit",
|
|
"value": "100"
|
|
},
|
|
{
|
|
"name": "order",
|
|
"value": "published_at"
|
|
},
|
|
{
|
|
"name": "direction",
|
|
"value": "desc"
|
|
}
|
|
]
|
|
},
|
|
"options": {}
|
|
}
|
|
},
|
|
{
|
|
"id": "8347ba6b-556f-4c9a-b22a-4fdcab3c2fc7",
|
|
"name": "Process Entries",
|
|
"type": "n8n-nodes-base.code",
|
|
"typeVersion": 2,
|
|
"position": [
|
|
-120,
|
|
0
|
|
],
|
|
"parameters": {
|
|
"mode": "runOnceForAllItems",
|
|
"jsCode": "const MINIFLUX_TOKEN = '3f439f2db55967ed5acd83398214d2aa76124927cde07115d945b9282597fecb';\nconst MINIFLUX_URL = 'http://192.168.1.40:17002';\n\nconst PAYWALL_NEWS = [\n 'spiegel.de','zeit.de','faz.net','sueddeutsche.de','handelsblatt.com',\n 'nytimes.com','wsj.com','ft.com','washingtonpost.com','thetimes.co.uk',\n 'wired.com','theatlantic.com','economist.com','bloomberg.com',\n 'hbr.org','foreignpolicy.com','newyorker.com','taz.de',\n 'telegraph.co.uk','businessinsider.com'\n];\n\nconst PAYWALL_SCIENCE = [\n 'nature.com','science.org','cell.com','nejm.org','thelancet.com',\n 'link.springer.com','springer.com','wiley.com','onlinelibrary.wiley.com',\n 'sciencedirect.com','pubs.acs.org','rsc.org','pubs.rsc.org',\n 'bmj.com','annualreviews.org','pnas.org',\n 'jamanetwork.com','ahajournals.org','tandfonline.com','academic.oup.com'\n];\n\nconst CONTENT_MIN = 800;\nconst ALREADY_PROCESSED = /[\\u{1F513}\\u{1F4D6}\u274c]$/u;\nfunction extractDoi(url) {\n let m = url.match(/doi\\.org\\/(10\\.\\d{4,}\\/[^\\s?#]+)/);\n if (m) return decodeURIComponent(m[1]);\n m = url.match(/\\/doi\\/(10\\.\\d{4,}\\/[^\\s?#]+)/);\n if (m) return decodeURIComponent(m[1].replace(/[?#].*$/, ''));\n return null;\n}\n\nfunction hostname(url) {\n try { return new URL(url).hostname.replace(/^www\\./, ''); } catch { return ''; }\n}\n\nfunction badge(source, type) {\n if (source === 'unpaywall') return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #4CAF50;padding:3px 8px;margin:0 0 1em\">\ud83d\udcda Open access via Unpaywall</p>';\n if (type === 'science') return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #2196F3;padding:3px 8px;margin:0 0 1em\">\ud83d\udcda Full text via crawl4ai (not OA)</p>';\n return '<p style=\"font-size:0.8em;color:#555;border-left:3px solid #FF9800;padding:3px 8px;margin:0 0 1em\">\ud83d\udcd6 Paywall bypass via crawl4ai</p>';\n}\n\nconst entries = $('Get Miniflux Entries').first().json.entries || [];\n\nconst results = [];\n\nfor (const e of entries) {\n if (ALREADY_PROCESSED.test(e.title)) continue;\n const host = hostname(e.url);\n const isScience = PAYWALL_SCIENCE.some(d => host === d || host.endsWith('.' + d));\n const isNews = PAYWALL_NEWS.some(d => host === d || host.endsWith('.' + d));\n const isShort = (e.content || '').replace(/<[^>]+>/g, '').trim().length < CONTENT_MIN;\n\n if (!isScience && !isNews && !isShort) continue;\n\n const type = isScience ? 'science' : isNews ? 'news' : 'short';\n let fetchUrl = e.url;\n let source = 'wallabag';\n\n // Try Unpaywall for scientific articles with a DOI\n if (isScience) {\n const doi = extractDoi(e.url);\n if (doi) {\n try {\n const up = await $helpers.httpRequest({\n method: 'GET',\n url: `https://api.unpaywall.org/v2/${encodeURIComponent(doi)}?email=fkrebs@nucli.de`,\n });\n if (up.is_oa && up.best_oa_location) {\n const oaUrl = up.best_oa_location.url || up.best_oa_location.url_for_pdf;\n if (oaUrl) { fetchUrl = oaUrl; source = 'unpaywall'; }\n }\n } catch (_) {}\n }\n }\n\n // Fetch full text via crawl4ai\n let wbContent = '';\n try {\n const cr = await $helpers.httpRequest({\n method: 'POST',\n url: 'http://mcp-crawl4ai:11235/md',\n headers: { 'Content-Type': 'application/json' },\n body: JSON.stringify({ url: fetchUrl, filter: 'fit' }),\n });\n if (cr.success && cr.markdown) wbContent = '<p>' + cr.markdown.replace(/\\n\\n/g, '</p><p>').replace(/\\n/g, '<br>') + '</p>';\n } catch (_) {}\n\n if (!wbContent) {\n // Mark title with failure emoji\n const failTitle = e.title.replace(/ [\ud83d\udd13\ud83d\udcd6\ud83d\udd12\u23f3\u274c]$/, '') + ' \u274c';\n try {\n await $helpers.httpRequest({\n method: 'PUT',\n url: `${MINIFLUX_URL}/v1/entries/${e.id}`,\n headers: { 'X-Auth-Token': MINIFLUX_TOKEN, 'Content-Type': 'application/json' },\n body: JSON.stringify({ title: failTitle }),\n });\n } catch (_) {}\n results.push({ json: { entry_id: e.id, title: failTitle, status: 'no_content', source, type } });\n continue;\n }\n\n const finalContent = badge(source, type) + wbContent;\n const titleEmoji = source === 'unpaywall' ? '\ud83d\udd13' : '\ud83d\udcd6';\n const newTitle = e.title.replace(/ [\ud83d\udd13\ud83d\udcd6\ud83d\udd12\u23f3\u274c]$/, '') + ' ' + titleEmoji;\n\n // Update Miniflux entry content + title\n try {\n await $helpers.httpRequest({\n method: 'PUT',\n url: `${MINIFLUX_URL}/v1/entries/${e.id}`,\n headers: { 'X-Auth-Token': MINIFLUX_TOKEN, 'Content-Type': 'application/json' },\n body: JSON.stringify({ content: finalContent, title: newTitle }),\n });\n results.push({ json: { entry_id: e.id, title: newTitle, status: 'updated', source, type } });\n } catch (err) {\n results.push({ json: { entry_id: e.id, title: e.title, status: 'update_error', error: String(err), source, type } });\n }\n}\n\nif (results.length === 0) return [{ json: { status: 'no_candidates', checked: entries.length } }];\nreturn results;"
|
|
}
|
|
}
|
|
],
|
|
"connections": {
|
|
"Get Miniflux Entries": {
|
|
"main": [
|
|
[
|
|
{
|
|
"node": "Process Entries",
|
|
"type": "main",
|
|
"index": 0
|
|
}
|
|
]
|
|
]
|
|
},
|
|
"Run Manually": {
|
|
"main": [
|
|
[
|
|
{
|
|
"node": "Get Miniflux Entries",
|
|
"type": "main",
|
|
"index": 0
|
|
}
|
|
]
|
|
]
|
|
}
|
|
},
|
|
"settings": {
|
|
"executionOrder": "v1"
|
|
},
|
|
"id": "66febffd-0d3b-4853-a82b-1e41c9c30af1"
|
|
} |