diff --git a/flows/paywall-bypass.json b/flows/paywall-bypass.json index 6c1989e..ebcf992 100644 --- a/flows/paywall-bypass.json +++ b/flows/paywall-bypass.json @@ -5,7 +5,6 @@ "id": "86fa097b-c230-484d-806d-45b421635c05", "name": "Every 5 Minutes", "type": "n8n-nodes-base.scheduleTrigger", - "typeVersion": 1.2, "position": [ -800, 0 @@ -19,30 +18,23 @@ } ] } - } + }, + "typeVersion": 1.2 }, { "id": "eb644570-cca3-4464-9b99-02084f199078", "name": "Get Miniflux Entries", "type": "n8n-nodes-base.httpRequest", - "typeVersion": 4.2, "position": [ -360, 0 ], "parameters": { - "method": "GET", "url": "http://192.168.1.40:17002/v1/entries", - "sendHeaders": true, - "headerParameters": { - "parameters": [ - { - "name": "X-Auth-Token", - "value": "3f439f2db55967ed5acd83398214d2aa76124927cde07115d945b9282597fecb" - } - ] - }, + "method": "GET", + "options": {}, "sendQuery": true, + "sendHeaders": true, "queryParameters": { "parameters": [ { @@ -63,22 +55,30 @@ } ] }, - "options": {} - } + "headerParameters": { + "parameters": [ + { + "name": "X-Auth-Token", + "value": "3f439f2db55967ed5acd83398214d2aa76124927cde07115d945b9282597fecb" + } + ] + } + }, + "typeVersion": 4.2 }, { "id": "f8024262-3163-4cd9-8233-3c91bfddcb01", "name": "Process Entries", "type": "n8n-nodes-base.code", - "typeVersion": 2, "position": [ -120, 0 ], "parameters": { "mode": "runOnceForAllItems", - "jsCode": "const MINIFLUX_TOKEN = '3f439f2db55967ed5acd83398214d2aa76124927cde07115d945b9282597fecb';\nconst MINIFLUX_URL = 'http://192.168.1.40:17002';\n\nconst PAYWALL_NEWS = [\n 'spiegel.de','zeit.de','faz.net','sueddeutsche.de','handelsblatt.com',\n 'nytimes.com','wsj.com','ft.com','washingtonpost.com','thetimes.co.uk',\n 'wired.com','theatlantic.com','economist.com','bloomberg.com',\n 'hbr.org','foreignpolicy.com','newyorker.com','taz.de',\n 'telegraph.co.uk','businessinsider.com'\n];\n\nconst PAYWALL_SCIENCE = [\n 'nature.com','science.org','cell.com','nejm.org','thelancet.com',\n 'link.springer.com','springer.com','wiley.com','onlinelibrary.wiley.com',\n 'sciencedirect.com','pubs.acs.org','rsc.org','pubs.rsc.org',\n 'bmj.com','annualreviews.org','pnas.org',\n 'jamanetwork.com','ahajournals.org','tandfonline.com','academic.oup.com'\n];\n\nconst CONTENT_MIN = 800;\nconst ALREADY_PROCESSED = /[\\u{1F513}\\u{1F4D6}\u274c]$/u;\n\nfunction extractDoi(url) {\n let m = url.match(/doi\\.org\\/(10\\.\\d{4,}\\/[^\\s?#]+)/);\n if (m) return decodeURIComponent(m[1]);\n m = url.match(/\\/doi\\/(10\\.\\d{4,}\\/[^\\s?#]+)/);\n if (m) return decodeURIComponent(m[1].replace(/[?#].*$/, ''));\n return null;\n}\n\nfunction hostname(url) {\n try { return new URL(url).hostname.replace(/^www\\./, ''); } catch { return ''; }\n}\n\nfunction badge(source, type) {\n if (source === 'unpaywall') return '
\ud83d\udd13 Open access via Unpaywall
';\n if (type === 'science') return '\ud83d\udcd6 Full text via crawl4ai (not OA)
';\n return '\ud83d\udcd6 Paywall bypass via crawl4ai
';\n}\n\nconst entries = $('Get Miniflux Entries').first().json.entries || [];\n\nconst results = [];\n\nfor (const e of entries) {\n // Skip already-processed entries (title ends with status emoji)\n if (ALREADY_PROCESSED.test(e.title)) continue;\n\n const host = hostname(e.url);\n const isScience = PAYWALL_SCIENCE.some(d => host === d || host.endsWith('.' + d));\n const isNews = PAYWALL_NEWS.some(d => host === d || host.endsWith('.' + d));\n const isShort = (e.content || '').replace(/<[^>]+>/g, '').trim().length < CONTENT_MIN;\n\n if (!isScience && !isNews && !isShort) continue;\n\n const type = isScience ? 'science' : isNews ? 'news' : 'short';\n let fetchUrl = e.url;\n let source = 'crawl4ai';\n\n // Try Unpaywall for scientific articles with a DOI\n if (isScience) {\n const doi = extractDoi(e.url);\n if (doi) {\n try {\n const up = await $helpers.httpRequest({\n method: 'GET',\n url: `https://api.unpaywall.org/v2/${encodeURIComponent(doi)}?email=fkrebs@nucli.de`,\n });\n if (up.is_oa && up.best_oa_location) {\n const oaUrl = up.best_oa_location.url || up.best_oa_location.url_for_pdf;\n if (oaUrl) { fetchUrl = oaUrl; source = 'unpaywall'; }\n }\n } catch (_) {}\n }\n }\n\n // Fetch full text via crawl4ai\n let fullContent = '';\n try {\n const cr = await $helpers.httpRequest({\n method: 'POST',\n url: 'http://mcp-crawl4ai:11235/md',\n headers: { 'Content-Type': 'application/json' },\n body: JSON.stringify({ url: fetchUrl, filter: 'fit' }),\n });\n if (cr.success && cr.markdown) fullContent = '' + cr.markdown.replace(/\\n\\n/g, '
').replace(/\\n/g, '
') + '
🔓 Open access via Unpaywall
';\n if (type === 'science') return '📖 Full text via crawl4ai (not OA)
';\n return '📖 Paywall bypass via crawl4ai
';\n}\n\nconst entries = $('Get Miniflux Entries').first().json.entries || [];\n\nconst results = [];\n\nfor (const e of entries) {\n if (ALREADY_PROCESSED.test(e.title)) continue;\n\n const host = hostname(e.url);\n const isScience = PAYWALL_SCIENCE.some(d => host === d || host.endsWith('.' + d));\n const isNews = PAYWALL_NEWS.some(d => host === d || host.endsWith('.' + d));\n const isShort = (e.content || '').replace(/<[^>]+>/g, '').trim().length < CONTENT_MIN;\n\n if (!isScience && !isNews && !isShort) continue;\n\n const type = isScience ? 'science' : isNews ? 'news' : 'short';\n let fetchUrl = e.url;\n let source = 'crawl4ai';\n\n if (isScience) {\n const doi = extractDoi(e.url);\n if (doi) {\n try {\n const upResp = await fetch(\n `https://api.unpaywall.org/v2/${encodeURIComponent(doi)}?email=fkrebs@nucli.de`\n );\n if (upResp.ok) {\n const up = await upResp.json();\n if (up.is_oa && up.best_oa_location) {\n const oaUrl = up.best_oa_location.url || up.best_oa_location.url_for_pdf;\n if (oaUrl) { fetchUrl = oaUrl; source = 'unpaywall'; }\n }\n }\n } catch (_) {}\n }\n }\n\n let fullContent = '';\n try {\n const crResp = await fetch('http://mcp-crawl4ai:11235/md', {\n method: 'POST',\n headers: { 'Content-Type': 'application/json' },\n body: JSON.stringify({ url: fetchUrl, filter: 'fit' }),\n });\n if (crResp.ok) {\n const cr = await crResp.json();\n if (cr.success && cr.markdown) {\n fullContent = '' + cr.markdown.replace(/\\n\\n/g, '
').replace(/\\n/g, '
') + '