From f42b9eab523d1a12623d773d480b68fb30c84fc0 Mon Sep 17 00:00:00 2001 From: fkrebs Date: Thu, 28 May 2026 00:11:07 +0200 Subject: [PATCH] paywall-bypass: add title emoji suffix per state MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit | 🔓 = open access via Unpaywall | 📖 = paywalled, fetched via Wallabag | ❌ = fetch attempted, no content returned --- flows/paywall-bypass.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/flows/paywall-bypass.json b/flows/paywall-bypass.json index 3ee95e1..7198dbb 100644 --- a/flows/paywall-bypass.json +++ b/flows/paywall-bypass.json @@ -118,7 +118,7 @@ ], "parameters": { "mode": "runOnceForAllItems", - "jsCode": "const MINIFLUX_TOKEN = '3f439f2db55967ed5acd83398214d2aa76124927cde07115d945b9282597fecb';\nconst WALLABAG_URL = 'http://192.168.1.40:17004';\nconst MINIFLUX_URL = 'http://192.168.1.40:17002';\n\nconst PAYWALL_NEWS = [\n 'spiegel.de','zeit.de','faz.net','sueddeutsche.de','handelsblatt.com',\n 'nytimes.com','wsj.com','ft.com','washingtonpost.com','thetimes.co.uk',\n 'wired.com','theatlantic.com','economist.com','bloomberg.com',\n 'hbr.org','foreignpolicy.com','newyorker.com','taz.de',\n 'telegraph.co.uk','businessinsider.com'\n];\n\nconst PAYWALL_SCIENCE = [\n 'nature.com','science.org','cell.com','nejm.org','thelancet.com',\n 'link.springer.com','springer.com','wiley.com','onlinelibrary.wiley.com',\n 'sciencedirect.com','pubs.acs.org','rsc.org','pubs.rsc.org',\n 'bmj.com','annualreviews.org','pnas.org',\n 'jamanetwork.com','ahajournals.org','tandfonline.com','academic.oup.com'\n];\n\nconst CONTENT_MIN = 800;\nconst cutoff = (Date.now() / 1000) - 360;\n\nfunction extractDoi(url) {\n let m = url.match(/doi\\.org\\/(10\\.\\d{4,}\\/[^\\s?#]+)/);\n if (m) return decodeURIComponent(m[1]);\n m = url.match(/\\/doi\\/(10\\.\\d{4,}\\/[^\\s?#]+)/);\n if (m) return decodeURIComponent(m[1].replace(/[?#].*$/, ''));\n return null;\n}\n\nfunction hostname(url) {\n try { return new URL(url).hostname.replace(/^www\\./, ''); } catch { return ''; }\n}\n\nfunction badge(source, type) {\n if (source === 'unpaywall') return '

\ud83d\udcda Open access via Unpaywall

';\n if (type === 'science') return '

\ud83d\udcda Full text via Wallabag (not OA)

';\n return '

\ud83d\udcd6 Paywall bypass via Wallabag

';\n}\n\nconst wbToken = $('Wallabag Token').first().json.access_token;\nconst entries = $('Get Miniflux Entries').first().json.entries || [];\n\nconst results = [];\n\nfor (const e of entries) {\n const pub = new Date(e.published_at).getTime() / 1000;\n if (pub < cutoff) continue;\n\n const host = hostname(e.url);\n const isScience = PAYWALL_SCIENCE.some(d => host === d || host.endsWith('.' + d));\n const isNews = PAYWALL_NEWS.some(d => host === d || host.endsWith('.' + d));\n const isShort = (e.content || '').replace(/<[^>]+>/g, '').trim().length < CONTENT_MIN;\n\n if (!isScience && !isNews && !isShort) continue;\n\n const type = isScience ? 'science' : isNews ? 'news' : 'short';\n let fetchUrl = e.url;\n let source = 'wallabag';\n\n // Try Unpaywall for scientific articles with a DOI\n if (isScience) {\n const doi = extractDoi(e.url);\n if (doi) {\n try {\n const up = await $helpers.httpRequest({\n method: 'GET',\n url: `https://api.unpaywall.org/v2/${encodeURIComponent(doi)}?email=fkrebs@nucli.de`,\n });\n if (up.is_oa && up.best_oa_location) {\n const oaUrl = up.best_oa_location.url || up.best_oa_location.url_for_pdf;\n if (oaUrl) { fetchUrl = oaUrl; source = 'unpaywall'; }\n }\n } catch (_) {}\n }\n }\n\n // Fetch full text via Wallabag\n let wbContent = '';\n try {\n const wb = await $helpers.httpRequest({\n method: 'POST',\n url: `${WALLABAG_URL}/api/entries`,\n headers: { 'Authorization': `Bearer ${wbToken}`, 'Content-Type': 'application/json' },\n body: JSON.stringify({ url: fetchUrl }),\n });\n wbContent = wb.content || '';\n } catch (_) {}\n\n if (!wbContent) {\n results.push({ json: { entry_id: e.id, title: e.title, status: 'no_content', source, type } });\n continue;\n }\n\n const finalContent = badge(source, type) + wbContent;\n\n // Update Miniflux entry content\n try {\n await $helpers.httpRequest({\n method: 'PUT',\n url: `${MINIFLUX_URL}/v1/entries/${e.id}`,\n headers: { 'X-Auth-Token': MINIFLUX_TOKEN, 'Content-Type': 'application/json' },\n body: JSON.stringify({ content: finalContent }),\n });\n results.push({ json: { entry_id: e.id, title: e.title, status: 'updated', source, type } });\n } catch (err) {\n results.push({ json: { entry_id: e.id, title: e.title, status: 'update_error', error: String(err), source, type } });\n }\n}\n\nif (results.length === 0) return [{ json: { status: 'no_candidates', checked: entries.length } }];\nreturn results;" + "jsCode": "const MINIFLUX_TOKEN = '3f439f2db55967ed5acd83398214d2aa76124927cde07115d945b9282597fecb';\nconst WALLABAG_URL = 'http://192.168.1.40:17004';\nconst MINIFLUX_URL = 'http://192.168.1.40:17002';\n\nconst PAYWALL_NEWS = [\n 'spiegel.de','zeit.de','faz.net','sueddeutsche.de','handelsblatt.com',\n 'nytimes.com','wsj.com','ft.com','washingtonpost.com','thetimes.co.uk',\n 'wired.com','theatlantic.com','economist.com','bloomberg.com',\n 'hbr.org','foreignpolicy.com','newyorker.com','taz.de',\n 'telegraph.co.uk','businessinsider.com'\n];\n\nconst PAYWALL_SCIENCE = [\n 'nature.com','science.org','cell.com','nejm.org','thelancet.com',\n 'link.springer.com','springer.com','wiley.com','onlinelibrary.wiley.com',\n 'sciencedirect.com','pubs.acs.org','rsc.org','pubs.rsc.org',\n 'bmj.com','annualreviews.org','pnas.org',\n 'jamanetwork.com','ahajournals.org','tandfonline.com','academic.oup.com'\n];\n\nconst CONTENT_MIN = 800;\nconst cutoff = (Date.now() / 1000) - 360;\n\nfunction extractDoi(url) {\n let m = url.match(/doi\\.org\\/(10\\.\\d{4,}\\/[^\\s?#]+)/);\n if (m) return decodeURIComponent(m[1]);\n m = url.match(/\\/doi\\/(10\\.\\d{4,}\\/[^\\s?#]+)/);\n if (m) return decodeURIComponent(m[1].replace(/[?#].*$/, ''));\n return null;\n}\n\nfunction hostname(url) {\n try { return new URL(url).hostname.replace(/^www\\./, ''); } catch { return ''; }\n}\n\nfunction badge(source, type) {\n if (source === 'unpaywall') return '

\ud83d\udcda Open access via Unpaywall

';\n if (type === 'science') return '

\ud83d\udcda Full text via Wallabag (not OA)

';\n return '

\ud83d\udcd6 Paywall bypass via Wallabag

';\n}\n\nconst wbToken = $('Wallabag Token').first().json.access_token;\nconst entries = $('Get Miniflux Entries').first().json.entries || [];\n\nconst results = [];\n\nfor (const e of entries) {\n const pub = new Date(e.published_at).getTime() / 1000;\n if (pub < cutoff) continue;\n\n const host = hostname(e.url);\n const isScience = PAYWALL_SCIENCE.some(d => host === d || host.endsWith('.' + d));\n const isNews = PAYWALL_NEWS.some(d => host === d || host.endsWith('.' + d));\n const isShort = (e.content || '').replace(/<[^>]+>/g, '').trim().length < CONTENT_MIN;\n\n if (!isScience && !isNews && !isShort) continue;\n\n const type = isScience ? 'science' : isNews ? 'news' : 'short';\n let fetchUrl = e.url;\n let source = 'wallabag';\n\n // Try Unpaywall for scientific articles with a DOI\n if (isScience) {\n const doi = extractDoi(e.url);\n if (doi) {\n try {\n const up = await $helpers.httpRequest({\n method: 'GET',\n url: `https://api.unpaywall.org/v2/${encodeURIComponent(doi)}?email=fkrebs@nucli.de`,\n });\n if (up.is_oa && up.best_oa_location) {\n const oaUrl = up.best_oa_location.url || up.best_oa_location.url_for_pdf;\n if (oaUrl) { fetchUrl = oaUrl; source = 'unpaywall'; }\n }\n } catch (_) {}\n }\n }\n\n // Fetch full text via Wallabag\n let wbContent = '';\n try {\n const wb = await $helpers.httpRequest({\n method: 'POST',\n url: `${WALLABAG_URL}/api/entries`,\n headers: { 'Authorization': `Bearer ${wbToken}`, 'Content-Type': 'application/json' },\n body: JSON.stringify({ url: fetchUrl }),\n });\n wbContent = wb.content || '';\n } catch (_) {}\n\n if (!wbContent) {\n // Mark title with failure emoji\n const failTitle = e.title.replace(/ [\ud83d\udd13\ud83d\udcd6\ud83d\udd12\u23f3\u274c]$/, '') + ' \u274c';\n try {\n await $helpers.httpRequest({\n method: 'PUT',\n url: `${MINIFLUX_URL}/v1/entries/${e.id}`,\n headers: { 'X-Auth-Token': MINIFLUX_TOKEN, 'Content-Type': 'application/json' },\n body: JSON.stringify({ title: failTitle }),\n });\n } catch (_) {}\n results.push({ json: { entry_id: e.id, title: failTitle, status: 'no_content', source, type } });\n continue;\n }\n\n const finalContent = badge(source, type) + wbContent;\n const titleEmoji = source === 'unpaywall' ? '\ud83d\udd13' : '\ud83d\udcd6';\n const newTitle = e.title.replace(/ [\ud83d\udd13\ud83d\udcd6\ud83d\udd12\u23f3\u274c]$/, '') + ' ' + titleEmoji;\n\n // Update Miniflux entry content + title\n try {\n await $helpers.httpRequest({\n method: 'PUT',\n url: `${MINIFLUX_URL}/v1/entries/${e.id}`,\n headers: { 'X-Auth-Token': MINIFLUX_TOKEN, 'Content-Type': 'application/json' },\n body: JSON.stringify({ content: finalContent, title: newTitle }),\n });\n results.push({ json: { entry_id: e.id, title: newTitle, status: 'updated', source, type } });\n } catch (err) {\n results.push({ json: { entry_id: e.id, title: e.title, status: 'update_error', error: String(err), source, type } });\n }\n}\n\nif (results.length === 0) return [{ json: { status: 'no_candidates', checked: entries.length } }];\nreturn results;" } } ],