From 54cb28c48816b9ad385c4346781b7c2795c5fb43 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E2=88=9A=28noham=29=C2=B2?= <100566912+NohamR@users.noreply.github.com> Date: Fri, 19 Jun 2026 21:34:01 +0200 Subject: [PATCH] Export as a lib --- README.md | 56 +++++++++++++++- package.json | 5 +- src/browser.js | 9 +-- src/index.js | 169 +++++++++++++------------------------------------ src/pdf.js | 16 +++-- src/pdfy.js | 133 ++++++++++++++++++++++++++++++++++++++ 6 files changed, 249 insertions(+), 139 deletions(-) create mode 100644 src/pdfy.js diff --git a/README.md b/README.md index df864e0..47b86a5 100644 --- a/README.md +++ b/README.md @@ -20,7 +20,7 @@ npm install The Reader View extension is downloaded and extracted automatically on first run. -## Usage +## CLI Usage ```bash node src/index.js --help @@ -90,6 +90,60 @@ npm test The test script reads URLs from `src/test/test.txt` and runs multiple scenarios (default, dark, sepia, custom CSS, etc.). +## Library Usage + +Use pdfy programmatically in your own Node.js applications: + +```js +import { convert, launchBrowser, BotProtectionError, VALID_THEMES } from 'pdfy' + +// Simple usage +const { title, pdfBuffer } = await convert('https://example.com/article') +fs.writeFileSync('article.pdf', pdfBuffer) + +// With options +const result = await convert('https://example.com/article', { + theme: 'dark', + css: 'body { color: #333 }', + prefs: { fontSize: 20 }, + output: './article.pdf' // saves to disk and returns buffer +}) +console.log(`Generated: ${result.title}`) + +// Reuse browser across conversions (for servers) +const browser = await launchBrowser() +const a = await convert('https://example.com/1', { browser, theme: 'dark' }) +const b = await convert('https://example.com/2', { browser, theme: 'sepia' }) +await browser.close() + +// Handle bot protection +try { + await convert('https://example.com/article') +} catch (err) { + if (err instanceof BotProtectionError) { + // Prompt user to provide the article HTML manually + } +} +``` + +### API + +**`convert(url, options?)`** — returns `{ title, pdfBuffer, outputPath }` + +| Option | Type | Default | Description | +|--------|------|---------|-------------| +| `theme` | string | `'light'` | Reader View theme | +| `css` | string | `null` | Custom CSS string | +| `prefs` | object | `{}` | Extension preferences | +| `output` | string | `null` | File path to write PDF (returns buffer when omitted) | +| `browser` | Browser | `null` | Reuse a Puppeteer browser instance | +| `html` | string | `null` | Pre-fetched HTML content (skips URL fetch) | + +**`launchBrowser()`** — launches a Puppeteer browser with the Reader View extension loaded. + +**`BotProtectionError`** — thrown when the target page appears to be behind bot protection. + + ## Credits - [Reader View](https://chromewebstore.google.com/detail/reader-view/ecabifbgmdmgdllomnfinbmaellmclnh) Chrome extension diff --git a/package.json b/package.json index 8b07dc2..df64cbe 100644 --- a/package.json +++ b/package.json @@ -4,7 +4,10 @@ "private": true, "type": "module", "description": "Convert web articles to PDF using Reader View and Puppeteer", - "main": "src/index.js", + "main": "src/pdfy.js", + "exports": { + ".": "./src/pdfy.js" + }, "scripts": { "start": "node src/index.js", "test": "node src/test/test.js" diff --git a/src/browser.js b/src/browser.js index 26235d6..c3960b5 100644 --- a/src/browser.js +++ b/src/browser.js @@ -14,20 +14,17 @@ export async function checkExtension () { try { await downloadExtension({ logger }) } catch (err) { - logger.error(`Failed to download extension: ${err.message}`) - process.exit(1) + throw new Error(`Failed to download extension: ${err.message}`) } if (!fs.existsSync(MANIFEST_PATH)) { - logger.error('Download completed but extension not found. Extraction may have failed.') - process.exit(1) + throw new Error('Download completed but extension not found. Extraction may have failed.') } } if (!fs.existsSync(READABILITY_PATH)) { - logger.error( + throw new Error( `Readability.js not found at: "${READABILITY_PATH}"\n` + 'The extension may be incomplete.' ) - process.exit(1) } } diff --git a/src/index.js b/src/index.js index f489cd6..dbbec46 100644 --- a/src/index.js +++ b/src/index.js @@ -2,31 +2,10 @@ import fs from 'node:fs' import process from 'node:process' import path from 'node:path' import readline from 'node:readline/promises' -import { parseArgs, VALID_THEMES } from './cli.js' +import { parseArgs } from './cli.js' import { loadConfigFile, loadPreferences, readCustomCss } from './config.js' +import { convert, BotProtectionError } from './pdfy.js' import logger from './logger.js' -import { - checkExtension, - launchBrowser, - discoverExtensionId, - getReadabilityPath -} from './browser.js' -import { - extractArticle, - storeArticle, - setExtensionPreferences, - waitForReaderView, - extractRenderedContent, - getArticleTitle -} from './reader.js' -import { - getOutputFilename, - resolveOutputPath, - enhanceHtml, - generatePdf -} from './pdf.js' - -const EXTENSION_BASE = 'chrome-extension://' async function promptForInput () { const rl = readline.createInterface({ @@ -41,120 +20,58 @@ async function promptForInput () { async function main () { const { url, opts } = parseArgs() - await checkExtension() + await loadConfigFile(opts.config) - const config = await loadConfigFile(opts.config) + const prefsPath = opts.prefs || null + const cssPath = opts.css || null + const outputPath = opts.output || null - const prefsPath = opts.prefs || config.prefs || null - const cssPath = opts.css || config.css || null - const outputPath = opts.output || config.output || null + const preferences = prefsPath ? loadPreferences(path.resolve(prefsPath)) : {} + const customCss = cssPath ? readCustomCss(path.resolve(cssPath)) : null - let preferences = {} - if (prefsPath) { - preferences = loadPreferences(path.resolve(prefsPath)) - } + let html + const currentUrl = url - const theme = opts.theme || config.theme || preferences.mode || 'light' - if (!VALID_THEMES.includes(theme)) { - logger.error( - `Invalid theme "${theme}". Valid themes: ${VALID_THEMES.join(', ')}` - ) - process.exit(1) - } - preferences.mode = theme - logger.info(`Theme: ${theme}`) + for (let attempt = 0; attempt < 2; attempt++) { + try { + const result = await convert(currentUrl, { + html, + theme: opts.theme || preferences.mode, + css: customCss, + prefs: preferences, + output: outputPath + }) - let customCss = null - if (cssPath) { - customCss = readCustomCss(path.resolve(cssPath)) - if (customCss !== null) { - preferences['user-css'] = customCss - logger.info(`Custom CSS loaded: "${path.resolve(cssPath)}"`) - } - } - - const browser = await launchBrowser() - - try { - const extId = await discoverExtensionId(browser) - - logger.info(`Fetching article: "${url}"`) - const articlePage = await browser.newPage() - await articlePage.goto(url, { waitUntil: 'networkidle0', timeout: 30000 }) - let article = await extractArticle(articlePage, getReadabilityPath()) - - if (article.title === 'Just a moment...' || article.length < 500) { - logger.warn('Cloudflare or bot protection detected..') - logger.warn('Open the URL in your browser, save the page as complete HTML (File > Save Page As),') - logger.warn('then paste the saved file path below and press Enter:') - - const htmlPath = await promptForInput() - const resolvedPath = path.resolve(htmlPath) - - if (!fs.existsSync(resolvedPath)) { - logger.error(`File not found: "${resolvedPath}"`) - process.exit(1) + const sizeKb = (result.pdfBuffer.length / 1024).toFixed(0) + logger.info(`PDF generated: "${result.title}" (${sizeKb} KB)`) + if (result.outputPath) { + logger.info(`Saved to: "${result.outputPath}"`) } + return + } catch (err) { + if (err instanceof BotProtectionError && attempt === 0) { + logger.warn('Bot protection or low-content page detected.') + logger.warn('Open the URL in your browser, save the page as complete HTML (File > Save Page As),') + logger.warn('then paste the saved file path below and press Enter:') - logger.info(`Loading article from local file: "${resolvedPath}"`) - await articlePage.goto(`file://${resolvedPath}`, { waitUntil: 'networkidle0', timeout: 30000 }) - article = await extractArticle(articlePage, getReadabilityPath()) + const htmlPath = await promptForInput() + const resolvedPath = path.resolve(htmlPath) + + if (!fs.existsSync(resolvedPath)) { + logger.error(`File not found: "${resolvedPath}"`) + process.exit(1) + } + + logger.info(`Loading article from local file: "${resolvedPath}"`) + html = fs.readFileSync(resolvedPath, 'utf-8') + continue + } + throw err } - - await articlePage.close() - - const extPage = await browser.newPage() - await storeArticle(extPage, extId, 1, article) - - const prefsToSet = Object.fromEntries( - Object.entries(preferences).filter( - ([_, v]) => v !== undefined && v !== null - ) - ) - if (Object.keys(prefsToSet).length > 0) { - await setExtensionPreferences(extPage, extId, prefsToSet) - } - await extPage.close() - - const readerUrl = [ - `${EXTENSION_BASE}${extId}/data/reader/index.html`, - '?id=1', - `&url=${encodeURIComponent(url)}` - ].join('') - logger.debug(`Opening Reader View: ${readerUrl}`) - - const readerPage = await browser.newPage() - await readerPage.goto(readerUrl, { waitUntil: 'load', timeout: 30000 }) - - await waitForReaderView(readerPage) - await new Promise((resolve) => setTimeout(resolve, 1000)) - - const title = await getArticleTitle(readerPage) - const outputFile = outputPath - ? resolveOutputPath(outputPath) - : getOutputFilename(title) - - let contentHtml = await extractRenderedContent(readerPage) - contentHtml = contentHtml.replace( - '
', - `