Commit 30f58c27 authored by 谢宇轩's avatar 谢宇轩

feat: support direct-method bundles; add wccftech-news source

- pack.js: --pattern (articleUrlPattern) for direct sources; recipe no
  longer required (and forbidden) for --method direct, since a recipe
  forces the browser path at harvest time
- source-manage.js install: carry bundle articleUrlPattern into config
- fetcher-direct.js: decode HTML entities in title/img alt
- new wccftech-news direct bundle (server-rendered WordPress, no login,
  no Chromium); verified end-to-end via harvest --preview
- docs: SKILL.md, recipe-guide.md, source-management.md updated
parent 782ddbb3
{
"format": "news-harvester-source",
"version": 1,
"exportedAt": "2026-08-24T10:11:57.469Z",
"source": {
"id": "wccftech-news",
"name": "Wccftech 新闻",
"homepageUrl": "https://wccftech.com/category/news/",
"method": "direct",
"vaultFolder": "Wccftech/新闻",
"articleUrlPattern": "wccftech\\.com/(?!category/|topic/|tag/|review/|how-to/|videos/|roundup/|tip-us/|feed/|author/|page/|ethics-statement|how-we-test)[a-z0-9-]{20,}"
},
"files": {}
}
\ No newline at end of file
...@@ -80,8 +80,8 @@ A `.nhsource.json` is plain JSON: ...@@ -80,8 +80,8 @@ A `.nhsource.json` is plain JSON:
| `format` | `"news-harvester-source"` (install validates this) | | `format` | `"news-harvester-source"` (install validates this) |
| `version` | `1` | | `version` | `1` |
| `exportedAt` | ISO timestamp | | `exportedAt` | ISO timestamp |
| `source` | `{ id, name, homepageUrl, method, vaultFolder }` — the config entry, minus `vaultPath`, `enabled`, `articleUrlPattern`, and `concurrency` (all set by install) | | `source` | `{ id, name, homepageUrl, method, vaultFolder }` — the config entry, minus `vaultPath`, `enabled`, and `concurrency` (all set by install). Direct bundles additionally carry `articleUrlPattern` (the regex `fetcher-direct.js` uses to find article links); install registers it into the config entry. |
| `files` | Map of relative path → file content. Includes `helpers/<id>.md` (the recipe) and `helpers/<id>.fixtures.json` (if present). | | `files` | Map of relative path → file content. Includes `helpers/<id>.md` (the recipe) and `helpers/<id>.fixtures.json` (if present). Direct bundles carry **no** files — a recipe would force the browser path regardless of `method`. |
### Install behavior ### Install behavior
......
...@@ -70,10 +70,21 @@ function extractAuthors(html) { ...@@ -70,10 +70,21 @@ function extractAuthors(html) {
return ''; return '';
} }
// Decode the HTML entities that survive into <title> / img alt text — numeric
// references (&#039; &#x27;) plus the common named ones. The <p> body path
// already does this inline; titles/alt skipped it, leaving literal "&#039;".
function decodeEntities(s) {
return s
.replace(/&#x([0-9a-f]+);/gi, (_, hex) => String.fromCodePoint(parseInt(hex, 16)))
.replace(/&#(\d+);/g, (_, dec) => String.fromCodePoint(parseInt(dec, 10)))
.replace(/&nbsp;/g, ' ').replace(/&amp;/g, '&').replace(/&lt;/g, '<')
.replace(/&gt;/g, '>').replace(/&quot;/g, '"').replace(/&#039;|&apos;/g, "'");
}
function extractArticle(html, url) { function extractArticle(html, url) {
// Title // Title
const titleMatch = html.match(/<title[^>]*>(.*?)<\/title>/is); const titleMatch = html.match(/<title[^>]*>(.*?)<\/title>/is);
let title = titleMatch ? titleMatch[1].trim() : ''; let title = titleMatch ? decodeEntities(titleMatch[1]).trim() : '';
title = title.replace(/ \| The Conversation| \| Reuters/g, '').trim(); title = title.replace(/ \| The Conversation| \| Reuters/g, '').trim();
// Date — capture the full ISO 8601 timestamp (incl. fractional seconds and // Date — capture the full ISO 8601 timestamp (incl. fractional seconds and
...@@ -91,7 +102,7 @@ function extractArticle(html, url) { ...@@ -91,7 +102,7 @@ function extractArticle(html, url) {
const seenSrc = new Set(); const seenSrc = new Set();
const imgs = imgMatches.map(m => { const imgs = imgMatches.map(m => {
const altMatch = m[0].match(/alt=["']([^"']*)["']/i); const altMatch = m[0].match(/alt=["']([^"']*)["']/i);
return { src: m[1], alt: altMatch ? altMatch[1] : '', width: 500 }; return { src: m[1], alt: altMatch ? decodeEntities(altMatch[1]) : '', width: 500 };
}).filter(i => !i.src.includes('logo') && !i.src.includes('icon') }).filter(i => !i.src.includes('logo') && !i.src.includes('icon')
&& !i.src.includes('facebook.com/tr') && i.src.length > 50 && !i.src.includes('facebook.com/tr') && i.src.length > 50
&& !seenSrc.has(i.src) && seenSrc.add(i.src)).slice(0, 6); && !seenSrc.has(i.src) && seenSrc.add(i.src)).slice(0, 6);
......
...@@ -171,10 +171,12 @@ switch (command) { ...@@ -171,10 +171,12 @@ switch (command) {
written.push(`helpers/${path.basename(dest)}`); written.push(`helpers/${path.basename(dest)}`);
} }
// Upsert the config entry. recipe sources don't carry articleUrlPattern // Upsert the config entry. Browser (recipe) sources don't carry
// (it's legacy-direct-only and ignored when a recipe exists). `paginate` // articleUrlPattern — it's ignored when helpers/<id>.md exists. Direct
// defaults to false on install but is intentionally NOT overwritten on // bundles from pack.js carry it (fetcher-direct needs it to find article
// re-install of an existing source (preserves the operator's setting). // links); carry it through when present. `paginate` defaults to false on
// install but is intentionally NOT overwritten on re-install of an
// existing source (preserves the operator's setting).
const entry = { const entry = {
id, id,
name: src.name || id, name: src.name || id,
...@@ -183,8 +185,9 @@ switch (command) { ...@@ -183,8 +185,9 @@ switch (command) {
vaultFolder: src.vaultFolder || src.name || id, vaultFolder: src.vaultFolder || src.name || id,
enabled: true, enabled: true,
concurrency: 1, // serial by default; user can raise to enable multi-tab fetching concurrency: 1, // serial by default; user can raise to enable multi-tab fetching
paginate: false // pagination switch — set true in config.json to enable multi-page harvesting paginate: false // pagination switch — set true in config.json to enable multi-page harvesting
}; };
if (src.articleUrlPattern) entry.articleUrlPattern = src.articleUrlPattern;
if (existing) { if (existing) {
// Preserve user-set paginate and concurrency across re-installs; both // Preserve user-set paginate and concurrency across re-installs; both
// stay at their defaults for new sources. // stay at their defaults for new sources.
......
...@@ -17,7 +17,8 @@ Analyze news websites, author fetch recipes, and produce installable `.nhsource. ...@@ -17,7 +17,8 @@ Analyze news websites, author fetch recipes, and produce installable `.nhsource.
| Inspect + write recipe | `node scripts/inspect-source.js <url> --scroll 3 --write <id> [--name "名称"]` | | Inspect + write recipe | `node scripts/inspect-source.js <url> --scroll 3 --write <id> [--name "名称"]` |
| Preview articles (read-only) | `node scripts/preview.js --url <url> --recipe <path> [--count 3] [--full]` | | Preview articles (read-only) | `node scripts/preview.js --url <url> --recipe <path> [--count 3] [--full]` |
| Run recipe regression tests | `node scripts/recipe-test.js <id> --url <homepageUrl>` | | Run recipe regression tests | `node scripts/recipe-test.js <id> --url <homepageUrl>` |
| Pack a bundle | `node scripts/pack.js --id <id> --name <name> --url <url> --method <browser\|direct> --folder <folder> --recipe <path> [--fixtures <path>] [--out <path>]` | | Pack a bundle | `node scripts/pack.js --id <id> --name <name> --url <url> --method browser --folder <folder> --recipe <path> [--fixtures <path>] [--out <path>]` |
| Pack a direct bundle | `node scripts/pack.js --id <id> --name <name> --url <url> --method direct --folder <folder> --pattern <regex> [--out <path>]` |
| Ensure Chromium | `node scripts/ensure-chromium.js` (or `--check`, `--list`, `--login`) | | Ensure Chromium | `node scripts/ensure-chromium.js` (or `--check`, `--list`, `--login`) |
> **All commands must be prefixed with `cd "<skill-root>" &&`**. Scripts live in `scripts/` — always reference them as `node scripts/inspect-source.js`, never `node inspect-source.js`. > **All commands must be prefixed with `cd "<skill-root>" &&`**. Scripts live in `scripts/` — always reference them as `node scripts/inspect-source.js`, never `node inspect-source.js`.
...@@ -128,6 +129,7 @@ Asserts: every article URL fetches `isArticle=true` with `bodyChars ≥ bodyMinC ...@@ -128,6 +129,7 @@ Asserts: every article URL fetches `isArticle=true` with `bodyChars ≥ bodyMinC
### Step 4. Pack into a bundle ### Step 4. Pack into a bundle
Browser source (recipe-driven):
```bash ```bash
node scripts/pack.js --id <id> --name <name> --url <homepageUrl> \ node scripts/pack.js --id <id> --name <name> --url <homepageUrl> \
--method browser --folder <vaultFolder> \ --method browser --folder <vaultFolder> \
...@@ -135,6 +137,15 @@ node scripts/pack.js --id <id> --name <name> --url <homepageUrl> \ ...@@ -135,6 +137,15 @@ node scripts/pack.js --id <id> --name <name> --url <homepageUrl> \
[--out bundles/<id>.nhsource.json] [--out bundles/<id>.nhsource.json]
``` ```
Direct source (server-rendered open site — no Chromium, no recipe):
```bash
node scripts/pack.js --id <id> --name <name> --url <listPageUrl> \
--method direct --folder <vaultFolder> --pattern '<articleUrlPattern regex>' \
[--out bundles/<id>.nhsource.json]
```
**When to choose direct**: the list page and article pages render server-side (verify with `curl -sL -A "Mozilla/5.0 ..." <url>` — real article links and `<p>` body present in the raw HTML, no JS challenge), and no login is needed. The harvester then uses `fetcher-direct.js` (plain HTTPS + regex extraction): links come from `--pattern` (matched as a substring of each `href`; exclude section/static pages via a `(?!category/|tag/|...)` lookahead + slug-length floor), body/metadata from JSON-LD / `<p>` fallbacks. Verify before packing by running the harvester's fetcher on a sample article: `cd <harvester-root> && node scripts/fetcher-direct.js <articleUrl>` (expect a full title, ISO date, authors, `body` > 500 chars). Direct bundles carry **no** `helpers/<id>.md` — a recipe would force the browser path at harvest time regardless of `method`.
Produces a self-contained `.nhsource.json` bundle. The harvester installs it with: Produces a self-contained `.nhsource.json` bundle. The harvester installs it with:
```bash ```bash
node scripts/source-manage.js install <bundle.nhsource.json> node scripts/source-manage.js install <bundle.nhsource.json>
......
...@@ -4,7 +4,7 @@ ...@@ -4,7 +4,7 @@
A recipe (`helpers/<id>.md`) is a YAML frontmatter file that declares the selectors, scroll behavior, and URL filters for a news source. When the harvester's `harvest.js` finds a recipe, it uses `fetcher-helper.js` (which reads the recipe's spec) to drive Chromium via CDP. Browser-type sources without a recipe are no longer supported (`fetcher-browser.js` has been removed); direct-type sources fall back to `fetcher-direct.js`. A recipe (`helpers/<id>.md`) is a YAML frontmatter file that declares the selectors, scroll behavior, and URL filters for a news source. When the harvester's `harvest.js` finds a recipe, it uses `fetcher-helper.js` (which reads the recipe's spec) to drive Chromium via CDP. Browser-type sources without a recipe are no longer supported (`fetcher-browser.js` has been removed); direct-type sources fall back to `fetcher-direct.js`.
> **Precedence**: when a recipe exists, its `urlPattern`/selectors are the source of truth and the config's `articleUrlPattern` is **ignored**. A recipe also forces Chromium use regardless of `method` (since `fetcher-helper.js` drives the browser). > **Precedence**: when a recipe exists, its `urlPattern`/selectors are the source of truth and the config's `articleUrlPattern` is **ignored**. A recipe also forces Chromium use regardless of `method` (since `fetcher-helper.js` drives the browser). For that reason **direct bundles never carry a recipe** — they ship `articleUrlPattern` instead (packed with `pack.js --method direct --pattern <regex>`), and the harvester serves them with the recipe-free `fetcher-direct.js` (plain HTTPS + regex extraction, no Chromium).
## Authoring workflow ## Authoring workflow
......
...@@ -8,16 +8,25 @@ ...@@ -8,16 +8,25 @@
* harvester's `source-manage.js install` can consume in one command. * harvester's `source-manage.js install` can consume in one command.
* *
* Usage: * Usage:
* Browser source (recipe-driven, fetcher-helper via CDP):
* node scripts/pack.js --id <id> --name <name> --url <homepageUrl> \ * node scripts/pack.js --id <id> --name <name> --url <homepageUrl> \
* --method <browser|direct> --folder <vaultFolder> \ * --method browser --folder <vaultFolder> \
* --recipe <path> [--fixtures <path>] [--out <path.nhsource.json>] * --recipe <path> [--fixtures <path>] [--out <path.nhsource.json>]
* *
* Direct source (plain HTTP, fetcher-direct regex extraction — no recipe:
* harvest.js forces the browser path whenever helpers/<id>.md exists, so a
* direct bundle must not carry one):
* node scripts/pack.js --id <id> --name <name> --url <homepageUrl> \
* --method direct --folder <vaultFolder> --pattern <articleUrlPattern> \
* [--out <path.nhsource.json>]
*
* The bundle format is identical to the old export output: * The bundle format is identical to the old export output:
* { * {
* "format": "news-harvester-source", * "format": "news-harvester-source",
* "version": 1, * "version": 1,
* "exportedAt": "<ISO timestamp>", * "exportedAt": "<ISO timestamp>",
* "source": { "id", "name", "homepageUrl", "method", "vaultFolder" }, * "source": { "id", "name", "homepageUrl", "method", "vaultFolder"
* [, "articleUrlPattern" — direct sources only] },
* "files": { "helpers/<id>.md": "<recipe content>", "helpers/<id>.fixtures.json": "..." } * "files": { "helpers/<id>.md": "<recipe content>", "helpers/<id>.fixtures.json": "..." }
* } * }
* *
...@@ -37,6 +46,7 @@ function parseArgs(argv) { ...@@ -37,6 +46,7 @@ function parseArgs(argv) {
else if (a === '--method') out.method = argv[++i]; else if (a === '--method') out.method = argv[++i];
else if (a === '--folder') out.folder = argv[++i]; else if (a === '--folder') out.folder = argv[++i];
else if (a === '--recipe') out.recipe = argv[++i]; else if (a === '--recipe') out.recipe = argv[++i];
else if (a === '--pattern') out.pattern = argv[++i];
else if (a === '--fixtures') out.fixtures = argv[++i]; else if (a === '--fixtures') out.fixtures = argv[++i];
else if (a === '--out') out.out = argv[++i]; else if (a === '--out') out.out = argv[++i];
else if (a === '-h' || a === '--help') out.help = true; else if (a === '-h' || a === '--help') out.help = true;
...@@ -48,17 +58,28 @@ function printHelp() { ...@@ -48,17 +58,28 @@ function printHelp() {
console.log(`pack.js — Pack a recipe + source metadata into a .nhsource.json bundle. console.log(`pack.js — Pack a recipe + source metadata into a .nhsource.json bundle.
Usage: Usage:
Browser source (recipe-driven, fetcher-helper via CDP):
node scripts/pack.js --id <id> --name <name> --url <homepageUrl> \\ node scripts/pack.js --id <id> --name <name> --url <homepageUrl> \\
--method <browser|direct> --folder <vaultFolder> \\ --method browser --folder <vaultFolder> \\
--recipe <path> [--fixtures <path>] [--out <path.nhsource.json>] --recipe <path> [--fixtures <path>] [--out <path.nhsource.json>]
Direct source (plain HTTP, fetcher-direct regex extraction, no Chromium):
node scripts/pack.js --id <id> --name <name> --url <homepageUrl> \\
--method direct --folder <vaultFolder> --pattern <regex> \\
[--out <path.nhsource.json>]
Required: Required:
--id <id> Source identifier (lowercase, no spaces, e.g. "reuters-world") --id <id> Source identifier (lowercase, no spaces, e.g. "reuters-world")
--name <name> Display name (e.g. "路透社世界") --name <name> Display name (e.g. "路透社世界")
--url <url> Homepage URL (e.g. "https://www.reuters.com/world/") --url <url> List-page URL (e.g. "https://www.reuters.com/world/")
--method <method> Fetch method: "browser" or "direct" --method <method> Fetch method: "browser" or "direct"
--folder <folder> Vault subfolder (e.g. "路透社/世界") --folder <folder> Vault subfolder (e.g. "路透社/世界")
--recipe <path> Path to the recipe file (helpers/<id>.md) --recipe <path> Recipe file (helpers/<id>.md) — required for --method browser,
forbidden for --method direct (a recipe would force the
browser path at harvest time)
--pattern <regex> articleUrlPattern regex source, matched as a substring of
each href — required for --method direct (fetcher-direct
finds article links with it)
Optional: Optional:
--fixtures <path> Path to fixtures file (helpers/<id>.fixtures.json) --fixtures <path> Path to fixtures file (helpers/<id>.fixtures.json)
...@@ -72,20 +93,49 @@ Examples: ...@@ -72,20 +93,49 @@ Examples:
node scripts/pack.js --id reuters-world --name "路透社世界" \\ node scripts/pack.js --id reuters-world --name "路透社世界" \\
--url https://www.reuters.com/world/ --method browser --folder "路透社/世界" \\ --url https://www.reuters.com/world/ --method browser --folder "路透社/世界" \\
--recipe helpers/reuters-world.md --fixtures helpers/reuters-world.fixtures.json \\ --recipe helpers/reuters-world.md --fixtures helpers/reuters-world.fixtures.json \\
--out bundles/reuters-world.nhsource.json`); --out bundles/reuters-world.nhsource.json
node scripts/pack.js --id wccftech-news --name "Wccftech 新闻" \\
--url https://wccftech.com/category/news/ --method direct --folder "Wccftech/新闻" \\
--pattern 'wccftech\\.com/(?!category/|topic/|tag/)[a-z0-9-]{20,}' \\
--out bundles/wccftech-news.nhsource.json`);
} }
async function main() { async function main() {
const args = parseArgs(process.argv.slice(2)); const args = parseArgs(process.argv.slice(2));
if (args.help) { printHelp(); process.exit(0); } if (args.help) { printHelp(); process.exit(0); }
// Validate required args // Validate required args. Browser sources carry a recipe; direct sources
const required = ['id', 'name', 'url', 'method', 'folder', 'recipe']; // carry an articleUrlPattern instead (fetcher-direct extracts links with it,
for (const field of required) { // and a recipe would force the browser path at harvest time regardless of
if (!args[field]) { // method — see harvest.js's helperPath dispatch).
console.error(`✗ --${field} is required`); if (args.method === 'direct') {
const required = ['id', 'name', 'url', 'method', 'folder', 'pattern'];
for (const field of required) {
if (!args[field]) {
console.error(`✗ --${field} is required for --method direct`);
process.exit(1);
}
}
if (args.recipe) {
console.error('✗ --recipe is not allowed with --method direct: harvest.js routes any source');
console.error(' with a helpers/<id>.md recipe to fetcher-helper (Chromium), defeating direct.');
console.error(' Direct sources describe link discovery via --pattern alone.');
process.exit(1);
}
try { new RegExp(args.pattern); }
catch (e) {
console.error(`✗ --pattern is not a valid regex: ${e.message}`);
process.exit(1); process.exit(1);
} }
} else {
const required = ['id', 'name', 'url', 'method', 'folder', 'recipe'];
for (const field of required) {
if (!args[field]) {
console.error(`✗ --${field} is required`);
process.exit(1);
}
}
} }
if (!['browser', 'direct'].includes(args.method)) { if (!['browser', 'direct'].includes(args.method)) {
...@@ -93,42 +143,47 @@ async function main() { ...@@ -93,42 +143,47 @@ async function main() {
process.exit(1); process.exit(1);
} }
// Read recipe file // Build files map (empty for direct sources — no recipe, no fixtures)
const recipePath = path.resolve(args.recipe);
if (!fs.existsSync(recipePath)) {
console.error(`✗ Recipe not found: ${recipePath}`);
process.exit(1);
}
const recipeContent = fs.readFileSync(recipePath, 'utf8');
// Build files map
const files = {}; const files = {};
files[`helpers/${args.id}.md`] = recipeContent; const included = [];
const included = [`helpers/${args.id}.md`];
if (args.recipe) {
// Optionally read fixtures const recipePath = path.resolve(args.recipe);
if (args.fixtures) { if (!fs.existsSync(recipePath)) {
const fixturesPath = path.resolve(args.fixtures); console.error(`✗ Recipe not found: ${recipePath}`);
if (!fs.existsSync(fixturesPath)) {
console.error(`✗ Fixtures not found: ${fixturesPath}`);
process.exit(1); process.exit(1);
} }
files[`helpers/${args.id}.fixtures.json`] = fs.readFileSync(fixturesPath, 'utf8'); files[`helpers/${args.id}.md`] = fs.readFileSync(recipePath, 'utf8');
included.push(`helpers/${args.id}.fixtures.json`); included.push(`helpers/${args.id}.md`);
// Optionally read fixtures
if (args.fixtures) {
const fixturesPath = path.resolve(args.fixtures);
if (!fs.existsSync(fixturesPath)) {
console.error(`✗ Fixtures not found: ${fixturesPath}`);
process.exit(1);
}
files[`helpers/${args.id}.fixtures.json`] = fs.readFileSync(fixturesPath, 'utf8');
included.push(`helpers/${args.id}.fixtures.json`);
}
} }
// Build bundle (same format as the old source-manage.js export) // Build bundle (same format as the old source-manage.js export). Direct
// sources carry articleUrlPattern so install can register it — it is the
// fetcher-direct equivalent of a recipe's urlPattern.
const source = {
id: args.id,
name: args.name,
homepageUrl: args.url,
method: args.method,
vaultFolder: args.folder
};
if (args.method === 'direct') source.articleUrlPattern = args.pattern;
const bundle = { const bundle = {
format: 'news-harvester-source', format: 'news-harvester-source',
version: 1, version: 1,
exportedAt: new Date().toISOString(), exportedAt: new Date().toISOString(),
source: { source,
id: args.id,
name: args.name,
homepageUrl: args.url,
method: args.method,
vaultFolder: args.folder
},
files files
}; };
...@@ -138,7 +193,10 @@ async function main() { ...@@ -138,7 +193,10 @@ async function main() {
console.log(`\n✅ Packed source "${args.name}" (id: ${args.id})`); console.log(`\n✅ Packed source "${args.name}" (id: ${args.id})`);
console.log(` Bundle: ${outPath}`); console.log(` Bundle: ${outPath}`);
console.log(` Files: ${included.join(', ')}`); console.log(` Files: ${included.length ? included.join(', ') : '(none — direct source, pattern-only)'}`);
if (args.method === 'direct') {
console.log(` Pattern: ${args.pattern}`);
}
console.log(`\n Install on the harvester with:`); console.log(`\n Install on the harvester with:`);
console.log(` node scripts/source-manage.js install ${path.basename(outPath)}\n`); console.log(` node scripts/source-manage.js install ${path.basename(outPath)}\n`);
} }
......
Markdown is supported
0% or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment