Commit 0f8f60f8 authored by 谢宇轩's avatar 谢宇轩

fix: add title filter rule

parent a5e2bf93
{
"format": "news-harvester-source",
"version": 1,
"exportedAt": "2026-09-08T07:28:27.604Z",
"exportedAt": "2026-09-09T06:50:14.008Z",
"source": {
"id": "ft",
"name": "金融时报",
......@@ -10,7 +10,7 @@
"vaultFolder": "金融时报"
},
"files": {
"helpers/ft.md": "---\nsource: ft\nlistSelector: \"div.o-teaser__heading\"\nlinkSelector: 'a[href]'\nurlPattern: 'www\\.ft\\.com/content/.+'\nexcludeUrlPattern: null\nscrollSteps: 3\nscrollWaitMs: 2000\nwaitMs: 15000\n# ── 翻页(可选)── 站点把文章列表分多页时启用。两种模式互斥,最多配一个。\n# 配了翻页字段后还需在 config.json 该 source 下设 paginate: true 才生效。\npageUrlTemplate: 'https://www.ft.com/news-feed?page={page}'\npageStart: 2\nmaxPages: 5\npageWaitMs: null\nbodySelectors:\n - '[data-testid^=\"paragraph\"]'\n - '[class*=\"articleBodyContent\"] p'\n - '[class*=\"article-body\"] p'\n - 'article p'\nbodyMinParagraphs: 3\ndateSelector: 'time[datetime]'\nauthorSelector: '[rel=\"author\"], [class*=\"author\"] a'\n# Scope to the article-body container (e.g. '[class*=\"ArticleBody\"] img, article img')\n# so bottom-of-page \"recommended articles\" thumbnails aren't downloaded into assets/.\nimageSelector: 'img'\nimageMinWidth: 200\nexcludeImage:\n - logo\n - icon\n - avatar\n - profile\n# excludeImageSelector drops imgs whose ancestor matches — use for\n# \"recommended/related articles\" thumbnails sharing the hero image's CDN host.\n# excludeImageSelector: '[class*=\"RelatedArticle\"], [class*=\"RecommendedArticle\"]'\ntitleStripSuffix: ''\nmaxBodyChars: 15000\n---\n# 金融时报 抓取备忘\n\n- 主文章流容器 `div.o-teaser__heading` 实测 25 个候选链接(inspect 自动生成)。\n- urlPattern `www\\.ft\\.com/content/.+` 由样本 URL 推断,如 preview 漏抓/误抓请手编收紧。\n- 首屏可能不全,scrollSteps=3 触发懒加载;反爬严重时调高 waitMs。\n- 验证: node scripts/preview.js --url https://www.ft.com/news-feed --recipe helpers/ft.md --count 3(链接为真实文章且 charCount>500 即通过)",
"helpers/ft.fixtures.json": "{\n \"articles\": [\n \"https://www.ft.com/content/0815cd34-bf4c-4771-8bf5-710eff286763\",\n \"https://www.ft.com/content/1f71d2d0-d2e2-46da-a96f-99d20936223e\",\n \"https://www.ft.com/content/0bdd8292-efaa-4329-b41e-d81a5063c4c0\"\n ],\n \"sections\": [\n \"https://www.ft.com/world\",\n \"https://www.ft.com/technology\"\n ],\n \"minLinks\": 15\n}\n"
"helpers/ft.md": "---\nsource: ft\nlistSelector: \"div.o-teaser__heading\"\nlinkSelector: 'a[href]'\nurlPattern: 'www\\.ft\\.com/content/.+'\nexcludeUrlPattern: null\nscrollSteps: 3\nscrollWaitMs: 2000\nwaitMs: 15000\n# ── 翻页(可选)── 站点把文章列表分多页时启用。两种模式互斥,最多配一个。\n# 配了翻页字段后还需在 config.json 该 source 下设 paginate: true 才生效。\npageUrlTemplate: 'https://www.ft.com/news-feed?page={page}'\npageStart: 2\nmaxPages: 5\npageWaitMs: null\nbodySelectors:\n - '[data-testid^=\"paragraph\"]'\n - '[class*=\"articleBodyContent\"] p'\n - '[class*=\"article-body\"] p'\n - 'article p'\nbodyMinParagraphs: 3\n# 高阶付费墙(如 Lex 专栏)超出账号订阅等级时返回 \"Subscribe to read\" 占位页。\n# 实测这类页面仍带完整 NewsArticle JSON-LD,且付费墙上方还留了几段导读文字,\n# 凑够 8 段/600+ 字——og:type/正文长度这套通用启发式反而判它是文章(isArticle\n# 会为 true),必须靠标题兜底。validateArticle 保留用于拦截真正的栏目页。\nvalidateArticle: true\nbodyMinChars: 500\nrejectTitlePattern: '^Subscribe to read$'\ndateSelector: 'time[datetime]'\nauthorSelector: '[rel=\"author\"], [class*=\"author\"] a'\n# Scope to the article-body container (e.g. '[class*=\"ArticleBody\"] img, article img')\n# so bottom-of-page \"recommended articles\" thumbnails aren't downloaded into assets/.\nimageSelector: 'img'\nimageMinWidth: 200\nexcludeImage:\n - logo\n - icon\n - avatar\n - profile\n# excludeImageSelector drops imgs whose ancestor matches — use for\n# \"recommended/related articles\" thumbnails sharing the hero image's CDN host.\n# excludeImageSelector: '[class*=\"RelatedArticle\"], [class*=\"RecommendedArticle\"]'\ntitleStripSuffix: ''\nmaxBodyChars: 15000\n---\n# 金融时报 抓取备忘\n\n- 主文章流容器 `div.o-teaser__heading` 实测 25 个候选链接(inspect 自动生成)。\n- urlPattern `www\\.ft\\.com/content/.+` 由样本 URL 推断,如 preview 漏抓/误抓请手编收紧。\n- 首屏可能不全,scrollSteps=3 触发懒加载;反爬严重时调高 waitMs。\n- **高阶付费墙过滤**:FT 的 Lex 专栏等超出账号订阅等级的内容,实测页面 `document.title === \"Subscribe to read\"`(h1 为专栏名如 \"Lex\")。曾尝试只用通用的 `validateArticle`(og:type=article 或 JSON-LD Article + 正文字数/段落数达标)来判 isArticle,但这类占位页仍带完整 NewsArticle JSON-LD,且付费墙上方保留的导读文字能凑够 8 段/600+ 字,两个门槛都过了,isArticle 反而是 true——示例 https://www.ft.com/content/462fddd0-850d-46e0-b2e7-5804d483ab52 实测 `ld=NewsArticle body=602c/8p`。所以给 fetcher-helper.js 加了通用字段 `rejectTitlePattern`(正则,匹配 document.title 就强制 isArticle=false,独立于 validateArticle),recipe 里设成 `'^Subscribe to read$'`,harvest.js 据此跳过。此改动同时打在了 news-source-analyzer 自己的 scripts/fetcher-helper.js 和开发仓库 skills/news-harvester/scripts/fetcher-helper.js——只有后者同步到实际使用的 harvester 部署后,这条规则才会在真实抓取中生效。\n- 验证: node scripts/preview.js --url https://www.ft.com/news-feed --recipe helpers/ft.md --count 3(链接为真实文章且 charCount>500 即通过)\n- 回归: node scripts/recipe-test.js ft --url https://www.ft.com/news-feed(fixtures 的 sections 里放了这条 Lex 高阶付费墙 URL,断言其 isArticle=false)",
"helpers/ft.fixtures.json": "{\n \"articles\": [\n \"https://www.ft.com/content/0815cd34-bf4c-4771-8bf5-710eff286763\",\n \"https://www.ft.com/content/1f71d2d0-d2e2-46da-a96f-99d20936223e\",\n \"https://www.ft.com/content/0bdd8292-efaa-4329-b41e-d81a5063c4c0\"\n ],\n \"sections\": [\n \"https://www.ft.com/world\",\n \"https://www.ft.com/technology\",\n \"https://www.ft.com/content/462fddd0-850d-46e0-b2e7-5804d483ab52\"\n ],\n \"minLinks\": 15\n}\n"
}
}
\ No newline at end of file
......@@ -72,6 +72,15 @@ const DEFAULTS = {
// default → existing recipes keep their original behaviour.
validateArticle: false,
bodyMinChars: 500,
// Regex source; if document.title (after titleStripSuffix) matches, the
// page is rejected (isArticle=false) regardless of validateArticle/body
// length. For a metered-paywall site whose higher subscription tiers still
// emit a full NewsArticle JSON-LD + a teaser body long enough to pass the
// og:type/bodyMinChars heuristic (e.g. FT's Lex column shows "Subscribe to
// read" as the page title while still returning ~600 chars of teaser text),
// og:type/body-length alone can't tell a real article from a locked one —
// the title is the reliable signal. null = off (existing recipes unaffected).
rejectTitlePattern: null,
dateSelector: 'time[datetime]',
authorSelector: '[rel="author"], [class*="author"] a',
imageSelector: 'img',
......@@ -178,6 +187,7 @@ function buildExtractExpr(spec) {
// so non-recipe / non-validating sources keep their original behaviour.
const validateArticle = !!spec.validateArticle;
const bodyMinChars = spec.bodyMinChars || 500;
const rejectTitlePattern = spec.rejectTitlePattern ? JSON.stringify(spec.rejectTitlePattern) : 'null';
return `(function(){
var title = document.title.replace(new RegExp(${titleStripRe}, 'g'), '').trim();
// JSON-LD fallback: many sites (Nikkei, Reuters, Yonhap) embed
......@@ -267,11 +277,16 @@ function buildExtractExpr(spec) {
&& body.length >= ${bodyMinChars}
&& bodyParagraphs >= ${bodyMinParagraphs};
}
// Title-based reject: independent of validateArticle, since a metered
// paywall can still satisfy the og:type/body-length heuristic above.
var rejectTitleRe = ${rejectTitlePattern};
var titleRejected = rejectTitleRe ? new RegExp(rejectTitleRe, 'i').test(title) : false;
if (titleRejected) isArticle = false;
return JSON.stringify({
title: title, date: date, authors: authors, imgs: imgs,
body: body.substring(0, ${maxBodyChars}),
isArticle: isArticle, ogType: ogType, ldType: ldType,
bodyChars: body.length, bodyParagraphs: bodyParagraphs
bodyChars: body.length, bodyParagraphs: bodyParagraphs, titleRejected: titleRejected
});
})()`;
}
......
......@@ -65,6 +65,7 @@ All fields optional except `urlPattern`; defaults reproduce legacy behaviour.
| `bodyMinParagraphs` | `3` | Min paragraph count for a selector to qualify |
| `validateArticle` | `false` | When `true`, `fetchPage` probes `og:type` + JSON-LD `@type` + body length and returns `isArticle`; `harvest.js` skips candidates where `isArticle === false`. Use with a *broad* `urlPattern` (so candidate URLs aren't pre-filtered by fragile slug heuristics) to reject section/category pages that slip through once the page is actually fetched. Off by default → existing recipes unchanged. |
| `bodyMinChars` | `500` | Floor on body length for `isArticle` (only consulted when `validateArticle: true`) |
| `rejectTitlePattern` | `null` | Regex source; if `document.title` (after `titleStripSuffix`) matches, `isArticle` is forced `false` — **independent of `validateArticle`**, applies even when that's off. Use for a metered paywall whose locked pages still emit a real `NewsArticle` JSON-LD and a teaser body long enough to pass the `validateArticle` heuristic (e.g. FT's Lex column: title is literally `"Subscribe to read"`, but the og:type/body-length checks alone say it's a real article). `null` = off. |
| `dateSelector` | `time[datetime]` | Falls back to JSON-LD `datePublished` |
| `authorSelector` | `[rel="author"], [class*="author"] a` | Falls back to JSON-LD `author.name` |
| `imageSelector` | `img` | |
......
......@@ -72,6 +72,15 @@ const DEFAULTS = {
// default → existing recipes keep their original behaviour.
validateArticle: false,
bodyMinChars: 500,
// Regex source; if document.title (after titleStripSuffix) matches, the
// page is rejected (isArticle=false) regardless of validateArticle/body
// length. For a metered-paywall site whose higher subscription tiers still
// emit a full NewsArticle JSON-LD + a teaser body long enough to pass the
// og:type/bodyMinChars heuristic (e.g. FT's Lex column shows "Subscribe to
// read" as the page title while still returning ~600 chars of teaser text),
// og:type/body-length alone can't tell a real article from a locked one —
// the title is the reliable signal. null = off (existing recipes unaffected).
rejectTitlePattern: null,
dateSelector: 'time[datetime]',
authorSelector: '[rel="author"], [class*="author"] a',
imageSelector: 'img',
......@@ -178,6 +187,7 @@ function buildExtractExpr(spec) {
// so non-recipe / non-validating sources keep their original behaviour.
const validateArticle = !!spec.validateArticle;
const bodyMinChars = spec.bodyMinChars || 500;
const rejectTitlePattern = spec.rejectTitlePattern ? JSON.stringify(spec.rejectTitlePattern) : 'null';
return `(function(){
var title = document.title.replace(new RegExp(${titleStripRe}, 'g'), '').trim();
// JSON-LD fallback: many sites (Nikkei, Reuters, Yonhap) embed
......@@ -267,11 +277,16 @@ function buildExtractExpr(spec) {
&& body.length >= ${bodyMinChars}
&& bodyParagraphs >= ${bodyMinParagraphs};
}
// Title-based reject: independent of validateArticle, since a metered
// paywall can still satisfy the og:type/body-length heuristic above.
var rejectTitleRe = ${rejectTitlePattern};
var titleRejected = rejectTitleRe ? new RegExp(rejectTitleRe, 'i').test(title) : false;
if (titleRejected) isArticle = false;
return JSON.stringify({
title: title, date: date, authors: authors, imgs: imgs,
body: body.substring(0, ${maxBodyChars}),
isArticle: isArticle, ogType: ogType, ldType: ldType,
bodyChars: body.length, bodyParagraphs: bodyParagraphs
bodyChars: body.length, bodyParagraphs: bodyParagraphs, titleRejected: titleRejected
});
})()`;
}
......
Markdown is supported
0% or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment