fix(news): count lead cards in degradation check, add draft dry-run script

Trying to push a WeChat draft locally surfaced two problems.

isNewsMorningContentDegraded matched cards with `class="card(?:\s|")`, which
never matches news002's `class="lead-card"`. Headline cards were therefore
invisible to the check, so a healthy news002 page could be flagged as degraded
and regenerated. Count cards through the shared extractor instead, which
handles card, lead-card and highlight-box alike.

Add scripts/dryrun-news-morning-draft.mjs to run a page through the dedup gate,
degradation check, inline conversion and footer layout without calling the
WeChat API, since local IPs are not in the official account whitelist
(errcode 40164) and a real push would write into the production draft box.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
john
2026-09-21 09:31:54 +08:00
parent 5708b333cc
commit 2049bc8c6e
5 changed files with 121 additions and 4 deletions
+85
View File
@@ -0,0 +1,85 @@
#!/usr/bin/env node
/**
* 本地 dry-run:把一张早报 HTML 走完「去重闸门 → 微信 inline 转换 → 统一页脚」全链路,
* 输出最终草稿正文,但不调用任何微信接口、不落 DB。
*
* 用于在本机 IP 不在公众号白名单时验证 news001 / news002 的草稿产出。
*
* 用法:
* node scripts/dryrun-news-morning-draft.mjs <页面.html> [输出目录]
*/
import fs from 'node:fs';
import path from 'node:path';
import {
dedupeNewsMorningHtml,
formatNewsMorningDedupeSummary,
} from '../wechat-news-morning-dedup.mjs';
import {
convertNewsPageHtmlToWechatArticle,
isDailyNewsFormat,
isNewsMorningContentDegraded,
} from '../wechat-news-morning-draft.mjs';
const MAX_WECHAT_CONTENT_CHARS = 20000;
const input = process.argv[2];
const outDir = process.argv[3] || path.dirname(input ?? '.');
if (!input || !fs.existsSync(input)) {
console.error('用法:node scripts/dryrun-news-morning-draft.mjs <页面.html> [输出目录]');
process.exit(1);
}
const publicUrl = 'https://m.tkmind.cn/MindSpace/dry-run/public/daily-news-dryrun.html';
const raw = fs.readFileSync(input, 'utf8');
console.log(`输入:${input}(${raw.length} 字符)`);
// 1. 去重闸门
const deduped = dedupeNewsMorningHtml(raw);
console.log(`\n[1] 去重闸门:${formatNewsMorningDedupeSummary(deduped)}`);
if (deduped.removedSectionIds.length) {
console.log(` 移除空栏目:${deduped.removedSectionIds.join(', ')}`);
}
// 2. 降级检测
const degraded = isNewsMorningContentDegraded(deduped.html);
console.log(`\n[2] 降级检测:${degraded ? '❌ 判定为降级页(卡片不足/检索失败)' : '✅ 正常'}`);
// 3. 识别版式 + 转换为微信草稿
console.log(`\n[3] 版式识别:${isDailyNewsFormat(deduped.html) ? '✅ daily-news' : '❌ 非早报版式'}`);
const article = convertNewsPageHtmlToWechatArticle(
deduped.html,
{ publicUrl, qrcodeImageUrl: 'https://mmbiz.qpic.cn/dryrun-qrcode.png' },
);
console.log('\n[4] 草稿产出:');
console.log(` 标题:${article.title}`);
console.log(` 摘要:${String(article.digest ?? '').slice(0, 60)}`);
console.log(` 正文长度:${article.content.length} / ${MAX_WECHAT_CONTENT_CHARS}`
+ (article.content.length > MAX_WECHAT_CONTENT_CHARS ? ' ❌ 超限' : ' ✅ 未超限'));
const checks = [
// 微信不支持 <style>,早报路径必须全部内联。
['无 <style> 块(已内联)', !/<style/i.test(article.content)],
['正文使用内联 style', (article.content.match(/style="/g) ?? []).length > 20],
['含「阅读原文」引导', /阅读原文/u.test(article.content)],
['含关注二维码区', /<img[^>]+dryrun-qrcode/i.test(article.content)],
['无 <script> 残留', !/<script/i.test(article.content)],
['无 MindSpace 平台水印', !/data-mindspace-page-tag/i.test(article.content)],
];
console.log('\n[5] 合规检查:');
for (const [label, ok] of checks) console.log(` ${ok ? '✅' : '❌'} ${label}`);
const base = path.basename(input, '.html');
const outHtml = path.join(outDir, `${base}.wechat-draft.html`);
fs.writeFileSync(outHtml, `<!DOCTYPE html><html><head><meta charset="utf-8">
<meta name="viewport" content="width=device-width,initial-scale=1">
<title>${article.title}(微信草稿 dry-run)</title></head>${article.content}</html>`);
console.log(`\n草稿正文已写入:${outHtml}`);
const outDeduped = path.join(outDir, `${base}.deduped.html`);
if (deduped.changed) {
fs.writeFileSync(outDeduped, deduped.html);
console.log(`去重后页面已写入:${outDeduped}`);
}
+6 -1
View File
@@ -17,7 +17,7 @@ import {
// 必须按 class token 精确匹配:`\bcard\b` 会误命中 weather-card / knowledge-card 等。
const CARD_OPEN_RE = /<(div|a)\b[^>]*\bclass="([^"]*)"[^>]*>/gi;
const SECTION_OPEN_RE = /<div\b[^>]*class="[^"]*\bsection\b[^"]*"[^>]*id="([^"]+)"[^>]*>/gi;
const CARD_CLASS_TOKENS = new Set(['card', 'lead-card']);
const CARD_CLASS_TOKENS = new Set(['card', 'lead-card', 'highlight-box']);
function hasCardClass(classAttr) {
return String(classAttr).split(/\s+/).some((token) => CARD_CLASS_TOKENS.has(token));
@@ -138,6 +138,11 @@ export function extractNewsMorningCards(html) {
return cards;
}
/** 统计页面里的新闻卡片数(含 news001 的 .card/.highlight-box 与 news002 的 .lead-card)。 */
export function countNewsMorningStoryCards(html) {
return extractNewsMorningCards(html).filter((card) => card.title).length;
}
/** 提取 section 区块边界,用于判断某个栏目是否被清空。 */
export function extractNewsMorningSections(html) {
const source = String(html ?? '');
+13
View File
@@ -162,3 +162,16 @@ test('formatNewsMorningDedupeSummary reports reasons and removed sections', () =
assert.match(summary, /剔除 1 条重复(event-key:1)/);
assert.match(summary, /移除空栏目 living/);
});
test('countNewsMorningStoryCards counts news001 and news002 card shapes', async () => {
const { countNewsMorningStoryCards } = await import('./wechat-news-morning-dedup.mjs');
const html = page([
['headlines', [
card({ title: '头条一条', cls: 'lead-card' }),
card({ title: '头条两条', cls: 'card highlight-box' }),
]],
['domestic', [card({ title: '国内一条' })]],
['weather', ['<div class="weather-card"><div class="wc">北京</div></div>']],
]);
assert.equal(countNewsMorningStoryCards(html), 3, 'lead-card/highlight-box 要计入,weather-card 不计入');
});
+4 -3
View File
@@ -21,6 +21,7 @@ import {
uploadWechatDraftBrandAssets,
} from './wechat-draft-article-layout.mjs';
import { getLocalParts, localDateKey, startOfLocalDay } from './schedule-time.mjs';
import { countNewsMorningStoryCards } from './wechat-news-morning-dedup.mjs';
import {
NEWS001_FRESH_CONTENT_SPEC,
NEWS001_SEO_SPEC,
@@ -110,9 +111,9 @@ export function isNewsMorningContentDegraded(html, { minStoryCards = 12 } = {})
if (/关于今日早报的重要说明|未填充任何当日具体新闻条目|未能获取当日可核验新闻/u.test(source)) {
return true;
}
const cards = (source.match(/class="card(?:\s|")/gi) ?? []).length;
const highlights = (source.match(/class="card highlight-box/gi) ?? []).length;
return cards + highlights < minStoryCards;
// 必须按 class token 统计:news002 的头条卡是 a.lead-card,
// 旧写法 `class="card(?:\s|")` 匹配不到,会把正常页误判成降级页。
return countNewsMorningStoryCards(source) < minStoryCards;
}
export function isNewsMorningPageDegraded(page, { minStoryCards = 6 } = {}) {
+13
View File
@@ -314,3 +314,16 @@ test('internals normalize booleans consistently', () => {
assert.equal(wechatNewsMorningDraftInternals.normalizeBoolean('1', false), true);
assert.equal(wechatNewsMorningDraftInternals.normalizeBoolean('off', true), false);
});
test('isNewsMorningContentDegraded counts news002 lead cards', async () => {
const { isNewsMorningContentDegraded } = await import('./wechat-news-morning-draft.mjs');
const leads = Array.from({ length: 4 }, (_, i) =>
`<a class="lead-card" href="#"><div class="body"><h3>头条${i}</h3><p>正文</p></div></a>`).join('');
const cards = Array.from({ length: 4 }, (_, i) =>
`<div class="card" data-event-key="k${i}"><div class="body"><h3>条目${i}</h3><p>正文</p></div></div>`).join('');
const html = `<body><div class="container">${leads}${cards}</div></body>`;
// 旧实现用 `class="card(?:\s|")` 统计,匹配不到 lead-card,4 条头条会被漏数。
assert.equal(isNewsMorningContentDegraded(html, { minStoryCards: 8 }), false);
assert.equal(isNewsMorningContentDegraded(html, { minStoryCards: 9 }), true);
});