Files
memind/wechat/push-content/news-section-extract.mjs
john 830f8a4011
Memind CI / Test, build, and release guards (push) Successful in 5m36s
feat(wechat): add service-account push subscriptions and news morning cover delivery
Ship the 10-item numeric subscription loop on WeChat MP, wire schedule delivery
for news/weather/quote/health/finance/tech/knowledge/night/surprise pushes, and
extend news morning draft cover generation plus a one-shot draft push script.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-09-20 16:15:15 +08:00

214 lines
6.6 KiB
JavaScript

import fs from 'node:fs';
import { buildPublicUrl } from '../../user-publish.mjs';
import {
findLatestNewsMorningPage,
isNewsMorningPageForToday,
} from '../../wechat-news-morning-draft.mjs';
import { resolveNewsMorningPushConfig } from './news-morning-delivery.mjs';
function stripHtml(value) {
return String(value ?? '')
.replace(/<br\s*\/?>/gi, '\n')
.replace(/<[^>]+>/g, '')
.replace(/\s+/g, ' ')
.trim();
}
function truncateText(value, maxChars = 96) {
const text = String(value ?? '').replace(/\s+/g, ' ').trim();
if (!maxChars || text.length <= maxChars) return text;
return `${text.slice(0, Math.max(0, maxChars - 1))}…`;
}
function extractDivBlock(html, className) {
const source = String(html ?? '');
const openRe = new RegExp(`<div class="${className}"[^>]*>`, 'i');
const openMatch = openRe.exec(source);
if (!openMatch) return '';
let pos = openMatch.index + openMatch[0].length;
let depth = 1;
while (pos < source.length && depth > 0) {
const nextOpen = source.indexOf('<div', pos);
const nextClose = source.indexOf('</div>', pos);
if (nextClose === -1) break;
if (nextOpen !== -1 && nextOpen < nextClose) {
depth += 1;
pos = nextOpen + 4;
} else {
depth -= 1;
pos = nextClose + 6;
}
}
return source.slice(openMatch.index, pos);
}
function extractClassDivBlocks(parentHtml, classPrefix) {
const source = String(parentHtml ?? '');
const blocks = [];
const openRe = new RegExp(`<div class="${classPrefix}[^"]*"[^>]*>`, 'gi');
let openMatch = openRe.exec(source);
while (openMatch) {
const start = openMatch.index;
let pos = start + openMatch[0].length;
let depth = 1;
while (pos < source.length && depth > 0) {
const nextOpen = source.indexOf('<div', pos);
const nextClose = source.indexOf('</div>', pos);
if (nextClose === -1) break;
if (nextOpen !== -1 && nextOpen < nextClose) {
depth += 1;
pos = nextOpen + 4;
} else {
depth -= 1;
pos = nextClose + 6;
}
}
blocks.push(source.slice(start, pos));
openMatch = openRe.exec(source);
}
return blocks;
}
export function extractDailyNewsSectionHtml(html, sectionId) {
const source = String(html ?? '');
const id = String(sectionId ?? '').trim();
if (!id) return '';
const sectionRe = new RegExp(
`<div class="section" id="${id}"[\\s\\S]*?(?=<div class="section" id=|<div class="footer"|<footer|</body>)`,
'i',
);
return source.match(sectionRe)?.[0] ?? '';
}
export function extractDailyNewsHeroDate(html) {
return stripHtml(
String(html).match(/<div class="date-badge"[^>]*>([\s\S]*?)<\/div>/i)?.[1]
?? String(html).match(/<span class="date"[^>]*>([\s\S]*?)<\/span>/i)?.[1]
?? '',
);
}
export function extractCardsFromHtmlFragment(html) {
const parts = String(html ?? '').split(/<div class="card(?:\s|")/i).slice(1);
return parts.map((block, index) => {
const title = stripHtml(
block.match(/<h3[^>]*>([\s\S]*?)<\/h3>/i)?.[1]
?? block.match(/<h2[^>]*>([\s\S]*?)<\/h2>/i)?.[1]
?? '',
);
const tag = stripHtml(
block.match(/<span class="[^"]*\btag\b[^"]*"[^>]*>([\s\S]*?)<\/span>/i)?.[1] ?? '',
);
const paragraphs = [...block.matchAll(/<p[^>]*>([\s\S]*?)<\/p>/gi)]
.map((item) => stripHtml(item[1]))
.filter(Boolean);
const context = paragraphs[0] ?? '';
return {
index: index + 1,
title,
tag,
context: truncateText(context, 120),
};
}).filter((item) => item.title);
}
export function extractKnowledgeItemsFromHtml(html) {
const sectionHtml = extractDailyNewsSectionHtml(html, 'knowledge');
const blocks = extractClassDivBlocks(sectionHtml, 'knowledge-item');
return blocks.map((block, index) => {
const icon = stripHtml(block.match(/<div class="k-icon"[^>]*>([\s\S]*?)<\/div>/i)?.[1] ?? '');
const title = stripHtml(block.match(/<h4[^>]*>([\s\S]*?)<\/h4>/i)?.[1] ?? '');
const body = truncateText(
stripHtml(block.match(/<p[^>]*>([\s\S]*?)<\/p>/i)?.[1] ?? ''),
180,
);
if (!title) return null;
return { index: index + 1, title: icon ? `${icon} ${title}` : title, body };
}).filter(Boolean);
}
export function extractCardsFromSectionIds(html, sectionIds = []) {
const cards = [];
for (const sectionId of sectionIds) {
const sectionHtml = extractDailyNewsSectionHtml(html, sectionId);
if (!sectionHtml) continue;
cards.push(...extractCardsFromHtmlFragment(sectionHtml));
}
return cards.map((card, index) => ({ ...card, index: index + 1 }));
}
export function buildSectionCardsPushText({
emoji,
label,
dateLabel = '',
cards = [],
maxItems = 5,
includeSummary = false,
publicUrl = '',
cancelId,
unavailableText = '内容正在整理中,稍后会推送到本服务号。',
}) {
const lines = [`${emoji} ${label}${dateLabel ? ` · ${dateLabel}` : ''}`];
const picked = cards.slice(0, Math.max(1, Math.min(10, Number(maxItems) || 5)));
if (!picked.length) {
lines.push('', unavailableText);
} else {
lines.push('');
for (const card of picked) {
const prefix = card.tag ? `[${card.tag}] ` : '';
if (includeSummary && card.context) {
lines.push(`${card.index}. ${prefix}${card.title}`, ` ${card.context}`, '');
} else {
lines.push(`${card.index}. ${prefix}${card.title}`);
}
}
if (includeSummary) {
while (lines.at(-1) === '') lines.pop();
}
}
if (publicUrl) lines.push('', `👉 阅读全文:${publicUrl}`);
if (cancelId) lines.push('', `回复「取消${cancelId}」关闭推送`);
return lines.join('\n');
}
export async function loadTodayNewsMorningHtml({
now = Date.now(),
timezone = 'Asia/Shanghai',
env = process.env,
h5Root = process.cwd(),
pool = null,
} = {}) {
const config = await resolveNewsMorningPushConfig({ pool, env });
const effectiveTimezone = config.timezone || timezone;
if (!config.sourceUserId) {
return { html: null, config, dateLabel: '', publicUrl: '' };
}
const page = findLatestNewsMorningPage({
h5Root,
userId: config.sourceUserId,
slugPattern: config.pageSlugPattern,
date: new Date(now),
timezone: effectiveTimezone,
});
if (
!page?.localPath
|| !fs.existsSync(page.localPath)
|| !isNewsMorningPageForToday(page, { date: new Date(now), timezone: effectiveTimezone })
) {
return { html: null, config, dateLabel: '', publicUrl: '' };
}
const html = fs.readFileSync(page.localPath, 'utf8');
return {
html,
config,
dateLabel: extractDailyNewsHeroDate(html),
publicUrl: buildPublicUrl(config.publicBaseUrl, config.sourceUserId, page.relativePath),
};
}