归档 artalk-cf 评论后端 + rss-robot 到 blog-admin(含技术选型/模块分布 README)

This commit is contained in:
zqlit committed 2026-10-04 08:45:40 +08:00
1 parent e1323bd260
commit 299ab57e4d
98 files changed
+15657

No files matched your search

+406
View File
@@ -0,0 +1,406 @@
// 抓取主流程:遍历订阅源 → 抓取 → 解析 → 去重 → 通知 → 写 KV
// 迁自 check-feeds.js 的 resolveFeed + main,作为 CF Cron 的 scheduled handler。
import type { Env } from '../../types';
import { fetchUrl } from './fetch';
import { discoverFeedUrl, tryCommonFeedPaths } from './discover';
import {
parseFeedXml,
parseJsonFeed,
parseCustomJson,
isJsonResponse,
type Article,
type FeedConfig,
type ParsedFeed,
} from './parse';
import { sendFeishuNotification } from './notify';
import { kvGetJson, kvPutJson } from './util';
const MAX_SEEN = 500;
/** 单源抓取超时(原来 30s:一个挂掉的源能把整批拖到 4 分钟以上) */
const FEED_TIMEOUT = 8000;
/** 连续失败这么多次就进入退避期,不再每次抓(否则每小时白等 8 秒) */
const FAIL_THRESHOLD = 3;
/** 退避时长:6 小时(期间只在其它源都抓完后才可能被跳过) */
const FAIL_BACKOFF_MS = 6 * 3600 * 1000;
interface FailRecord { n: number; at: number }
/** 抓取 + 解析单个订阅源(含 feed 自动发现)。失败返回 null。 */
async function resolveFeed(env: Env, feedConfig: FeedConfig): Promise<ParsedFeed | null> {
const url = feedConfig.url;
const format = feedConfig.format || 'auto';
const forceProxy = feedConfig.proxy === true;
let response;
try {
response = await fetchUrl(env, url, FEED_TIMEOUT, forceProxy);
} catch {
// 直连 + 代理都失败:不再重复抓同一个地址(原来会白等一个超时),
// 直接把常见的 feed 路径猜一遍就放弃。
response = undefined;
}
if (!response) {
const guessedUrl = await tryCommonFeedPaths(env, url);
if (!guessedUrl) return null;
try {
response = await fetchUrl(env, guessedUrl, FEED_TIMEOUT, forceProxy);
} catch {
return null;
}
}
const { text, contentType } = response;
const isJson = isJsonResponse(text, contentType);
if (format === 'json' || (format === 'auto' && isJson)) {
try {
const json = JSON.parse(text);
if (
json.version?.includes('jsonfeed.org') ||
(json.items && Array.isArray(json.items) && !feedConfig.path)
) {
return parseJsonFeed(json, url);
}
return parseCustomJson(json, feedConfig);
} catch {
return null;
}
}
if (
text.includes('<rss') ||
text.includes('<feed') ||
text.includes('<channel') ||
text.includes('<entry')
) {
return parseFeedXml(text, url);
}
// HTML 发现
const feedUrl = discoverFeedUrl(text, url);
if (feedUrl) {
const feedResp = await fetchUrl(env, feedUrl, FEED_TIMEOUT, forceProxy);
if (feedResp) {
if (isJsonResponse(feedResp.text, feedResp.contentType)) {
try {
const json = JSON.parse(feedResp.text);
if (json.version?.includes('jsonfeed.org') || json.items) return parseJsonFeed(json, feedUrl);
return parseCustomJson(json, { url: feedUrl });
} catch {
return null;
}
}
return parseFeedXml(feedResp.text, feedUrl);
}
}
const guessedUrl = await tryCommonFeedPaths(env, url);
if (guessedUrl) {
const guessResp = await fetchUrl(env, guessedUrl, FEED_TIMEOUT, forceProxy);
if (guessResp) {
if (isJsonResponse(guessResp.text, guessResp.contentType)) {
try {
const json = JSON.parse(guessResp.text);
if (json.version?.includes('jsonfeed.org') || json.items) return parseJsonFeed(json, guessedUrl);
return parseCustomJson(json, { url: guessedUrl });
} catch {
return null;
}
}
return parseFeedXml(guessResp.text, guessedUrl);
}
}
return null;
}
/** 从远程 JSON / KV 加载订阅源列表 */
async function loadFeeds(env: Env): Promise<FeedConfig[]> {
// 优先从 KV 读(管理页维护的最新订阅源)
const kvFeeds = await kvGetJson<{ feeds: FeedConfig[] }>(env, 'feeds_config', { feeds: [] });
if (kvFeeds.feeds && kvFeeds.feeds.length > 0) {
return kvFeeds.feeds.map((f) => (typeof f === 'string' ? { url: f } : f));
}
// 回退:FEEDS_URL 远程 JSON
if (env.FEEDS_URL) {
try {
const res = await fetch(env.FEEDS_URL, {
headers: { 'User-Agent': 'Mozilla/5.0 (compatible; RSSBot/1.0)' },
});
if (res.ok) {
const json = (await res.json()) as any;
if (Array.isArray(json)) {
return json.map((item: string | FeedConfig) =>
typeof item === 'string' ? { url: item } : item,
);
}
if (json.feeds && Array.isArray(json.feeds)) {
return json.feeds.map((item: string | FeedConfig) =>
typeof item === 'string' ? { url: item } : item,
);
}
}
} catch {
// 忽略,走空
}
}
return [];
}
interface SeenData {
lastCheck: string | null;
articles: Record<string, string>;
}
/** scheduled 入口:每天抓取一次 */
/**
* 抓取一批订阅源。
* 免费版 Worker 每次调用限 50 个 subrequest,66 个源全量一把抓会炸,
* 所以按 offset/limit 分批轮转(一天两批跑完全部);latest 按 feed 维度合并写入。
*/
export interface CronStats {
/** 本批计划抓取的源数 */
batch: number;
/** 订阅源总数 */
total: number;
ok: number;
failed: number;
/** 本批抓到的文章条数(去重前) */
articles: number;
/** 判定为新文章的条数 */
newArticles: number;
/** 是否 dry-run(不通知、不写 KV) */
dry: boolean;
durationMs: number;
/** 因连续失败处于退避期、本次跳过的源数 */
skipped: number;
/** 最慢的几个源,用于排查拖后腿的 feed */
slowest: { url: string; ms: number }[];
/** 本次失败的源(最多列 10 个,便于排查) */
failedUrls: string[];
}
export async function runCron(
env: Env,
opts?: { offset?: number; limit?: number; dry?: boolean; rotate?: boolean },
): Promise<CronStats> {
const startedAt = Date.now();
const dry = opts?.dry === true;
const rotate = opts?.rotate === true;
const offset = Math.max(opts?.offset ?? 0, 0);
const batchLimit = Math.max(opts?.limit ?? 0, 0);
const timings: { url: string; ms: number }[] = [];
let okCount = 0;
let failedCount = 0;
let articleCount = 0;
let skippedCount = 0;
const failedUrls: string[] = [];
const emptyStats: CronStats = {
batch: 0, total: 0, ok: 0, failed: 0, articles: 0, newArticles: 0,
dry, durationMs: 0, skipped: 0, slowest: [], failedUrls: [],
};
console.log('🔍 开始检查 RSS Feed...');
const allFeeds = await loadFeeds(env);
if (allFeeds.length === 0) {
console.log('⚠️ 没有可用的订阅源');
return emptyStats;
}
// 失败退避表:连续失败 >= 阈值的源,在退避期内不再每次白等超时
const failMap = await kvGetJson<Record<string, FailRecord>>(env, 'rss_fail', {});
const nowTs = Date.now();
const inBackoff = (u: string): boolean => {
const r = failMap[u];
return !!r && r.n >= FAIL_THRESHOLD && nowTs - r.at < FAIL_BACKOFF_MS;
};
const markFail = (u: string): void => {
const r = failMap[u];
failMap[u] = { n: (r?.n ?? 0) + 1, at: nowTs };
};
const clearFail = (u: string): void => {
if (failMap[u]) delete failMap[u];
};
console.log(`📋 共 ${allFeeds.length} 个订阅源`);
for (const u of Object.keys(failMap)) if (!inBackoff(u)) delete failMap[u]; // 退避期满自动重试
// 环形取本批:offset 开始取 limit 个(limit=0 表示全量)。
// rotate=true 时从 KV 里的游标继续(定时任务用,保证每轮都能覆盖到所有源)。
const feeds: FeedConfig[] = [];
let cursorAfter = offset;
if (batchLimit > 0 && allFeeds.length > batchLimit) {
let start = offset % allFeeds.length;
if (rotate && !dry) {
const saved = await env.RSS_KV.get('rss_cursor');
const n = parseInt(saved || '0', 10);
if (Number.isFinite(n)) start = ((n % allFeeds.length) + allFeeds.length) % allFeeds.length;
}
let examined = 0;
let i = start;
while (feeds.length < batchLimit && examined < allFeeds.length) {
const f = allFeeds[i % allFeeds.length];
i += 1;
examined += 1;
if (inBackoff(f.url)) {
skippedCount += 1;
continue;
}
feeds.push(f);
}
cursorAfter = i % allFeeds.length;
if (rotate && !dry) await env.RSS_KV.put('rss_cursor', String(cursorAfter));
console.log(
`📋 本批 ${feeds.length} 个(cursor=${start}${skippedCount ? `,跳过退避中 ${skippedCount} 个` : ''})`,
);
} else {
feeds.push(...allFeeds);
}
const seenData = await kvGetJson<SeenData>(env, 'seen_articles', {
lastCheck: null,
articles: {},
});
const allNewArticles: Article[] = [];
const allFetchedArticles: (Article & { siteUrl: string })[] = [];
for (const feedConfig of feeds) {
const url = feedConfig.url;
console.log(`🔍 检查: ${url}`);
const t0 = Date.now();
try {
const result = await resolveFeed(env, feedConfig);
timings.push({ url, ms: Date.now() - t0 });
if (!result) {
failedCount++;
markFail(url);
if (failedUrls.length < 10) failedUrls.push(url);
console.log(' ❌ 未找到 Feed');
continue;
}
okCount++;
articleCount += result.articles.length;
clearFail(url);
const { feedTitle, articles } = result;
console.log(` 📰 ${feedTitle || '未知'} - 共 ${articles.length} 篇`);
const recentArticles = articles.slice(0, 10);
allFetchedArticles.push(
...recentArticles.map((a) => ({ ...a, feedTitle: feedTitle || '未知', siteUrl: url })),
);
const newOnes = recentArticles.filter((a) => !seenData.articles[a.link]);
if (newOnes.length > 0) {
console.log(` 🆕 发现 ${newOnes.length} 篇新文章`);
allNewArticles.push(...newOnes);
} else {
console.log(' ✅ 无新文章');
}
for (const a of recentArticles) {
seenData.articles[a.link] = a.pubDate;
}
} catch (err) {
timings.push({ url, ms: Date.now() - t0 });
failedCount++;
markFail(url);
if (failedUrls.length < 10) failedUrls.push(url);
console.error(` ❌ 抓取失败: ${(err as Error).message}`);
}
}
// 通知(仅最近 24h 内的)
const oneDayAgo = Date.now() - 86400000;
const recentNew = allNewArticles.filter(
(a) => new Date(a.pubDate).getTime() > oneDayAgo || !seenData.lastCheck,
);
if (dry) {
console.log(`🧪 dry-run:跳过通知(本应通知 ${recentNew.length} 篇)与 KV 写入`);
} else if (recentNew.length > 0) {
await sendFeishuNotification(env, recentNew);
} else if (allNewArticles.length > 0) {
console.log('ℹ️ 发现新文章但超过 24 小时,不推送');
} else {
console.log('✅ 所有博客均无新文章');
}
// 写 latest 缓存(供 /api/results、/api/articles 读)
// 分批模式:本批源的条目替换旧缓存里的同源条目,其他源的保留
const byFeed: Record<string, { articles: Article[]; siteUrl: string }> = {};
for (const a of allFetchedArticles) {
const key = a.feedTitle || '未知';
if (!byFeed[key]) byFeed[key] = { articles: [], siteUrl: a.siteUrl || '' };
byFeed[key].articles.push({ title: a.title, link: a.link, pubDate: a.pubDate, author: a.author, feedTitle: key });
}
if (!dry) {
try {
await kvPutJson(env, 'rss_fail', failMap);
} catch {
/* 失败表写失败不影响主流程 */
}
}
const durationMs = Date.now() - startedAt;
const stats: CronStats = {
batch: feeds.length,
total: allFeeds.length,
ok: okCount,
failed: failedCount,
articles: articleCount,
newArticles: allNewArticles.length,
dry,
durationMs,
skipped: skippedCount,
slowest: timings.sort((a, b) => b.ms - a.ms).slice(0, 5),
failedUrls,
};
if (dry) return stats;
const prev = await kvGetJson<{ timestamp: string | null; total: number; feeds: { name: string; siteUrl: string; favicon: string; articles: Article[] }[] }>(
env,
'latest',
{ timestamp: null, total: 0, feeds: [] },
);
const batchUrls = new Set(feeds.map((f) => f.url));
const kept = prev.feeds.filter((f) => !batchUrls.has(f.siteUrl || f.name));
const mergedFeeds = [
...kept,
...Object.entries(byFeed).map(([name, { articles, siteUrl }]) => ({
name,
siteUrl,
// 必须写绝对地址:友圈页在 usj.cc 上,相对路径 /api/* 会 404 → 退化成第三方默认头像
favicon: siteUrl
? `${env.PUBLIC_API_BASE || 'https://api.200181.xyz'}/api/favicon?url=${encodeURIComponent(siteUrl)}`
: '',
articles,
})),
];
const mergedTotal = mergedFeeds.reduce((n, f) => n + f.articles.length, 0);
await kvPutJson(env, 'latest', {
timestamp: new Date().toISOString(),
total: mergedTotal,
feeds: mergedFeeds,
});
// 清理 + 保存去重记录
const entries = Object.entries(seenData.articles);
if (entries.length > MAX_SEEN) {
entries.sort((a, b) => new Date(b[1]).getTime() - new Date(a[1]).getTime());
seenData.articles = Object.fromEntries(entries.slice(0, MAX_SEEN));
}
seenData.lastCheck = new Date().toISOString();
await kvPutJson(env, 'seen_articles', seenData);
console.log(`📝 已保存去重记录 (${Object.keys(seenData.articles).length} 条)`);
return stats;
}
+51
View File
@@ -0,0 +1,51 @@
// Feed 自动发现:从 HTML 里找 RSS/Atom/JSON Feed 链接,或试常见路径
// 迁自 check-feeds.js 的 discoverFeedUrl / tryCommonFeedPaths
import { probeDirect } from './fetch';
import type { Env } from '../../types';
export function discoverFeedUrl(html: string, baseUrl: string): string | null {
const patterns = [
/<link[^>]+type=["']application\/rss\+xml["'][^>]+href=["']([^"']+)["']/i,
/<link[^>]+href=["']([^"']+)["'][^>]+type=["']application\/rss\+xml["']/i,
/<link[^>]+type=["']application\/atom\+xml["'][^>]+href=["']([^"']+)["']/i,
/<link[^>]+href=["']([^"']+)["'][^>]+type=["']application\/atom\+xml["']/i,
/<link[^>]+type=["']application\/feed\+json["'][^>]+href=["']([^"']+)["']/i,
/<link[^>]+href=["']([^"']+)["'][^>]+type=["']application\/feed\+json["']/i,
/<link[^>]+type=["']application\/json["'][^>]+title=["'][^"']*feed[^"']*["'][^>]+href=["']([^"']+)["']/i,
];
for (const pattern of patterns) {
const match = html.match(pattern);
if (match) {
let href = match[1];
const urlObj = new URL(baseUrl);
if (href.startsWith('/')) {
href = `${urlObj.protocol}//${urlObj.host}${href}`;
} else if (!href.startsWith('http')) {
href = `${urlObj.protocol}//${urlObj.host}/${href}`;
}
return href;
}
}
return null;
}
export async function tryCommonFeedPaths(env: Env, baseUrl: string): Promise<string | null> {
const urlObj = new URL(baseUrl);
// 只留最常见的三个:候选多一个,最坏情况就多等一个超时(原来的 9 个候选能把
// 单个源拖到 180 秒以上,整批跑几分钟)
const paths = ['/feed', '/feed.xml', '/atom.xml'];
for (const path of paths) {
const candidate = `${urlObj.protocol}//${urlObj.host}${path}`;
try {
const { contentType } = await probeDirect(candidate, 4000);
const ct = contentType.toLowerCase();
if (ct.includes('xml') || ct.includes('rss') || ct.includes('atom') || ct.includes('json')) {
return candidate;
}
} catch {
// 忽略,继续试
}
}
return null;
}
+152
View File
@@ -0,0 +1,152 @@
// 抓取逻辑:直连 + 腾讯云 SCF 国内代理回退
// 迁自 check-feeds.js 的 fetchUrl / fetchViaProxy。
//
// 关键:CF Worker 跑在境外节点,抓国内博客可能被拦。
// 复用已搭好的腾讯云 SCF(scfapi.usj.cc,国内 IP)作为回退代理。
// 原 EdgeOne 的 /api/proxy 也是境外节点,迁移后不再需要——Worker 直连即等价。
import type { Env } from '../../types';
// 单源超时:原来 30s 太长(一个挂掉的源能把整批拖到 4 分钟以上)。
// 正常 RSS 1-3 秒就能回来,8 秒足够;代理(国内 SCF)多给 4 秒余量。
const REQUEST_TIMEOUT = 8000;
const PROXY_TIMEOUT_EXTRA = 4000;
// ── 代理熔断 ─────────────────────────────────────────────────────────────
// 国内代理(scfapi.usj.cc)证书过期/挂掉时,每个源都要先直连超时、再代理超时,
// 一批 20 多个源能白等好几分钟。连续失败若干次后就暂时跳过代理,省掉这一半时间。
let proxyFailStreak = 0;
let proxySkipUntil = 0;
const PROXY_FAIL_THRESHOLD = 5;
const PROXY_SKIP_MS = 10 * 60 * 1000;
export interface FetchResult {
text: string;
contentType: string;
via: 'direct' | 'proxy';
}
/** 直连抓取 */
async function fetchDirect(url: string, timeout: number): Promise<FetchResult> {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeout);
try {
const res = await fetch(url, {
signal: controller.signal,
redirect: 'follow',
headers: {
'User-Agent': 'Mozilla/5.0 (compatible; RSSBot/1.0)',
Accept: 'text/html,application/xhtml+xml,application/xml,application/json;q=0.9,*/*;q=0.8',
},
});
if (!res.ok) throw new Error(`HTTP ${res.status}`);
const text = await res.text();
const contentType = (res.headers.get('content-type') || '').toLowerCase();
return { text, contentType, via: 'direct' };
} finally {
clearTimeout(timer);
}
}
/** 通过腾讯云 SCF 国内代理抓取 */
async function fetchViaSCF(
env: Env,
url: string,
timeout: number,
): Promise<FetchResult> {
const proxyUrl = env.SCF_PROXY_URL;
if (!proxyUrl) throw new Error('SCF_PROXY_URL not configured');
if (Date.now() < proxySkipUntil) throw new Error('proxy temporarily disabled (recent failures)');
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeout);
try {
const res = await fetch(proxyUrl, {
method: 'POST',
signal: controller.signal,
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ url, timeout }),
});
// 代理这一层通了(HTTP 2xx)就不算代理故障
if (res.ok) {
proxyFailStreak = 0;
proxySkipUntil = 0;
}
const data = (await res.json().catch(() => ({}))) as {
ok?: boolean;
status?: number;
error?: string;
body?: string;
contentType?: string;
};
if (!res.ok) {
throw new Error(`HTTP ${res.status}`);
}
if (!data.ok) {
// 代理连得上,是目标站点抓不动(站点挂了/被墙)→ 不熔断代理
throw new Error(data.error || 'target fetch failed');
}
if (data.status && data.status >= 400) throw new Error(`目标 HTTP ${data.status}`);
proxyFailStreak = 0;
return {
text: data.body || '',
contentType: (data.contentType || '').toLowerCase(),
via: 'proxy',
};
} catch (err) {
const msg = (err as Error).message || '';
// 只有"连不上代理/代理返回错误状态"才计故障;目标站点失败不算
const proxyLevelFailure = /^(HTTP \d|proxy temporarily disabled|SCF_PROXY_URL)/.test(msg) ||
/fetch failed|Network|TLS|abort/i.test(msg);
if (proxyLevelFailure && !/target fetch failed/.test(msg)) {
proxyFailStreak += 1;
if (proxyFailStreak >= PROXY_FAIL_THRESHOLD) {
proxySkipUntil = Date.now() + PROXY_SKIP_MS;
proxyFailStreak = 0;
console.log('⚠️ 国内代理连续失败,暂时跳过代理(10 分钟)');
}
}
throw err;
} finally {
clearTimeout(timer);
}
}
/**
* 只走直连的轻量探测:用于"猜常见 feed 路径"。
* 猜路径最多试几个候选,如果每个都再走一遍代理,一个挂掉的源能拖 3 分钟
* (实测有单源 237 秒),所以探测只用直连 + 短超时。
*/
export async function probeDirect(url: string, timeout = 4000): Promise<FetchResult> {
return fetchDirect(url, timeout);
}
/**
* 抓取一个 URL:先直连,失败回退 SCF 代理。
* 与 check-feeds.js 一致:feed.proxy === true 时强制走代理。
*/
export async function fetchUrl(
env: Env,
url: string,
timeout = REQUEST_TIMEOUT,
forceProxy = false,
): Promise<FetchResult> {
const proxyTimeout = timeout + PROXY_TIMEOUT_EXTRA;
if (forceProxy) {
return fetchViaSCF(env, url, proxyTimeout);
}
try {
return await fetchDirect(url, timeout);
} catch (directErr) {
// 直连失败 → 回退国内代理
try {
const viaProxy = await fetchViaSCF(env, url, proxyTimeout);
return viaProxy;
} catch (proxyErr) {
throw new Error(
`直连失败(${(directErr as Error).message}),代理也失败(${(proxyErr as Error).message})`,
);
}
}
}
+85
View File
@@ -0,0 +1,85 @@
// 默认问候语词库(迁自 edgeone/functions/api/ai/greeting.js 的 defaultPool)
export interface Greeting {
text: string;
time: string | null;
holiday: string | null;
}
export function defaultPool(): Greeting[] {
return [
{ text: '早上好,今天的咖啡够浓吗', time: 'weekday-morning', holiday: null },
{ text: '新的一天,新的 bug 等着你', time: 'weekday-morning', holiday: null },
{ text: '周一的闹钟总是响得特别早', time: 'weekday-morning', holiday: null },
{ text: '周二了,距离周末还有 3 天', time: 'weekday-morning', holiday: null },
{ text: '周三,一周的折返点', time: 'weekday-morning', holiday: null },
{ text: '周四了,胜利在望', time: 'weekday-morning', holiday: null },
{ text: '周五早晨的空气都是甜的', time: 'weekday-morning', holiday: null },
{ text: '上班前来看看博客吧', time: 'weekday-morning', holiday: null },
{ text: '通勤路上,刷一篇好文章', time: 'weekday-morning', holiday: null },
{ text: '阳光正好,写点什么吧', time: 'morning', holiday: null },
{ text: '一杯茶,一篇文章,一个上午', time: 'morning', holiday: null },
{ text: '灵感总在上午悄悄来访', time: 'morning', holiday: null },
{ text: '窗外鸟鸣,键盘轻敲', time: 'morning', holiday: null },
{ text: '午饭吃好了吗,来读篇博客', time: 'noon', holiday: null },
{ text: '午休时间,偷得浮生一刻闲', time: 'noon', holiday: null },
{ text: '饱了才有力气写代码', time: 'noon', holiday: null },
{ text: '午餐后的惬意,属于博客时光', time: 'noon', holiday: null },
{ text: '午后阳光很暖,文字也很温柔', time: 'afternoon', holiday: null },
{ text: '来杯下午茶,配一篇好博客', time: 'afternoon', holiday: null },
{ text: '不想工作的时候,就读博客吧', time: 'afternoon', holiday: null },
{ text: '下午三点,正是摸鱼好时光', time: 'afternoon', holiday: null },
{ text: '代码写累了,换个脑子', time: 'afternoon', holiday: null },
{ text: '夕阳之下,该给今天收个尾了', time: 'evening', holiday: null },
{ text: '晚霞温柔,适合安静地读点东西', time: 'evening', holiday: null },
{ text: '下班了吗,博客等你回家', time: 'evening', holiday: null },
{ text: '暮色四合,一天又悄悄过去了', time: 'evening', holiday: null },
{ text: '夜深了,只有你还在折腾博客吧', time: 'night', holiday: null },
{ text: '凌晨三点,灵感比白天更活跃', time: 'night', holiday: null },
{ text: '熬夜冠军,博客世界永远亮着灯', time: 'night', holiday: null },
{ text: '星星都睡了,你的博客还醒着', time: 'night', holiday: null },
{ text: '深夜的代码写给自己看', time: 'night', holiday: null },
{ text: '失眠的夜晚,幸好有博客陪伴', time: 'night', holiday: null },
{ text: '周末不用早起,但可以早起写博客', time: 'weekend', holiday: null },
{ text: '窝在沙发里,手机刷博客', time: 'weekend', holiday: null },
{ text: '周末的早晨,适合赖床和码字', time: 'weekend', holiday: null },
{ text: '终于有空了,把攒了一周的文章读完', time: 'weekend', holiday: null },
{ text: '周末宅家,博客是最好的伴侣', time: 'weekend', holiday: null },
{ text: '咖啡 + 面包 + 博客 = 完美周末', time: 'weekend', holiday: null },
{ text: '没有 deadline 的周末,写点想写的', time: 'weekend', holiday: null },
{ text: '元旦快乐,新的一年从一篇博客开始', time: null, holiday: '01-01' },
{ text: '新年新气象,博客也要更新啦', time: null, holiday: '01-01' },
{ text: '元旦快乐,今年第一篇写什么', time: null, holiday: '01-01' },
{ text: '情人节快乐,代码和爱情可以兼得', time: null, holiday: '02-14' },
{ text: '今天不写代码,陪 ta 看看博客', time: null, holiday: '02-14' },
{ text: '三八妇女节,致敬所有闪闪发光的她', time: null, holiday: '03-08' },
{ text: '愚人节快乐,今天看到什么都别信', time: null, holiday: '04-01' },
{ text: '劳动节快乐,今天不写代码', time: null, holiday: '05-01' },
{ text: '五一劳动节,劳动者最光荣', time: null, holiday: '05-01' },
{ text: '五四青年节,趁年轻多写点博客', time: null, holiday: '05-04' },
{ text: '六一快乐,谁还不是个孩子呢', time: null, holiday: '06-01' },
{ text: '国庆快乐,祖国繁荣昌盛', time: null, holiday: '10-01' },
{ text: '假期余额不多,抓紧时间写博客', time: null, holiday: '10-01' },
{ text: '圣诞快乐,博客就是你的圣诞老人', time: null, holiday: '12-25' },
{ text: '圣诞夜,许个愿,明年博客涨粉', time: null, holiday: '12-25' },
{ text: '立春了,博客也要焕发新生', time: null, holiday: '02-04' },
{ text: '春分时节,昼夜平分,灵感均分', time: null, holiday: '03-20' },
{ text: '夏至已至,白天很长,文章也可以很长', time: null, holiday: '06-21' },
{ text: '秋分,收获的季节,盘点一下今年的博客', time: null, holiday: '09-23' },
{ text: '冬至了,吃碗饺子暖暖心,写篇博客暖暖手', time: null, holiday: '12-22' },
{ text: '又是美好的一天', time: null, holiday: null },
{ text: '今天想写点什么吗', time: null, holiday: null },
{ text: '来博客串个门吧', time: null, holiday: null },
{ text: '保持好奇心,世界不会无趣', time: null, holiday: null },
{ text: '你有多久没有好好写点东西了', time: null, holiday: null },
{ text: '每个博客都是一扇窗', time: null, holiday: null },
{ text: '写作是和自己的对话', time: null, holiday: null },
{ text: '今天遇到什么有趣的事了吗', time: null, holiday: null },
{ text: '读别人的故事,写自己的心情', time: null, holiday: null },
{ text: '博客是一个人的宇宙', time: null, holiday: null },
{ text: '别让灵感溜走,赶紧码下来', time: null, holiday: null },
{ text: '有人默默关注着你的博客呢', time: null, holiday: null },
{ text: '每个字都是时间的印记', time: null, holiday: null },
{ text: '博客不老,我们永远年轻', time: null, holiday: null },
];
}
+67
View File
@@ -0,0 +1,67 @@
// 通知推送:飞书(邮件后续可加)
// 迁自 check-feeds.js 的 sendFeishuNotification + formatRelativeTime
import type { Env } from '../../types';
import type { Article } from './parse';
function formatRelativeTime(isoDate: string): string {
const diff = Date.now() - new Date(isoDate).getTime();
const minutes = Math.floor(diff / 60000);
if (minutes < 1) return '刚刚';
if (minutes < 60) return `${minutes} 分钟前`;
const hours = Math.floor(minutes / 60);
if (hours < 24) return `${hours} 小时前`;
const days = Math.floor(hours / 24);
if (days < 30) return `${days} 天前`;
return new Date(isoDate).toISOString().slice(0, 10);
}
export async function sendFeishuNotification(env: Env, newArticles: Article[]): Promise<void> {
if (!env.FEISHU_WEBHOOK_URL) {
console.log('⚠️ 未配置 FEISHU_WEBHOOK_URL,跳过飞书通知');
return;
}
const grouped: Record<string, Article[]> = {};
for (const article of newArticles) {
const key = article.feedTitle || '未知博客';
if (!grouped[key]) grouped[key] = [];
grouped[key].push(article);
}
const elements: unknown[] = [];
for (const [feedName, articles] of Object.entries(grouped)) {
elements.push({ tag: 'markdown', content: `**📰 ${feedName}**` });
const articleLines = articles.map(
(a) => `- [${a.title}](${a.link}) <font color="grey">${formatRelativeTime(a.pubDate)}</font>`,
);
elements.push({ tag: 'markdown', content: articleLines.join('\n') });
}
const card = {
msg_type: 'interactive',
card: {
header: {
title: { tag: 'plain_text', content: `🔔 博客更新 · ${newArticles.length} 篇新文章` },
template: 'blue',
},
elements,
},
};
try {
const res = await fetch(env.FEISHU_WEBHOOK_URL, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(card),
});
const result = (await res.json()) as { code?: number };
if (result.code === 0) {
console.log(`✅ 飞书通知发送成功 (${newArticles.length} 篇新文章)`);
} else {
console.error('❌ 飞书通知发送失败:', result);
}
} catch (err) {
console.error('❌ 飞书通知发送异常:', (err as Error).message);
}
}
+208
View File
@@ -0,0 +1,208 @@
// RSS / Atom / JSON Feed / 自定义 JSON 解析
// 迁自 check-feeds.js 的 parseFeedXml / parseJsonFeed / parseCustomJson 等。
export interface Article {
title: string;
link: string;
pubDate: string;
author: string;
feedTitle: string;
}
export interface FeedConfig {
url: string;
feedTitle?: string;
format?: string;
path?: string;
proxy?: boolean;
mapping?: Record<string, string>;
}
export interface ParsedFeed {
feedTitle: string;
articles: Article[];
}
function extractTag(xml: string, tag: string): string {
const regex = new RegExp(`<${tag}[^>]*>([\\s\\S]*?)<\\/${tag}>`, 'i');
const match = xml.match(regex);
return match ? match[1] : '';
}
function decodeXml(str: string): string {
return str
.replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
.replace(/&amp;/g, '&')
.replace(/&lt;/g, '<')
.replace(/&gt;/g, '>')
.replace(/&quot;/g, '"')
.replace(/&apos;/g, "'");
}
export function parseDate(val: unknown): string | null {
if (!val) return null;
if (typeof val === 'number') {
const d = new Date(val > 9999999999 ? val : val * 1000);
return isNaN(d.getTime()) ? null : d.toISOString();
}
const str = String(val).trim();
if (!str) return null;
const d = new Date(str);
if (isNaN(d.getTime())) return null;
try {
return d.toISOString();
} catch {
return null;
}
}
export function parseFeedXml(xml: string, feedUrl: string): ParsedFeed {
const articles: Article[] = [];
let feedTitle = '';
const titleMatch = xml.match(/<title[^>]*>([\s\S]*?)<\/title>/);
if (titleMatch) feedTitle = decodeXml(titleMatch[1]).trim();
const itemRegex = /<item[\s>]?>([\s\S]*?)<\/item>/gi;
const entryRegex = /<entry[\s>]?>([\s\S]*?)<\/entry>/gi;
const parseItem = (itemXml: string) => {
const title = extractTag(itemXml, 'title');
let link = '';
const rssLink = extractTag(itemXml, 'link');
if (rssLink) link = rssLink;
const atomLinkMatch = itemXml.match(/<link[^>]+href=["']([^"']+)["'][^>]*>/i);
if (atomLinkMatch && atomLinkMatch[1]) link = atomLinkMatch[1];
const altLinkMatch = itemXml.match(
/<link[^>]+rel=["']alternate["'][^>]+href=["']([^"']+)["']/i,
);
if (altLinkMatch) link = altLinkMatch[1];
const pubDate =
extractTag(itemXml, 'pubDate') ||
extractTag(itemXml, 'published') ||
extractTag(itemXml, 'updated') ||
extractTag(itemXml, 'dc:date') ||
extractTag(itemXml, 'lastBuildDate') ||
extractTag(itemXml, 'date') ||
extractTag(itemXml, 'modified') ||
extractTag(itemXml, 'created');
const author = extractTag(itemXml, 'dc:creator') || extractTag(itemXml, 'author') || '';
const authorName = author.replace(/<name>([\s\S]*?)<\/name>/gi, '$1').trim();
if (title && link) {
articles.push({
title: decodeXml(title).trim(),
link: link.trim(),
pubDate: parseDate(pubDate) || new Date().toISOString(),
author: decodeXml(authorName).trim(),
feedTitle: feedTitle || feedUrl,
});
}
};
let match: RegExpExecArray | null;
while ((match = itemRegex.exec(xml)) !== null) parseItem(match[1]);
while ((match = entryRegex.exec(xml)) !== null) parseItem(match[1]);
return { feedTitle, articles };
}
export function parseJsonFeed(json: any, feedUrl: string): ParsedFeed {
const articles: Article[] = [];
const feedTitle = json.title || feedUrl;
const items = json.items || [];
for (const item of items) {
const title = item.title || '';
const link = item.url || item.id || '';
const pubDate =
item.date_published || item.date_modified || item.date || item.published || item.modified || item.created_at || item.createdAt || item.pubDate || item.timestamp || null;
const author = Array.isArray(item.authors)
? item.authors.map((a: any) => a.name).join(', ')
: item.author?.name || item.author || '';
if (title && link) {
articles.push({
title: title.trim(),
link: link.trim(),
pubDate: parseDate(pubDate) || new Date().toISOString(),
author: typeof author === 'string' ? author.trim() : '',
feedTitle,
});
}
}
return { feedTitle, articles };
}
function getNestedValue(obj: any, path?: string): unknown {
if (!obj || !path) return undefined;
return path.split('.').reduce((o: any, key) => o?.[key], obj);
}
export function parseCustomJson(json: any, config: FeedConfig): ParsedFeed {
const articles: Article[] = [];
const mapping = config.mapping || {};
const path = config.path || '';
let data: any = json;
if (path) {
for (const key of path.split('.')) {
if (data && typeof data === 'object') data = data[key];
}
}
const items = Array.isArray(data) ? data : [];
const titleKey = mapping.title || 'title';
const linkKey = mapping.link || 'link';
const pubDateKey = mapping.pubDate || 'pubDate';
const authorKey = mapping.author || 'author';
const feedTitleKey = mapping.feedTitle || 'feedTitle';
for (const item of items) {
const title = getNestedValue(item, titleKey) || '';
const link = getNestedValue(item, linkKey) || getNestedValue(item, 'url') || '';
const pubDate =
getNestedValue(item, pubDateKey) ||
getNestedValue(item, 'publishedAt') ||
getNestedValue(item, 'createdAt') ||
getNestedValue(item, 'created_at') ||
getNestedValue(item, 'date') ||
getNestedValue(item, 'published') ||
getNestedValue(item, 'updatedAt') ||
getNestedValue(item, 'timestamp') ||
getNestedValue(item, 'datePublished') ||
getNestedValue(item, 'dateModified') ||
null;
const author = getNestedValue(item, authorKey) || '';
const feedTitle = getNestedValue(item, feedTitleKey) || config.feedTitle || config.url || '';
if (title && link) {
articles.push({
title: String(title).trim(),
link: String(link).trim(),
pubDate: parseDate(pubDate) || new Date().toISOString(),
author: String(typeof author === 'object' ? '' : author).trim(),
feedTitle: String(feedTitle),
});
}
}
return { feedTitle: config.feedTitle || config.url || '', articles };
}
export function isJsonResponse(text: string, contentType: string): boolean {
if (contentType?.includes('json')) return true;
try {
const parsed = JSON.parse(text);
return typeof parsed === 'object' && parsed !== null;
} catch {
return false;
}
}
+93
View File
@@ -0,0 +1,93 @@
// 国内代理(腾讯云 SCF)体检。
//
// 为什么要在服务器端做:代理挂掉(最常见的原因是 SCF 自定义域名的免费证书 90 天到期、
// 而它不会自动续)会让抓国内博客全部失败,但这件事本地电脑开着才会被想起来。
// 这里放进 Worker 自己的每日 cron:电脑关着也能发提醒邮件。
//
// 判断方式不看证书日期,直接"能不能抓到东西" —— 用国内知名站点探测,
// 任意一个成功就认为代理可用(比解析证书更贴近实际效果)。
import type { Env } from '../../types';
import { notifyAdmin } from '../admin-notify';
/** 探测站:国内可直连、响应快、不易挂 */
const PROBE_URLS = ['https://www.rz.sb', 'https://blog.qydzz.cn', 'https://t-t.live'];
const PROBE_TIMEOUT_MS = 12000;
export interface ProxyProbe {
url: string;
ok: boolean;
ms: number;
error?: string;
}
export interface ProxyHealth {
/** 至少一个探测站成功 */
ok: boolean;
probes: ProxyProbe[];
checkedAt: string;
}
/** 通过代理抓一个 URL(与 RSS 抓取走同一条链路) */
async function probe(env: Env, target: string): Promise<ProxyProbe> {
const started = Date.now();
const proxyUrl = env.SCF_PROXY_URL;
if (!proxyUrl) return { url: target, ok: false, ms: 0, error: 'SCF_PROXY_URL not configured' };
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), PROBE_TIMEOUT_MS);
try {
const res = await fetch(proxyUrl, {
method: 'POST',
signal: controller.signal,
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ url: target, timeout: PROBE_TIMEOUT_MS }),
});
const data = (await res.json().catch(() => ({}))) as { ok?: boolean; error?: string };
const ms = Date.now() - started;
if (res.ok && data.ok) return { url: target, ok: true, ms };
return { url: target, ok: false, ms, error: data.error || `HTTP ${res.status}` };
} catch (e) {
return { url: target, ok: false, ms: Date.now() - started, error: (e as Error).message };
} finally {
clearTimeout(timer);
}
}
export async function checkProxyHealth(env: Env): Promise<ProxyHealth> {
const probes: ProxyProbe[] = [];
for (const target of PROBE_URLS) {
const r = await probe(env, target);
probes.push(r);
if (r.ok) break; // 有一个通就算代理没问题,不用继续
}
return {
ok: probes.some((p) => p.ok),
probes,
checkedAt: new Date().toISOString(),
};
}
/** 代理不可用 → 给站长发提醒(48 小时最多一封,避免重复刷屏) */
export async function notifyProxyDown(env: Env, health: ProxyHealth) {
const lines = health.probes
.map((p) => `· ${p.url}:${p.ok ? `OK(${p.ms}ms)` : `失败(${p.error || '未知'})`}`)
.join('\n');
const text =
`国内代理 scfapi.usj.cc 体检不通过:${health.probes.length} 个探测站全部抓不到。\n\n` +
`探测结果:\n${lines}\n\n` +
'影响:Cloudflare Worker 抓国内博客时的代理回退会全部失败,友圈/订阅数据会停止更新(直连只对少数境外可访问的站点有效)。\n\n' +
'最常见原因:scfapi.usj.cc 用的是腾讯云 SCF 自定义域名上的免费 DV 证书,90 天到期且不会自动续。\n\n' +
'续期步骤(约 5 分钟):\n' +
'1) 腾讯云 SSL 证书控制台 → 申请免费证书:域名 scfapi.usj.cc,验证方式选「自动 DNS 验证」(usj.cc 托管在 DNSPod,几分钟签发)\n' +
'2) 打开 https://console.cloud.tencent.com/scf/domain?rid=1 (云函数 → 高级能力 → 自定义域名,广州地域)→ 找到 scfapi.usj.cc → 编辑 → HTTPS 里重新选择刚签发的新证书 → 保存\n' +
'3) 验证:curl -s -X POST https://scfapi.usj.cc -H "Content-Type: application/json" -d \'{"url":"https://www.rz.sb","timeout":15000}\' 返回 {"ok":true,...} 即恢复\n\n' +
'(本提醒由服务器端每日任务发出,同一状态最多两天一封;恢复后自动停止。)';
return notifyAdmin(env, {
subject: '【提醒】国内代理 scfapi.usj.cc 不可用(多半是证书过期)',
text,
dedupeKey: 'scf-proxy-down',
minGapHours: 48,
});
}
+46
View File
@@ -0,0 +1,46 @@
// 通用工具:鉴权、CORS、KV 封装
// 把原 EdgeOne 每个 api/*.js 里重复的 requireAuth / respond / corsOptions 收拢到一处
import type { Env } from '../../types';
export const CORS_HEADERS = {
'Content-Type': 'application/json',
'Access-Control-Allow-Origin': '*',
'Access-Control-Allow-Methods': 'GET, POST, DELETE, OPTIONS',
'Access-Control-Allow-Headers': 'Content-Type',
} as const;
export function respond(data: unknown, status = 200): Response {
return new Response(JSON.stringify(data), { status, headers: CORS_HEADERS });
}
export function corsOptions(): Response {
return new Response(null, { status: 204, headers: CORS_HEADERS });
}
/** 校验站点口令:URL ?token= / Cookie site_token=。KV 无口令时放行(与原实现一致)。 */
export async function requireAuth(request: Request, env: Env): Promise<boolean> {
const url = new URL(request.url);
const cookie = request.headers.get('Cookie') || '';
const cookieMatch = cookie.match(/site_token=([^;]+)/);
const token = url.searchParams.get('token') || (cookieMatch ? cookieMatch[1] : '');
const password = await env.RSS_KV.get('site_password');
if (!password) return true; // 未设置口令 → 放行
return token === password;
}
/** 读 KV 里的 JSON,解析失败或不存在返回 fallback */
export async function kvGetJson<T>(env: Env, key: string, fallback: T): Promise<T> {
const raw = await env.RSS_KV.get(key);
if (!raw) return fallback;
try {
return JSON.parse(raw) as T;
} catch {
return fallback;
}
}
export async function kvPutJson(env: Env, key: string, value: unknown): Promise<void> {
await env.RSS_KV.put(key, JSON.stringify(value));
}