// RSS / Atom / JSON Feed / 自定义 JSON 解析 // 迁自 check-feeds.js 的 parseFeedXml / parseJsonFeed / parseCustomJson 等。 export interface Article { title: string; link: string; pubDate: string; author: string; feedTitle: string; } export interface FeedConfig { url: string; feedTitle?: string; format?: string; path?: string; proxy?: boolean; mapping?: Record; } export interface ParsedFeed { feedTitle: string; articles: Article[]; } function extractTag(xml: string, tag: string): string { const regex = new RegExp(`<${tag}[^>]*>([\\s\\S]*?)<\\/${tag}>`, 'i'); const match = xml.match(regex); return match ? match[1] : ''; } function decodeXml(str: string): string { return str .replace(//g, '$1') .replace(/&/g, '&') .replace(/</g, '<') .replace(/>/g, '>') .replace(/"/g, '"') .replace(/'/g, "'"); } export function parseDate(val: unknown): string | null { if (!val) return null; if (typeof val === 'number') { const d = new Date(val > 9999999999 ? val : val * 1000); return isNaN(d.getTime()) ? null : d.toISOString(); } const str = String(val).trim(); if (!str) return null; const d = new Date(str); if (isNaN(d.getTime())) return null; try { return d.toISOString(); } catch { return null; } } export function parseFeedXml(xml: string, feedUrl: string): ParsedFeed { const articles: Article[] = []; let feedTitle = ''; const titleMatch = xml.match(/]*>([\s\S]*?)<\/title>/); if (titleMatch) feedTitle = decodeXml(titleMatch[1]).trim(); const itemRegex = /]?>([\s\S]*?)<\/item>/gi; const entryRegex = /]?>([\s\S]*?)<\/entry>/gi; const parseItem = (itemXml: string) => { const title = extractTag(itemXml, 'title'); let link = ''; const rssLink = extractTag(itemXml, 'link'); if (rssLink) link = rssLink; const atomLinkMatch = itemXml.match(/]+href=["']([^"']+)["'][^>]*>/i); if (atomLinkMatch && atomLinkMatch[1]) link = atomLinkMatch[1]; const altLinkMatch = itemXml.match( /]+rel=["']alternate["'][^>]+href=["']([^"']+)["']/i, ); if (altLinkMatch) link = altLinkMatch[1]; const pubDate = extractTag(itemXml, 'pubDate') || extractTag(itemXml, 'published') || extractTag(itemXml, 'updated') || extractTag(itemXml, 'dc:date') || extractTag(itemXml, 'lastBuildDate') || extractTag(itemXml, 'date') || extractTag(itemXml, 'modified') || extractTag(itemXml, 'created'); const author = extractTag(itemXml, 'dc:creator') || extractTag(itemXml, 'author') || ''; const authorName = author.replace(/([\s\S]*?)<\/name>/gi, '$1').trim(); if (title && link) { articles.push({ title: decodeXml(title).trim(), link: link.trim(), pubDate: parseDate(pubDate) || new Date().toISOString(), author: decodeXml(authorName).trim(), feedTitle: feedTitle || feedUrl, }); } }; let match: RegExpExecArray | null; while ((match = itemRegex.exec(xml)) !== null) parseItem(match[1]); while ((match = entryRegex.exec(xml)) !== null) parseItem(match[1]); return { feedTitle, articles }; } export function parseJsonFeed(json: any, feedUrl: string): ParsedFeed { const articles: Article[] = []; const feedTitle = json.title || feedUrl; const items = json.items || []; for (const item of items) { const title = item.title || ''; const link = item.url || item.id || ''; const pubDate = item.date_published || item.date_modified || item.date || item.published || item.modified || item.created_at || item.createdAt || item.pubDate || item.timestamp || null; const author = Array.isArray(item.authors) ? item.authors.map((a: any) => a.name).join(', ') : item.author?.name || item.author || ''; if (title && link) { articles.push({ title: title.trim(), link: link.trim(), pubDate: parseDate(pubDate) || new Date().toISOString(), author: typeof author === 'string' ? author.trim() : '', feedTitle, }); } } return { feedTitle, articles }; } function getNestedValue(obj: any, path?: string): unknown { if (!obj || !path) return undefined; return path.split('.').reduce((o: any, key) => o?.[key], obj); } export function parseCustomJson(json: any, config: FeedConfig): ParsedFeed { const articles: Article[] = []; const mapping = config.mapping || {}; const path = config.path || ''; let data: any = json; if (path) { for (const key of path.split('.')) { if (data && typeof data === 'object') data = data[key]; } } const items = Array.isArray(data) ? data : []; const titleKey = mapping.title || 'title'; const linkKey = mapping.link || 'link'; const pubDateKey = mapping.pubDate || 'pubDate'; const authorKey = mapping.author || 'author'; const feedTitleKey = mapping.feedTitle || 'feedTitle'; for (const item of items) { const title = getNestedValue(item, titleKey) || ''; const link = getNestedValue(item, linkKey) || getNestedValue(item, 'url') || ''; const pubDate = getNestedValue(item, pubDateKey) || getNestedValue(item, 'publishedAt') || getNestedValue(item, 'createdAt') || getNestedValue(item, 'created_at') || getNestedValue(item, 'date') || getNestedValue(item, 'published') || getNestedValue(item, 'updatedAt') || getNestedValue(item, 'timestamp') || getNestedValue(item, 'datePublished') || getNestedValue(item, 'dateModified') || null; const author = getNestedValue(item, authorKey) || ''; const feedTitle = getNestedValue(item, feedTitleKey) || config.feedTitle || config.url || ''; if (title && link) { articles.push({ title: String(title).trim(), link: String(link).trim(), pubDate: parseDate(pubDate) || new Date().toISOString(), author: String(typeof author === 'object' ? '' : author).trim(), feedTitle: String(feedTitle), }); } } return { feedTitle: config.feedTitle || config.url || '', articles }; } export function isJsonResponse(text: string, contentType: string): boolean { if (contentType?.includes('json')) return true; try { const parsed = JSON.parse(text); return typeof parsed === 'object' && parsed !== null; } catch { return false; } }