Files
blog/blog-admin/src/lib/rss/parse.ts
T

208 lines
6.3 KiB
TypeScript
Raw Normal View History

// RSS / Atom / JSON Feed / 自定义 JSON 解析
// 迁自 check-feeds.js 的 parseFeedXml / parseJsonFeed / parseCustomJson 等。
export interface Article {
title: string;
link: string;
pubDate: string;
author: string;
feedTitle: string;
}
export interface FeedConfig {
url: string;
feedTitle?: string;
format?: string;
path?: string;
proxy?: boolean;
mapping?: Record<string, string>;
}
export interface ParsedFeed {
feedTitle: string;
articles: Article[];
}
function extractTag(xml: string, tag: string): string {
const regex = new RegExp(`<${tag}[^>]*>([\\s\\S]*?)<\\/${tag}>`, 'i');
const match = xml.match(regex);
return match ? match[1] : '';
}
function decodeXml(str: string): string {
return str
.replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, '$1')
.replace(/&amp;/g, '&')
.replace(/&lt;/g, '<')
.replace(/&gt;/g, '>')
.replace(/&quot;/g, '"')
.replace(/&apos;/g, "'");
}
export function parseDate(val: unknown): string | null {
if (!val) return null;
if (typeof val === 'number') {
const d = new Date(val > 9999999999 ? val : val * 1000);
return isNaN(d.getTime()) ? null : d.toISOString();
}
const str = String(val).trim();
if (!str) return null;
const d = new Date(str);
if (isNaN(d.getTime())) return null;
try {
return d.toISOString();
} catch {
return null;
}
}
export function parseFeedXml(xml: string, feedUrl: string): ParsedFeed {
const articles: Article[] = [];
let feedTitle = '';
const titleMatch = xml.match(/<title[^>]*>([\s\S]*?)<\/title>/);
if (titleMatch) feedTitle = decodeXml(titleMatch[1]).trim();
const itemRegex = /<item[\s>]?>([\s\S]*?)<\/item>/gi;
const entryRegex = /<entry[\s>]?>([\s\S]*?)<\/entry>/gi;
const parseItem = (itemXml: string) => {
const title = extractTag(itemXml, 'title');
let link = '';
const rssLink = extractTag(itemXml, 'link');
if (rssLink) link = rssLink;
const atomLinkMatch = itemXml.match(/<link[^>]+href=["']([^"']+)["'][^>]*>/i);
if (atomLinkMatch && atomLinkMatch[1]) link = atomLinkMatch[1];
const altLinkMatch = itemXml.match(
/<link[^>]+rel=["']alternate["'][^>]+href=["']([^"']+)["']/i,
);
if (altLinkMatch) link = altLinkMatch[1];
const pubDate =
extractTag(itemXml, 'pubDate') ||
extractTag(itemXml, 'published') ||
extractTag(itemXml, 'updated') ||
extractTag(itemXml, 'dc:date') ||
extractTag(itemXml, 'lastBuildDate') ||
extractTag(itemXml, 'date') ||
extractTag(itemXml, 'modified') ||
extractTag(itemXml, 'created');
const author = extractTag(itemXml, 'dc:creator') || extractTag(itemXml, 'author') || '';
const authorName = author.replace(/<name>([\s\S]*?)<\/name>/gi, '$1').trim();
if (title && link) {
articles.push({
title: decodeXml(title).trim(),
link: link.trim(),
pubDate: parseDate(pubDate) || new Date().toISOString(),
author: decodeXml(authorName).trim(),
feedTitle: feedTitle || feedUrl,
});
}
};
let match: RegExpExecArray | null;
while ((match = itemRegex.exec(xml)) !== null) parseItem(match[1]);
while ((match = entryRegex.exec(xml)) !== null) parseItem(match[1]);
return { feedTitle, articles };
}
export function parseJsonFeed(json: any, feedUrl: string): ParsedFeed {
const articles: Article[] = [];
const feedTitle = json.title || feedUrl;
const items = json.items || [];
for (const item of items) {
const title = item.title || '';
const link = item.url || item.id || '';
const pubDate =
item.date_published || item.date_modified || item.date || item.published || item.modified || item.created_at || item.createdAt || item.pubDate || item.timestamp || null;
const author = Array.isArray(item.authors)
? item.authors.map((a: any) => a.name).join(', ')
: item.author?.name || item.author || '';
if (title && link) {
articles.push({
title: title.trim(),
link: link.trim(),
pubDate: parseDate(pubDate) || new Date().toISOString(),
author: typeof author === 'string' ? author.trim() : '',
feedTitle,
});
}
}
return { feedTitle, articles };
}
function getNestedValue(obj: any, path?: string): unknown {
if (!obj || !path) return undefined;
return path.split('.').reduce((o: any, key) => o?.[key], obj);
}
export function parseCustomJson(json: any, config: FeedConfig): ParsedFeed {
const articles: Article[] = [];
const mapping = config.mapping || {};
const path = config.path || '';
let data: any = json;
if (path) {
for (const key of path.split('.')) {
if (data && typeof data === 'object') data = data[key];
}
}
const items = Array.isArray(data) ? data : [];
const titleKey = mapping.title || 'title';
const linkKey = mapping.link || 'link';
const pubDateKey = mapping.pubDate || 'pubDate';
const authorKey = mapping.author || 'author';
const feedTitleKey = mapping.feedTitle || 'feedTitle';
for (const item of items) {
const title = getNestedValue(item, titleKey) || '';
const link = getNestedValue(item, linkKey) || getNestedValue(item, 'url') || '';
const pubDate =
getNestedValue(item, pubDateKey) ||
getNestedValue(item, 'publishedAt') ||
getNestedValue(item, 'createdAt') ||
getNestedValue(item, 'created_at') ||
getNestedValue(item, 'date') ||
getNestedValue(item, 'published') ||
getNestedValue(item, 'updatedAt') ||
getNestedValue(item, 'timestamp') ||
getNestedValue(item, 'datePublished') ||
getNestedValue(item, 'dateModified') ||
null;
const author = getNestedValue(item, authorKey) || '';
const feedTitle = getNestedValue(item, feedTitleKey) || config.feedTitle || config.url || '';
if (title && link) {
articles.push({
title: String(title).trim(),
link: String(link).trim(),
pubDate: parseDate(pubDate) || new Date().toISOString(),
author: String(typeof author === 'object' ? '' : author).trim(),
feedTitle: String(feedTitle),
});
}
}
return { feedTitle: config.feedTitle || config.url || '', articles };
}
export function isJsonResponse(text: string, contentType: string): boolean {
if (contentType?.includes('json')) return true;
try {
const parsed = JSON.parse(text);
return typeof parsed === 'object' && parsed !== null;
} catch {
return false;
}
}