Files
blog/scripts/generate_circle_data.js
T

181 lines
6.5 KiB
JavaScript
Raw Normal View History

2026-01-21 23:18:35 +08:00
const fs = require('fs');
const path = require('path');
const Parser = require('rss-parser');
const LINK_LITE_PATH = path.join(__dirname, '../themes/Ying/static/json/link_lite.json');
const OUTPUT_PATH = path.join(__dirname, '../themes/Ying/static/json/friend_circle_data.json');
const MAX_POSTS_PER_FRIEND = 5;
const MAX_TOTAL_POSTS = 100; // Limit total output size
const parser = new Parser({
2026-02-01 13:25:16 +08:00
timeout: 5000,
2026-01-21 23:18:35 +08:00
headers: {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'
}
});
async function fetchFeed(url) {
try {
const feed = await parser.parseURL(url);
return feed;
} catch (error) {
// console.error(`Failed to fetch ${url}: ${error.message}`);
2026-02-01 13:25:16 +08:00
throw error;
2026-01-21 23:18:35 +08:00
}
}
async function getRssUrl(friend) {
// If RSS is explicitly provided (4th element in array)
if (friend.length >= 4 && friend[3]) {
return friend[3];
}
// Otherwise, try to guess
const baseUrl = friend[1].replace(/\/$/, '');
const guesses = [
`${baseUrl}/atom.xml`,
`${baseUrl}/rss.xml`,
`${baseUrl}/feed`,
`${baseUrl}/feed/`,
`${baseUrl}/index.xml`
];
for (const url of guesses) {
try {
// Simple HEAD request or just try to parse
// Since we use rss-parser, just trying to parse is easiest check
// But doing this for every friend is slow.
// For now, we only use explicit RSS if provided, or simple guess if not.
// To speed up, we'll just try the first guess or return null if strict.
// But to be helpful, let's try the most common one: atom.xml
// ideally we should check HEAD, but let's just return the first guess for now
// and let the parser fail if it's wrong.
// BETTER STRATEGY: Return a list of candidates and try them.
} catch (e) {}
}
return `${baseUrl}/atom.xml`; // Default guess
}
async function main() {
console.log('Starting Friend Circle generation...');
if (!fs.existsSync(LINK_LITE_PATH)) {
console.error('link_lite.json not found!');
process.exit(1);
}
const data = JSON.parse(fs.readFileSync(LINK_LITE_PATH, 'utf8'));
// data.friends is an array of arrays: [name, url, avatar, rss?]
let allPosts = [];
// Process friends in batches to avoid overwhelming network
2026-02-01 13:25:16 +08:00
const BATCH_SIZE = 10;
2026-01-21 23:18:35 +08:00
const friends = data.friends;
for (let i = 0; i < friends.length; i += BATCH_SIZE) {
const batch = friends.slice(i, i + BATCH_SIZE);
const promises = batch.map(async (friend) => {
const name = friend[0];
const blogUrl = friend[1];
const avatar = friend[2];
let rssUrl = friend.length >= 4 ? friend[3] : null;
// Heuristics for RSS URL
const candidates = [];
if (rssUrl) {
candidates.push(rssUrl);
} else {
const cleanUrl = blogUrl.replace(/\/$/, '');
candidates.push(`${cleanUrl}/atom.xml`);
candidates.push(`${cleanUrl}/rss.xml`);
candidates.push(`${cleanUrl}/feed`);
candidates.push(`${cleanUrl}/feed/`);
candidates.push(`${cleanUrl}/feed.xml`);
candidates.push(`${cleanUrl}/index.xml`);
candidates.push(`${cleanUrl}/rss`);
candidates.push(`${cleanUrl}/atom`);
}
let feed = null;
2026-02-01 13:25:16 +08:00
try {
// Try all candidates in parallel and take the first success
feed = await Promise.any(candidates.map(url => fetchFeed(url)));
// console.log(`[${name}] Fetched RSS successfully.`);
} catch (err) {
// All candidates failed
2026-02-01 13:47:51 +08:00
// console.error(`[${name}] All RSS candidates failed:`, err.errors);
console.log(`[${name}] No valid RSS found. Candidates: ${candidates.join(', ')}`);
2026-01-21 23:18:35 +08:00
return;
}
// Extract posts
const posts = feed.items.slice(0, MAX_POSTS_PER_FRIEND).map(item => {
let img = null;
// Try to find an image in content
// Priority: content:encoded > content > description
const content = item['content:encoded'] || item.content || item.description || '';
// Better regex to capture src with single or double quotes
const imgMatch = content.match(/<img[^>]+src=['"]([^'"]+)['"]/i);
if (imgMatch) {
img = imgMatch[1];
} else if (item.enclosure && item.enclosure.url && item.enclosure.type && item.enclosure.type.startsWith('image')) {
img = item.enclosure.url;
}
// Get a short snippet
let snippet = '';
if (content) {
// Remove HTML tags
snippet = content.replace(/<[^>]+>/g, '');
// Remove excessive whitespace
snippet = snippet.replace(/\s+/g, ' ').trim();
// Truncate
if (snippet.length > 120) {
snippet = snippet.substring(0, 120) + '...';
}
}
return {
title: item.title,
link: item.link,
date: item.isoDate || item.pubDate, // RSS parser standardizes this
author: name,
avatar: avatar,
blogUrl: blogUrl,
image: img,
description: snippet
};
});
allPosts = allPosts.concat(posts);
});
await Promise.all(promises);
process.stdout.write(`Processed ${Math.min(i + BATCH_SIZE, friends.length)}/${friends.length} friends...\r`);
}
console.log('\nSorting and saving...');
// Sort by date desc
allPosts.sort((a, b) => {
return new Date(b.date) - new Date(a.date);
});
// Limit total
const finalPosts = allPosts.slice(0, MAX_TOTAL_POSTS);
const output = {
updated: new Date().toISOString(),
posts: finalPosts
};
fs.writeFileSync(OUTPUT_PATH, JSON.stringify(output, null, 2));
console.log(`Saved ${finalPosts.length} posts to ${OUTPUT_PATH}`);
}
main();