const fs = require('fs'); const path = require('path'); const Parser = require('rss-parser'); const LINK_LITE_PATH = path.join(__dirname, '../themes/Ying/static/json/link_lite.json'); const OUTPUT_PATH = path.join(__dirname, '../themes/Ying/static/json/friend_circle_data.json'); const MAX_POSTS_PER_FRIEND = 5; const MAX_TOTAL_POSTS = 100; // Limit total output size const parser = new Parser({ timeout: 5000, headers: { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36' } }); async function fetchFeed(url) { try { const feed = await parser.parseURL(url); return feed; } catch (error) { // console.error(`Failed to fetch ${url}: ${error.message}`); throw error; } } async function getRssUrl(friend) { // If RSS is explicitly provided (4th element in array) if (friend.length >= 4 && friend[3]) { return friend[3]; } // Otherwise, try to guess const baseUrl = friend[1].replace(/\/$/, ''); const guesses = [ `${baseUrl}/atom.xml`, `${baseUrl}/rss.xml`, `${baseUrl}/feed`, `${baseUrl}/feed/`, `${baseUrl}/index.xml` ]; for (const url of guesses) { try { // Simple HEAD request or just try to parse // Since we use rss-parser, just trying to parse is easiest check // But doing this for every friend is slow. // For now, we only use explicit RSS if provided, or simple guess if not. // To speed up, we'll just try the first guess or return null if strict. // But to be helpful, let's try the most common one: atom.xml // ideally we should check HEAD, but let's just return the first guess for now // and let the parser fail if it's wrong. // BETTER STRATEGY: Return a list of candidates and try them. } catch (e) {} } return `${baseUrl}/atom.xml`; // Default guess } async function main() { console.log('Starting Friend Circle generation...'); if (!fs.existsSync(LINK_LITE_PATH)) { console.error('link_lite.json not found!'); process.exit(1); } const data = JSON.parse(fs.readFileSync(LINK_LITE_PATH, 'utf8')); // data.friends is an array of arrays: [name, url, avatar, rss?] let allPosts = []; // Process friends in batches to avoid overwhelming network const BATCH_SIZE = 10; const friends = data.friends; for (let i = 0; i < friends.length; i += BATCH_SIZE) { const batch = friends.slice(i, i + BATCH_SIZE); const promises = batch.map(async (friend) => { const name = friend[0]; const blogUrl = friend[1]; const avatar = friend[2]; let rssUrl = friend.length >= 4 ? friend[3] : null; // Heuristics for RSS URL const candidates = []; if (rssUrl) { candidates.push(rssUrl); } else { const cleanUrl = blogUrl.replace(/\/$/, ''); candidates.push(`${cleanUrl}/atom.xml`); candidates.push(`${cleanUrl}/rss.xml`); candidates.push(`${cleanUrl}/feed`); candidates.push(`${cleanUrl}/feed/`); candidates.push(`${cleanUrl}/feed.xml`); candidates.push(`${cleanUrl}/index.xml`); candidates.push(`${cleanUrl}/rss`); candidates.push(`${cleanUrl}/atom`); } let feed = null; try { // Try all candidates in parallel and take the first success feed = await Promise.any(candidates.map(url => fetchFeed(url))); // console.log(`[${name}] Fetched RSS successfully.`); } catch (err) { // All candidates failed // console.error(`[${name}] All RSS candidates failed:`, err.errors); console.log(`[${name}] No valid RSS found. Candidates: ${candidates.join(', ')}`); return; } // Extract posts const posts = feed.items.slice(0, MAX_POSTS_PER_FRIEND).map(item => { let img = null; // Try to find an image in content // Priority: content:encoded > content > description const content = item['content:encoded'] || item.content || item.description || ''; // Better regex to capture src with single or double quotes const imgMatch = content.match(/]+src=['"]([^'"]+)['"]/i); if (imgMatch) { img = imgMatch[1]; } else if (item.enclosure && item.enclosure.url && item.enclosure.type && item.enclosure.type.startsWith('image')) { img = item.enclosure.url; } // Get a short snippet let snippet = ''; if (content) { // Remove HTML tags snippet = content.replace(/<[^>]+>/g, ''); // Remove excessive whitespace snippet = snippet.replace(/\s+/g, ' ').trim(); // Truncate if (snippet.length > 120) { snippet = snippet.substring(0, 120) + '...'; } } return { title: item.title, link: item.link, date: item.isoDate || item.pubDate, // RSS parser standardizes this author: name, avatar: avatar, blogUrl: blogUrl, image: img, description: snippet }; }); allPosts = allPosts.concat(posts); }); await Promise.all(promises); process.stdout.write(`Processed ${Math.min(i + BATCH_SIZE, friends.length)}/${friends.length} friends...\r`); } console.log('\nSorting and saving...'); // Sort by date desc allPosts.sort((a, b) => { return new Date(b.date) - new Date(a.date); }); // Limit total const finalPosts = allPosts.slice(0, MAX_TOTAL_POSTS); const output = { updated: new Date().toISOString(), posts: finalPosts }; fs.writeFileSync(OUTPUT_PATH, JSON.stringify(output, null, 2)); console.log(`Saved ${finalPosts.length} posts to ${OUTPUT_PATH}`); } main();