#!/usr/bin/env node /** * 生成「本次部署需要刷新的 CDN URL 清单」(每行一个 URL,输出到 stdout) * * 背景(2026-10-05 实测): * 又拍云 CDN 对 HTML 是 max-age=691200(8 天),流水线里只刷固定入口页 * (/ sitemap rss archives posts comment links)。于是**不在白名单里的页面 * 即使源站已更新,CDN 仍会发最多 8 天的旧副本**:本次 about.html 就是这样 * 卡在旧副本上(源站 47935B / CDN 41454B)。 * * 策略:固定入口页(保底,逻辑不变) * + **非「纯内容改动」时,追加全站所有 HTML** * + 纯内容改动(diff 只含 content/**)时,只追加改动内容对应的输出页 * - 输出页靠「frontmatter 的 url / slug / 目录名 / 文件名」推导,并**校验 * public/ 下确实存在该文件**才加入,避免刷不存在的 URL(刷了也无害,但脏)。 * - 拿不到 git 基线(单提交仓库 / 浅克隆)时:push 事件按「全量」处理(宁多刷 * 不漏刷,代价只是刷新耗时),定时任务只输出固定入口页。 * * 用法:node scripts/purge_list.js [BASE_REF] BASE_REF 默认 HEAD~1 */ const { execSync } = require('child_process'); const fs = require('fs'); const path = require('path'); const SITE = 'https://usj.cc'; // 固定入口页:首页 / 聚合页 / 站点地图(只刷首页会让后面这些一直发旧副本) const FIXED = [ '/', 'sitemap.xml', 'rss.xml', 'archives.html', 'posts.html', 'comment.html', 'links.html', 'about.html', 'circles.html', 'tiaozhuan.html', ]; function sh(cmd) { try { return execSync(cmd, { stdio: ['ignore', 'pipe', 'ignore'], encoding: 'utf8' }); } catch { return ''; } } // ★ 必须用 -z:git 默认会把非 ASCII 路径写成 "\344\270\255..." 的带引号转义形式, // 那种字符串在 fs.existsSync 上永远查不到,动态部分会静默失效。 const base = process.argv[2] || 'HEAD~1'; let changed = sh(`git -c core.quotepath=false diff --name-only -z ${base} HEAD`).split('\0').filter(Boolean); if (!changed.length) changed = sh('git -c core.quotepath=false diff --name-only -z HEAD').split('\0').filter(Boolean); // 递归列出 public 下所有 .html function walkHtml(dir, out = []) { let entries; try { entries = fs.readdirSync(dir, { withFileTypes: true }); } catch { return out; } for (const e of entries) { const p = path.join(dir, e.name); if (e.isDirectory()) walkHtml(p, out); else if (e.name.endsWith('.html')) out.push(p); } return out; } const urls = new Set(FIXED.map((p) => SITE + '/' + p.replace(/^\//, ''))); // ★ 全量刷新判定:**只有「纯内容改动」才走精细刷新,其它一律全量** // 为什么必须这么严:模板用 resources.Fingerprint 给 JS/CSS 生成带 hash 的 // 文件名,而 upx sync 带 --delete(远端不存在的文件会被删)。只要产物的资源 // 文件名变了,CDN 上还在的旧 HTML(TTL 8 天)就会引用**已被删掉的旧文件** // → 评论区/交互直接断链。所以凡是产物可能变的改动,都得让 HTML 重新拉取。 // 为什么不用「dirs 白名单」:2026-10-05 踩过 —— 主题改动与 purge 规则改动分成 // 两个 commit 时,后一个 commit 的 diff 里没有主题文件 → 被判成非全局 → 文章页 // 漏刷。改成「非 content 即全量」后,不再依赖改动被切进哪个 commit。 const isCron = process.env.CNB_IS_CRONEVENT === 'true'; const isContentOnly = changed.length > 0 && changed.every((f) => /^content\//.test(f)); const isGlobal = changed.length > 0 ? !isContentOnly : !isCron; if (isGlobal) { for (const p of walkHtml('public')) { const rel = p.replace(/^public[\\/]/, '').split(path.sep).join('/'); if (rel) urls.add(SITE + '/' + rel); } } // 诊断输出(进构建日志,便于事后核对判定是否正确) process.stderr.write( `[purge] base=${base} 变更=${changed.length} 条 纯内容=${isContentOnly} ` + `全量=${isGlobal}${changed.length ? '' : '(无基线)'} → 清单=${urls.size} 条\n` ); for (const f of changed) { if (!/^content\/.*\.md$/.test(f) || !fs.existsSync(f)) continue; const src = fs.readFileSync(f, 'utf8'); const fm = (src.match(/^---\r?\n([\s\S]*?)\r?\n---/) || [])[1] || ''; const grab = (k) => { const m = fm.match(new RegExp('^' + k + ':[ \\t]*["\']?([^"\'\\s]+)', 'm')); return m ? m[1] : ''; }; const cands = []; const rawUrl = grab('url'); if (rawUrl) cands.push(rawUrl.replace(/^\/+|\/+$/g, '') + '.html'); const slug = grab('slug'); if (slug) cands.push(slug + '.html'); const name = path.basename(f, '.md'); const parent = path.basename(path.dirname(f)); if (name === 'index') cands.push(parent + '.html'); // 目录型内容(如 content/posts/xxx/index.md) else cands.push(name + '.html'); for (const c of cands) { const rel = c.replace(/^\/+/, ''); if (!rel || rel.includes('..')) continue; if (fs.existsSync(path.join('public', rel))) urls.add(SITE + '/' + rel); } } process.stdout.write([...urls].join('\n') + '\n');