上一版用「themes/layouts/assets 等 dirs 白名单」判全局,实测被绕过: build 2 的 diff 只有 scripts/purge_list.js(主题改动在上一个 commit), 判成非全局 → 只刷了 10 条固定入口页 → 文章页仍引用已被 --delete 删掉的 旧 page-only.min.1e47477b…(靠 CDN 缓存才 200,缓存一过期即 404)。 改法: - 判定反转:diff 只含 content/** 才算「纯内容改动」(走精细刷新), 其它一律全量刷 public 下所有 .html。 - 拿不到基线时:push 事件按全量(宁多刷不漏刷),定时任务只刷固定入口页。 - 输出一行 stderr 诊断(base/变更数/纯内容/全量/清单条数)便于事后核对。 - .cnb.yml:清单按 50 条/批 + 批间 6s 调用 upx purge(又拍云限制 单次 ≤50、每分钟 ≤600),并逐批打日志;不再一次性提交 1300+ 条。
125 lines
5.1 KiB
JavaScript
125 lines
5.1 KiB
JavaScript
#!/usr/bin/env node
|
||
/**
|
||
* 生成「本次部署需要刷新的 CDN URL 清单」(每行一个 URL,输出到 stdout)
|
||
*
|
||
* 背景(2026-10-05 实测):
|
||
* 又拍云 CDN 对 HTML 是 max-age=691200(8 天),流水线里只刷固定入口页
|
||
* (/ sitemap rss archives posts comment links)。于是**不在白名单里的页面
|
||
* 即使源站已更新,CDN 仍会发最多 8 天的旧副本**:本次 about.html 就是这样
|
||
* 卡在旧副本上(源站 47935B / CDN 41454B)。
|
||
*
|
||
* 策略:固定入口页(保底,逻辑不变)
|
||
* + **非「纯内容改动」时,追加全站所有 HTML**
|
||
* + 纯内容改动(diff 只含 content/**)时,只追加改动内容对应的输出页
|
||
* - 输出页靠「frontmatter 的 url / slug / 目录名 / 文件名」推导,并**校验
|
||
* public/ 下确实存在该文件**才加入,避免刷不存在的 URL(刷了也无害,但脏)。
|
||
* - 拿不到 git 基线(单提交仓库 / 浅克隆)时:push 事件按「全量」处理(宁多刷
|
||
* 不漏刷,代价只是刷新耗时),定时任务只输出固定入口页。
|
||
*
|
||
* 用法:node scripts/purge_list.js [BASE_REF] BASE_REF 默认 HEAD~1
|
||
*/
|
||
const { execSync } = require('child_process');
|
||
const fs = require('fs');
|
||
const path = require('path');
|
||
|
||
const SITE = 'https://usj.cc';
|
||
|
||
// 固定入口页:首页 / 聚合页 / 站点地图(只刷首页会让后面这些一直发旧副本)
|
||
const FIXED = [
|
||
'/',
|
||
'sitemap.xml',
|
||
'rss.xml',
|
||
'archives.html',
|
||
'posts.html',
|
||
'comment.html',
|
||
'links.html',
|
||
'about.html',
|
||
'circles.html',
|
||
'tiaozhuan.html',
|
||
];
|
||
|
||
function sh(cmd) {
|
||
try {
|
||
return execSync(cmd, { stdio: ['ignore', 'pipe', 'ignore'], encoding: 'utf8' });
|
||
} catch {
|
||
return '';
|
||
}
|
||
}
|
||
|
||
// ★ 必须用 -z:git 默认会把非 ASCII 路径写成 "\344\270\255..." 的带引号转义形式,
|
||
// 那种字符串在 fs.existsSync 上永远查不到,动态部分会静默失效。
|
||
const base = process.argv[2] || 'HEAD~1';
|
||
let changed = sh(`git -c core.quotepath=false diff --name-only -z ${base} HEAD`).split('\0').filter(Boolean);
|
||
if (!changed.length) changed = sh('git -c core.quotepath=false diff --name-only -z HEAD').split('\0').filter(Boolean);
|
||
|
||
// 递归列出 public 下所有 .html
|
||
function walkHtml(dir, out = []) {
|
||
let entries;
|
||
try {
|
||
entries = fs.readdirSync(dir, { withFileTypes: true });
|
||
} catch {
|
||
return out;
|
||
}
|
||
for (const e of entries) {
|
||
const p = path.join(dir, e.name);
|
||
if (e.isDirectory()) walkHtml(p, out);
|
||
else if (e.name.endsWith('.html')) out.push(p);
|
||
}
|
||
return out;
|
||
}
|
||
|
||
const urls = new Set(FIXED.map((p) => SITE + '/' + p.replace(/^\//, '')));
|
||
|
||
// ★ 全量刷新判定:**只有「纯内容改动」才走精细刷新,其它一律全量**
|
||
// 为什么必须这么严:模板用 resources.Fingerprint 给 JS/CSS 生成带 hash 的
|
||
// 文件名,而 upx sync 带 --delete(远端不存在的文件会被删)。只要产物的资源
|
||
// 文件名变了,CDN 上还在的旧 HTML(TTL 8 天)就会引用**已被删掉的旧文件**
|
||
// → 评论区/交互直接断链。所以凡是产物可能变的改动,都得让 HTML 重新拉取。
|
||
// 为什么不用「dirs 白名单」:2026-10-05 踩过 —— 主题改动与 purge 规则改动分成
|
||
// 两个 commit 时,后一个 commit 的 diff 里没有主题文件 → 被判成非全局 → 文章页
|
||
// 漏刷。改成「非 content 即全量」后,不再依赖改动被切进哪个 commit。
|
||
const isCron = process.env.CNB_IS_CRONEVENT === 'true';
|
||
const isContentOnly = changed.length > 0 && changed.every((f) => /^content\//.test(f));
|
||
const isGlobal = changed.length > 0 ? !isContentOnly : !isCron;
|
||
|
||
if (isGlobal) {
|
||
for (const p of walkHtml('public')) {
|
||
const rel = p.replace(/^public[\\/]/, '').split(path.sep).join('/');
|
||
if (rel) urls.add(SITE + '/' + rel);
|
||
}
|
||
}
|
||
|
||
// 诊断输出(进构建日志,便于事后核对判定是否正确)
|
||
process.stderr.write(
|
||
`[purge] base=${base} 变更=${changed.length} 条 纯内容=${isContentOnly} ` +
|
||
`全量=${isGlobal}${changed.length ? '' : '(无基线)'} → 清单=${urls.size} 条\n`
|
||
);
|
||
|
||
for (const f of changed) {
|
||
if (!/^content\/.*\.md$/.test(f) || !fs.existsSync(f)) continue;
|
||
const src = fs.readFileSync(f, 'utf8');
|
||
const fm = (src.match(/^---\r?\n([\s\S]*?)\r?\n---/) || [])[1] || '';
|
||
const grab = (k) => {
|
||
const m = fm.match(new RegExp('^' + k + ':[ \\t]*["\']?([^"\'\\s]+)', 'm'));
|
||
return m ? m[1] : '';
|
||
};
|
||
|
||
const cands = [];
|
||
const rawUrl = grab('url');
|
||
if (rawUrl) cands.push(rawUrl.replace(/^\/+|\/+$/g, '') + '.html');
|
||
const slug = grab('slug');
|
||
if (slug) cands.push(slug + '.html');
|
||
const name = path.basename(f, '.md');
|
||
const parent = path.basename(path.dirname(f));
|
||
if (name === 'index') cands.push(parent + '.html'); // 目录型内容(如 content/posts/xxx/index.md)
|
||
else cands.push(name + '.html');
|
||
|
||
for (const c of cands) {
|
||
const rel = c.replace(/^\/+/, '');
|
||
if (!rel || rel.includes('..')) continue;
|
||
if (fs.existsSync(path.join('public', rel))) urls.add(SITE + '/' + rel);
|
||
}
|
||
}
|
||
|
||
process.stdout.write([...urls].join('\n') + '\n');
|