Files
blog/scripts/purge_list.js
T
zqlit c1f9a0c6f0 fix(ci): 全量刷新判定改为「非纯内容改动即全量」+ 刷新分批(又拍云 50/次)
上一版用「themes/layouts/assets 等 dirs 白名单」判全局,实测被绕过:
build 2 的 diff 只有 scripts/purge_list.js(主题改动在上一个 commit),
判成非全局 → 只刷了 10 条固定入口页 → 文章页仍引用已被 --delete 删掉的
旧 page-only.min.1e47477b…(靠 CDN 缓存才 200,缓存一过期即 404)。

改法:
- 判定反转:diff 只含 content/** 才算「纯内容改动」(走精细刷新),
  其它一律全量刷 public 下所有 .html。
- 拿不到基线时:push 事件按全量(宁多刷不漏刷),定时任务只刷固定入口页。
- 输出一行 stderr 诊断(base/变更数/纯内容/全量/清单条数)便于事后核对。
- .cnb.yml:清单按 50 条/批 + 批间 6s 调用 upx purge(又拍云限制
  单次 ≤50、每分钟 ≤600),并逐批打日志;不再一次性提交 1300+ 条。
2026-10-05 17:57:51 +08:00

125 lines
5.1 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env node
/**
* 生成「本次部署需要刷新的 CDN URL 清单」(每行一个 URL,输出到 stdout)
*
* 背景(2026-10-05 实测):
* 又拍云 CDN 对 HTML 是 max-age=691200(8 天),流水线里只刷固定入口页
* (/ sitemap rss archives posts comment links)。于是**不在白名单里的页面
* 即使源站已更新,CDN 仍会发最多 8 天的旧副本**:本次 about.html 就是这样
* 卡在旧副本上(源站 47935B / CDN 41454B)。
*
* 策略:固定入口页(保底,逻辑不变)
* + **非「纯内容改动」时,追加全站所有 HTML**
* + 纯内容改动(diff 只含 content/**)时,只追加改动内容对应的输出页
* - 输出页靠「frontmatter 的 url / slug / 目录名 / 文件名」推导,并**校验
* public/ 下确实存在该文件**才加入,避免刷不存在的 URL(刷了也无害,但脏)。
* - 拿不到 git 基线(单提交仓库 / 浅克隆)时:push 事件按「全量」处理(宁多刷
* 不漏刷,代价只是刷新耗时),定时任务只输出固定入口页。
*
* 用法:node scripts/purge_list.js [BASE_REF] BASE_REF 默认 HEAD~1
*/
const { execSync } = require('child_process');
const fs = require('fs');
const path = require('path');
const SITE = 'https://usj.cc';
// 固定入口页:首页 / 聚合页 / 站点地图(只刷首页会让后面这些一直发旧副本)
const FIXED = [
'/',
'sitemap.xml',
'rss.xml',
'archives.html',
'posts.html',
'comment.html',
'links.html',
'about.html',
'circles.html',
'tiaozhuan.html',
];
function sh(cmd) {
try {
return execSync(cmd, { stdio: ['ignore', 'pipe', 'ignore'], encoding: 'utf8' });
} catch {
return '';
}
}
// ★ 必须用 -z:git 默认会把非 ASCII 路径写成 "\344\270\255..." 的带引号转义形式,
// 那种字符串在 fs.existsSync 上永远查不到,动态部分会静默失效。
const base = process.argv[2] || 'HEAD~1';
let changed = sh(`git -c core.quotepath=false diff --name-only -z ${base} HEAD`).split('\0').filter(Boolean);
if (!changed.length) changed = sh('git -c core.quotepath=false diff --name-only -z HEAD').split('\0').filter(Boolean);
// 递归列出 public 下所有 .html
function walkHtml(dir, out = []) {
let entries;
try {
entries = fs.readdirSync(dir, { withFileTypes: true });
} catch {
return out;
}
for (const e of entries) {
const p = path.join(dir, e.name);
if (e.isDirectory()) walkHtml(p, out);
else if (e.name.endsWith('.html')) out.push(p);
}
return out;
}
const urls = new Set(FIXED.map((p) => SITE + '/' + p.replace(/^\//, '')));
// ★ 全量刷新判定:**只有「纯内容改动」才走精细刷新,其它一律全量**
// 为什么必须这么严:模板用 resources.Fingerprint 给 JS/CSS 生成带 hash 的
// 文件名,而 upx sync 带 --delete(远端不存在的文件会被删)。只要产物的资源
// 文件名变了,CDN 上还在的旧 HTML(TTL 8 天)就会引用**已被删掉的旧文件**
// → 评论区/交互直接断链。所以凡是产物可能变的改动,都得让 HTML 重新拉取。
// 为什么不用「dirs 白名单」:2026-10-05 踩过 —— 主题改动与 purge 规则改动分成
// 两个 commit 时,后一个 commit 的 diff 里没有主题文件 → 被判成非全局 → 文章页
// 漏刷。改成「非 content 即全量」后,不再依赖改动被切进哪个 commit。
const isCron = process.env.CNB_IS_CRONEVENT === 'true';
const isContentOnly = changed.length > 0 && changed.every((f) => /^content\//.test(f));
const isGlobal = changed.length > 0 ? !isContentOnly : !isCron;
if (isGlobal) {
for (const p of walkHtml('public')) {
const rel = p.replace(/^public[\\/]/, '').split(path.sep).join('/');
if (rel) urls.add(SITE + '/' + rel);
}
}
// 诊断输出(进构建日志,便于事后核对判定是否正确)
process.stderr.write(
`[purge] base=${base} 变更=${changed.length} 条 纯内容=${isContentOnly} ` +
`全量=${isGlobal}${changed.length ? '' : '(无基线)'} → 清单=${urls.size} 条\n`
);
for (const f of changed) {
if (!/^content\/.*\.md$/.test(f) || !fs.existsSync(f)) continue;
const src = fs.readFileSync(f, 'utf8');
const fm = (src.match(/^---\r?\n([\s\S]*?)\r?\n---/) || [])[1] || '';
const grab = (k) => {
const m = fm.match(new RegExp('^' + k + ':[ \\t]*["\']?([^"\'\\s]+)', 'm'));
return m ? m[1] : '';
};
const cands = [];
const rawUrl = grab('url');
if (rawUrl) cands.push(rawUrl.replace(/^\/+|\/+$/g, '') + '.html');
const slug = grab('slug');
if (slug) cands.push(slug + '.html');
const name = path.basename(f, '.md');
const parent = path.basename(path.dirname(f));
if (name === 'index') cands.push(parent + '.html'); // 目录型内容(如 content/posts/xxx/index.md)
else cands.push(name + '.html');
for (const c of cands) {
const rel = c.replace(/^\/+/, '');
if (!rel || rel.includes('..')) continue;
if (fs.existsSync(path.join('public', rel))) urls.add(SITE + '/' + rel);
}
}
process.stdout.write([...urls].join('\n') + '\n');