Files
blog/scripts/purge_list.js
T
zqlit 5a50df6e23 fix(ci): 主题/layouts/assets 等全局改动时,purge 清单追加全站 HTML
问题:模板用 resources.Fingerprint 生成带 hash 的 JS/CSS 文件名,而 upx sync
带 --delete(远端不存在的文件会被删)。只改主题、不动 content/*.md 时,
purge_list 只输出固定入口页 + content 页 → 文章页的 HTML 仍留在 CDN(TTL 8 天),
里面引用的是**已被删掉的旧指纹 JS** → 评论区/交互断链(实测文章页仍引用
page-only.min.1e47477b…,该文件本轮已不在源站,只是还在 CDN 缓存里)。

修法:diff 命中 themes/ layouts/ assets/ static/ data/ archetypes/ 或
hugo.toml/config.* 时,walk public/ 把所有 .html 加入刷新清单
(本机验证 1336/1336 覆盖)。
2026-10-05 17:39:59 +08:00

115 lines
4.5 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env node
/**
* 生成「本次部署需要刷新的 CDN URL 清单」(每行一个 URL,输出到 stdout)
*
* 背景(2026-10-05 实测):
* 又拍云 CDN 对 HTML 是 max-age=691200(8 天),流水线里只刷固定入口页
* (/ sitemap rss archives posts comment links)。于是**不在白名单里的页面
* 即使源站已更新,CDN 仍会发最多 8 天的旧副本**:本次 about.html 就是这样
* 卡在旧副本上(源站 47935B / CDN 41454B)。
*
* 策略:固定入口页(保底,逻辑不变)
* + 本次改动过的 content/*.md 对应的输出页
* + **全局性改动(themes/layouts/assets/static/data、hugo/config)时,追加全站所有 HTML**。
* - 输出页靠「frontmatter 的 url / slug / 目录名 / 文件名」推导,并**校验
* public/ 下确实存在该文件**才加入,避免刷不存在的 URL(刷了也无害,但脏)。
* - 拿不到 git 基线(单提交仓库 / 浅克隆)时自动降级:只输出固定入口页。
*
* 用法:node scripts/purge_list.js [BASE_REF] BASE_REF 默认 HEAD~1
*/
const { execSync } = require('child_process');
const fs = require('fs');
const path = require('path');
const SITE = 'https://usj.cc';
// 固定入口页:首页 / 聚合页 / 站点地图(只刷首页会让后面这些一直发旧副本)
const FIXED = [
'/',
'sitemap.xml',
'rss.xml',
'archives.html',
'posts.html',
'comment.html',
'links.html',
'about.html',
'circles.html',
'tiaozhuan.html',
];
function sh(cmd) {
try {
return execSync(cmd, { stdio: ['ignore', 'pipe', 'ignore'], encoding: 'utf8' });
} catch {
return '';
}
}
// ★ 必须用 -z:git 默认会把非 ASCII 路径写成 "\344\270\255..." 的带引号转义形式,
// 那种字符串在 fs.existsSync 上永远查不到,动态部分会静默失效。
const base = process.argv[2] || 'HEAD~1';
let changed = sh(`git -c core.quotepath=false diff --name-only -z ${base} HEAD`).split('\0').filter(Boolean);
if (!changed.length) changed = sh('git -c core.quotepath=false diff --name-only -z HEAD').split('\0').filter(Boolean);
// 递归列出 public 下所有 .html
function walkHtml(dir, out = []) {
let entries;
try {
entries = fs.readdirSync(dir, { withFileTypes: true });
} catch {
return out;
}
for (const e of entries) {
const p = path.join(dir, e.name);
if (e.isDirectory()) walkHtml(p, out);
else if (e.name.endsWith('.html')) out.push(p);
}
return out;
}
const urls = new Set(FIXED.map((p) => SITE + '/' + p.replace(/^\//, '')));
// ★ 全局性改动 → 刷新**全部** HTML(2026-10-05 新增)
// 为什么必须这么做:模板用 resources.Fingerprint 给 JS/CSS 生成带 hash 的文件名,
// 而 upx sync 带 --delete(远端不存在的文件会被删)。主题一改,page-only / core 等
// 包的文件名就变了 → 老指纹文件被删。文章页的 HTML 若还留在 CDN(TTL 8 天),
// 它引用的就是**已被删掉的旧 JS** → 评论区/交互直接断链。
// 所以凡是会改变「全站 HTML 引用资源」的改动,都必须让所有 HTML 重新拉取。
const GLOBAL_DIRS = /^(themes|layouts|assets|static|data|archetypes)\//;
const GLOBAL_FILES = /^(hugo|config)\.(toml|yaml|yml|json)$/;
const isGlobal = changed.some((f) => GLOBAL_DIRS.test(f) || GLOBAL_FILES.test(f));
if (isGlobal) {
for (const p of walkHtml('public')) {
const rel = p.replace(/^public[\\/]/, '').split(path.sep).join('/');
if (rel) urls.add(SITE + '/' + rel);
}
}
for (const f of changed) {
if (!/^content\/.*\.md$/.test(f) || !fs.existsSync(f)) continue;
const src = fs.readFileSync(f, 'utf8');
const fm = (src.match(/^---\r?\n([\s\S]*?)\r?\n---/) || [])[1] || '';
const grab = (k) => {
const m = fm.match(new RegExp('^' + k + ':[ \\t]*["\']?([^"\'\\s]+)', 'm'));
return m ? m[1] : '';
};
const cands = [];
const rawUrl = grab('url');
if (rawUrl) cands.push(rawUrl.replace(/^\/+|\/+$/g, '') + '.html');
const slug = grab('slug');
if (slug) cands.push(slug + '.html');
const name = path.basename(f, '.md');
const parent = path.basename(path.dirname(f));
if (name === 'index') cands.push(parent + '.html'); // 目录型内容(如 content/posts/xxx/index.md)
else cands.push(name + '.html');
for (const c of cands) {
const rel = c.replace(/^\/+/, '');
if (!rel || rel.includes('..')) continue;
if (fs.existsSync(path.join('public', rel))) urls.add(SITE + '/' + rel);
}
}
process.stdout.write([...urls].join('\n') + '\n');