2026-10-05 12:18:53 +08:00
|
|
|
|
#!/usr/bin/env node
|
|
|
|
|
|
/**
|
|
|
|
|
|
* 生成「本次部署需要刷新的 CDN URL 清单」(每行一个 URL,输出到 stdout)
|
|
|
|
|
|
*
|
|
|
|
|
|
* 背景(2026-10-05 实测):
|
|
|
|
|
|
* 又拍云 CDN 对 HTML 是 max-age=691200(8 天),流水线里只刷固定入口页
|
|
|
|
|
|
* (/ sitemap rss archives posts comment links)。于是**不在白名单里的页面
|
|
|
|
|
|
* 即使源站已更新,CDN 仍会发最多 8 天的旧副本**:本次 about.html 就是这样
|
|
|
|
|
|
* 卡在旧副本上(源站 47935B / CDN 41454B)。
|
|
|
|
|
|
*
|
2026-10-05 17:39:59 +08:00
|
|
|
|
* 策略:固定入口页(保底,逻辑不变)
|
2026-10-05 17:57:51 +08:00
|
|
|
|
* + **非「纯内容改动」时,追加全站所有 HTML**
|
|
|
|
|
|
* + 纯内容改动(diff 只含 content/**)时,只追加改动内容对应的输出页
|
2026-10-05 12:18:53 +08:00
|
|
|
|
* - 输出页靠「frontmatter 的 url / slug / 目录名 / 文件名」推导,并**校验
|
|
|
|
|
|
* public/ 下确实存在该文件**才加入,避免刷不存在的 URL(刷了也无害,但脏)。
|
2026-10-05 17:57:51 +08:00
|
|
|
|
* - 拿不到 git 基线(单提交仓库 / 浅克隆)时:push 事件按「全量」处理(宁多刷
|
|
|
|
|
|
* 不漏刷,代价只是刷新耗时),定时任务只输出固定入口页。
|
2026-10-05 12:18:53 +08:00
|
|
|
|
*
|
|
|
|
|
|
* 用法:node scripts/purge_list.js [BASE_REF] BASE_REF 默认 HEAD~1
|
|
|
|
|
|
*/
|
|
|
|
|
|
const { execSync } = require('child_process');
|
|
|
|
|
|
const fs = require('fs');
|
|
|
|
|
|
const path = require('path');
|
|
|
|
|
|
|
|
|
|
|
|
const SITE = 'https://usj.cc';
|
|
|
|
|
|
|
|
|
|
|
|
// 固定入口页:首页 / 聚合页 / 站点地图(只刷首页会让后面这些一直发旧副本)
|
|
|
|
|
|
const FIXED = [
|
|
|
|
|
|
'/',
|
|
|
|
|
|
'sitemap.xml',
|
|
|
|
|
|
'rss.xml',
|
|
|
|
|
|
'archives.html',
|
|
|
|
|
|
'posts.html',
|
|
|
|
|
|
'comment.html',
|
|
|
|
|
|
'links.html',
|
|
|
|
|
|
'about.html',
|
|
|
|
|
|
'circles.html',
|
|
|
|
|
|
'tiaozhuan.html',
|
|
|
|
|
|
];
|
|
|
|
|
|
|
|
|
|
|
|
function sh(cmd) {
|
|
|
|
|
|
try {
|
|
|
|
|
|
return execSync(cmd, { stdio: ['ignore', 'pipe', 'ignore'], encoding: 'utf8' });
|
|
|
|
|
|
} catch {
|
|
|
|
|
|
return '';
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
// ★ 必须用 -z:git 默认会把非 ASCII 路径写成 "\344\270\255..." 的带引号转义形式,
|
|
|
|
|
|
// 那种字符串在 fs.existsSync 上永远查不到,动态部分会静默失效。
|
|
|
|
|
|
const base = process.argv[2] || 'HEAD~1';
|
|
|
|
|
|
let changed = sh(`git -c core.quotepath=false diff --name-only -z ${base} HEAD`).split('\0').filter(Boolean);
|
|
|
|
|
|
if (!changed.length) changed = sh('git -c core.quotepath=false diff --name-only -z HEAD').split('\0').filter(Boolean);
|
|
|
|
|
|
|
2026-10-05 17:39:59 +08:00
|
|
|
|
// 递归列出 public 下所有 .html
|
|
|
|
|
|
function walkHtml(dir, out = []) {
|
|
|
|
|
|
let entries;
|
|
|
|
|
|
try {
|
|
|
|
|
|
entries = fs.readdirSync(dir, { withFileTypes: true });
|
|
|
|
|
|
} catch {
|
|
|
|
|
|
return out;
|
|
|
|
|
|
}
|
|
|
|
|
|
for (const e of entries) {
|
|
|
|
|
|
const p = path.join(dir, e.name);
|
|
|
|
|
|
if (e.isDirectory()) walkHtml(p, out);
|
|
|
|
|
|
else if (e.name.endsWith('.html')) out.push(p);
|
|
|
|
|
|
}
|
|
|
|
|
|
return out;
|
|
|
|
|
|
}
|
|
|
|
|
|
|
2026-10-05 12:18:53 +08:00
|
|
|
|
const urls = new Set(FIXED.map((p) => SITE + '/' + p.replace(/^\//, '')));
|
|
|
|
|
|
|
2026-10-05 17:57:51 +08:00
|
|
|
|
// ★ 全量刷新判定:**只有「纯内容改动」才走精细刷新,其它一律全量**
|
|
|
|
|
|
// 为什么必须这么严:模板用 resources.Fingerprint 给 JS/CSS 生成带 hash 的
|
|
|
|
|
|
// 文件名,而 upx sync 带 --delete(远端不存在的文件会被删)。只要产物的资源
|
|
|
|
|
|
// 文件名变了,CDN 上还在的旧 HTML(TTL 8 天)就会引用**已被删掉的旧文件**
|
|
|
|
|
|
// → 评论区/交互直接断链。所以凡是产物可能变的改动,都得让 HTML 重新拉取。
|
|
|
|
|
|
// 为什么不用「dirs 白名单」:2026-10-05 踩过 —— 主题改动与 purge 规则改动分成
|
|
|
|
|
|
// 两个 commit 时,后一个 commit 的 diff 里没有主题文件 → 被判成非全局 → 文章页
|
|
|
|
|
|
// 漏刷。改成「非 content 即全量」后,不再依赖改动被切进哪个 commit。
|
|
|
|
|
|
const isCron = process.env.CNB_IS_CRONEVENT === 'true';
|
|
|
|
|
|
const isContentOnly = changed.length > 0 && changed.every((f) => /^content\//.test(f));
|
|
|
|
|
|
const isGlobal = changed.length > 0 ? !isContentOnly : !isCron;
|
|
|
|
|
|
|
2026-10-05 17:39:59 +08:00
|
|
|
|
if (isGlobal) {
|
|
|
|
|
|
for (const p of walkHtml('public')) {
|
|
|
|
|
|
const rel = p.replace(/^public[\\/]/, '').split(path.sep).join('/');
|
|
|
|
|
|
if (rel) urls.add(SITE + '/' + rel);
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
|
2026-10-05 17:57:51 +08:00
|
|
|
|
// 诊断输出(进构建日志,便于事后核对判定是否正确)
|
|
|
|
|
|
process.stderr.write(
|
|
|
|
|
|
`[purge] base=${base} 变更=${changed.length} 条 纯内容=${isContentOnly} ` +
|
|
|
|
|
|
`全量=${isGlobal}${changed.length ? '' : '(无基线)'} → 清单=${urls.size} 条\n`
|
|
|
|
|
|
);
|
|
|
|
|
|
|
2026-10-05 12:18:53 +08:00
|
|
|
|
for (const f of changed) {
|
|
|
|
|
|
if (!/^content\/.*\.md$/.test(f) || !fs.existsSync(f)) continue;
|
|
|
|
|
|
const src = fs.readFileSync(f, 'utf8');
|
|
|
|
|
|
const fm = (src.match(/^---\r?\n([\s\S]*?)\r?\n---/) || [])[1] || '';
|
|
|
|
|
|
const grab = (k) => {
|
|
|
|
|
|
const m = fm.match(new RegExp('^' + k + ':[ \\t]*["\']?([^"\'\\s]+)', 'm'));
|
|
|
|
|
|
return m ? m[1] : '';
|
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
|
|
const cands = [];
|
|
|
|
|
|
const rawUrl = grab('url');
|
|
|
|
|
|
if (rawUrl) cands.push(rawUrl.replace(/^\/+|\/+$/g, '') + '.html');
|
|
|
|
|
|
const slug = grab('slug');
|
|
|
|
|
|
if (slug) cands.push(slug + '.html');
|
|
|
|
|
|
const name = path.basename(f, '.md');
|
|
|
|
|
|
const parent = path.basename(path.dirname(f));
|
|
|
|
|
|
if (name === 'index') cands.push(parent + '.html'); // 目录型内容(如 content/posts/xxx/index.md)
|
|
|
|
|
|
else cands.push(name + '.html');
|
|
|
|
|
|
|
|
|
|
|
|
for (const c of cands) {
|
|
|
|
|
|
const rel = c.replace(/^\/+/, '');
|
|
|
|
|
|
if (!rel || rel.includes('..')) continue;
|
|
|
|
|
|
if (fs.existsSync(path.join('public', rel))) urls.add(SITE + '/' + rel);
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
process.stdout.write([...urls].join('\n') + '\n');
|