332 lines
10 KiB
JavaScript
332 lines
10 KiB
JavaScript
/**
|
||||
|
|
* 极简 YAML front matter 子集:解析 + 序列化(零依赖)。
|
|||
|
|
*
|
|||
|
|
* 为什么不用 gray-matter / js-yaml:
|
|||
|
|
* 这个后端只服务「编辑自己的 Hugo 文章」这一件事,语料是固定的 100 多篇
|
|||
|
|
* Markdown。为此拖进一棵依赖树不值得,镜像也能小一圈。
|
|||
|
|
*
|
|||
|
|
* 覆盖的形态(Hugo front matter 实际会用到的全部):
|
|||
|
|
* key: 标量 key: '单引号' key: "双引号"
|
|||
|
|
* key: 123 / true key: [a, b] key: []
|
|||
|
|
* key: (空值 → null)
|
|||
|
|
* - a
|
|||
|
|
* - b (列表,缩进两格)
|
|||
|
|
* key: | / |- / |+ / > / >- / >+ (块标量)
|
|||
|
|
* key:
|
|||
|
|
* sub: 1 (一层嵌套 map,Hugo 的 params 会用到)
|
|||
|
|
*
|
|||
|
|
* 遇到覆盖不到的结构一律 **抛错**,由调用方降级为「原文照存」——
|
|||
|
|
* 宁可少解析,绝不猜错后把用户文章写坏。
|
|||
|
|
*/
|
|||
|
|
|
|||
|
|
const OPEN = /^---[ \t]*\n/;
|
|||
|
|
|
|||
|
|
/**
|
|||
|
|
* 把整篇文本切成 front matter 原文 / 正文 / 原文件换行风格。
|
|||
|
|
*
|
|||
|
|
* ★ 换行符必须原样保留:仓库 `.gitattributes` 是 `* text=auto`,
|
|||
|
|
* 仓库里存的是 LF,但 Windows 工作区检出是 CRLF。如果写回时统一成 LF,
|
|||
|
|
* 在 Windows 侧就会产生「整文件换行变更」的巨型 diff。所以这里记住原风格,
|
|||
|
|
* 交给 joinFrontMatter 还原。
|
|||
|
|
*
|
|||
|
|
* body **保留**它前面的空行,这样 join 出来的结果与原文件逐字节相同。
|
|||
|
|
*/
|
|||
|
|
export function splitFrontMatter(text) {
|
|||
|
|
const eol = text.includes('\r\n') ? '\r\n' : '\n';
|
|||
|
|
const norm = text.replace(/\r\n/g, '\n');
|
|||
|
|
|
|||
|
|
const open = OPEN.exec(norm);
|
|||
|
|
if (!open) return { raw: null, body: norm, eol };
|
|||
|
|
|
|||
|
|
const rest = norm.slice(open[0].length);
|
|||
|
|
const lines = rest.split('\n');
|
|||
|
|
|
|||
|
|
for (let i = 0; i < lines.length; i++) {
|
|||
|
|
if (lines[i].trim() === '---') {
|
|||
|
|
return {
|
|||
|
|
raw: lines.slice(0, i).join('\n'),
|
|||
|
|
body: lines.slice(i + 1).join('\n'),
|
|||
|
|
eol,
|
|||
|
|
};
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
return { raw: null, body: norm, eol };
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
/** 组装回整篇文本(raw 为 null 表示本来就没有 front matter) */
|
|||
|
|
export function joinFrontMatter(raw, body, eol = '\n') {
|
|||
|
|
const s = raw == null ? body : '---\n' + raw.replace(/[ \t]+$/, '') + '\n---\n' + body;
|
|||
|
|
return eol === '\n' ? s : s.replace(/\n/g, eol);
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// ---------------------------------------------------------------- 解析
|
|||
|
|
|
|||
|
|
export function parse(raw) {
|
|||
|
|
const lines = raw.split('\n');
|
|||
|
|
const out = {};
|
|||
|
|
let i = 0;
|
|||
|
|
|
|||
|
|
while (i < lines.length) {
|
|||
|
|
const line = lines[i];
|
|||
|
|
if (line.trim() === '' || line.trimStart().startsWith('#')) {
|
|||
|
|
i++;
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
if (/^[ \t]/.test(line)) {
|
|||
|
|
throw new Error('顶层出现意外缩进: ' + JSON.stringify(line));
|
|||
|
|
}
|
|||
|
|
const m = /^([^:\s][^:]*):(.*)$/.exec(line);
|
|||
|
|
if (!m) throw new Error('无法解析的行: ' + JSON.stringify(line));
|
|||
|
|
|
|||
|
|
const key = m[1].trim();
|
|||
|
|
const rest = m[2].trim();
|
|||
|
|
i++;
|
|||
|
|
|
|||
|
|
// 空值:看后面有没有缩进块
|
|||
|
|
if (rest === '') {
|
|||
|
|
const block = [];
|
|||
|
|
while (i < lines.length && (lines[i].trim() === '' || /^[ \t]/.test(lines[i]))) {
|
|||
|
|
block.push(lines[i]);
|
|||
|
|
i++;
|
|||
|
|
}
|
|||
|
|
while (block.length && block[block.length - 1].trim() === '') block.pop();
|
|||
|
|
if (!block.length) {
|
|||
|
|
out[key] = null;
|
|||
|
|
} else if (/^-[ \t]?/.test(block[0].trimStart())) {
|
|||
|
|
out[key] = parseList(block);
|
|||
|
|
} else {
|
|||
|
|
out[key] = parseMap(block);
|
|||
|
|
}
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// 块标量
|
|||
|
|
if (/^[|>][+-]?$/.test(rest)) {
|
|||
|
|
const block = [];
|
|||
|
|
while (i < lines.length && (lines[i].trim() === '' || /^[ \t]/.test(lines[i]))) {
|
|||
|
|
block.push(lines[i]);
|
|||
|
|
i++;
|
|||
|
|
}
|
|||
|
|
out[key] = parseBlockScalar(block, rest);
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
out[key] = parseScalar(rest);
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
return out;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
function dedent(block) {
|
|||
|
|
let min = Infinity;
|
|||
|
|
for (const l of block) {
|
|||
|
|
if (l.trim() === '') continue;
|
|||
|
|
const n = l.match(/^[ \t]*/)[0].replace(/\t/g, ' ').length;
|
|||
|
|
if (n < min) min = n;
|
|||
|
|
}
|
|||
|
|
if (!isFinite(min)) min = 0;
|
|||
|
|
return block.map((l) => (l.trim() === '' ? '' : l.replace(/\t/g, ' ').slice(min)));
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
function parseBlockScalar(block, header) {
|
|||
|
|
const lines = dedent(block);
|
|||
|
|
while (lines.length && lines[lines.length - 1] === '') lines.pop();
|
|||
|
|
|
|||
|
|
let text;
|
|||
|
|
if (header[0] === '|') {
|
|||
|
|
text = lines.join('\n');
|
|||
|
|
} else {
|
|||
|
|
// 折叠:单个换行变空格,空行保留为换行
|
|||
|
|
let acc = '';
|
|||
|
|
for (const l of lines) {
|
|||
|
|
if (l === '') {
|
|||
|
|
acc += '\n';
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
if (acc !== '' && !acc.endsWith('\n')) acc += ' ';
|
|||
|
|
acc += l;
|
|||
|
|
}
|
|||
|
|
text = acc;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
const chomp = header[1];
|
|||
|
|
if (chomp === '-') return text.replace(/\n+$/, '');
|
|||
|
|
if (chomp === '+') return text + '\n';
|
|||
|
|
return text.replace(/\n+$/, '') + '\n';
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
function parseList(block) {
|
|||
|
|
const lines = dedent(block);
|
|||
|
|
const out = [];
|
|||
|
|
for (let i = 0; i < lines.length; i++) {
|
|||
|
|
const l = lines[i];
|
|||
|
|
if (l.trim() === '') continue;
|
|||
|
|
const m = /^-[ \t]?(.*)$/.exec(l);
|
|||
|
|
if (!m) throw new Error('列表中出现了非列表项: ' + JSON.stringify(l));
|
|||
|
|
if (/^[ \t]/.test(l)) throw new Error('不支持多级列表');
|
|||
|
|
out.push(parseScalar(m[1]));
|
|||
|
|
}
|
|||
|
|
return out;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
function parseMap(block) {
|
|||
|
|
const lines = dedent(block);
|
|||
|
|
const out = {};
|
|||
|
|
for (const l of lines) {
|
|||
|
|
if (l.trim() === '') continue;
|
|||
|
|
const m = /^([^:\s][^:]*):(.*)$/.exec(l);
|
|||
|
|
if (!m) throw new Error('嵌套 map 中出现无法解析的行: ' + JSON.stringify(l));
|
|||
|
|
out[m[1].trim()] = parseScalar(m[2].trim());
|
|||
|
|
}
|
|||
|
|
return out;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
function parseScalar(s) {
|
|||
|
|
s = s.trim();
|
|||
|
|
if (s === '' || s === '~' || s === 'null') return null;
|
|||
|
|
if (s === 'true') return true;
|
|||
|
|
if (s === 'false') return false;
|
|||
|
|
if (/^'.*'$/.test(s)) return s.slice(1, -1).replace(/''/g, "'");
|
|||
|
|
if (/^".*"$/.test(s)) {
|
|||
|
|
try {
|
|||
|
|
return JSON.parse(s);
|
|||
|
|
} catch {
|
|||
|
|
// JSON.parse 不认 YAML 专属的 \Uxxxxxxxx(8 位)与 \xXX 转义 ——
|
|||
|
|
// 仓库里就有这种标题(Telegram bot 写入的 emoji,如 \U0001F605)。
|
|||
|
|
// 先把它们解成真实字符,再交给 JSON.parse 处理其余转义(\n、\" 等)。
|
|||
|
|
const pre = s
|
|||
|
|
.slice(1, -1)
|
|||
|
|
.replace(/\\U([0-9a-fA-F]{8})/g, (_, h) => String.fromCodePoint(parseInt(h, 16)))
|
|||
|
|
.replace(/\\x([0-9a-fA-F]{2})/g, (_, h) => String.fromCodePoint(parseInt(h, 16)));
|
|||
|
|
try {
|
|||
|
|
return JSON.parse('"' + pre + '"');
|
|||
|
|
} catch {
|
|||
|
|
return pre;
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
if (s.startsWith('[') && s.endsWith(']')) return parseInlineList(s);
|
|||
|
|
if (/^-?\d+$/.test(s)) return parseInt(s, 10);
|
|||
|
|
if (/^-?\d+\.\d+$/.test(s)) return parseFloat(s);
|
|||
|
|
return s;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
function parseInlineList(s) {
|
|||
|
|
const inner = s.slice(1, -1).trim();
|
|||
|
|
if (!inner) return [];
|
|||
|
|
const out = [];
|
|||
|
|
let cur = '';
|
|||
|
|
let quote = null;
|
|||
|
|
for (const ch of inner) {
|
|||
|
|
if (quote) {
|
|||
|
|
cur += ch;
|
|||
|
|
if (ch === quote) quote = null;
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
if (ch === "'" || ch === '"') {
|
|||
|
|
quote = ch;
|
|||
|
|
cur += ch;
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
if (ch === ',') {
|
|||
|
|
out.push(parseScalar(cur));
|
|||
|
|
cur = '';
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
cur += ch;
|
|||
|
|
}
|
|||
|
|
if (cur.trim() !== '') out.push(parseScalar(cur));
|
|||
|
|
return out;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// ---------------------------------------------------------------- 序列化
|
|||
|
|
|
|||
|
|
const PLAIN_OK = /^[^\s\-?:,[\]{}#&*!|>'"%@`][^:#\n]*$/;
|
|||
|
|
|
|||
|
|
function toYamlScalar(v) {
|
|||
|
|
if (v === null || v === undefined) return '';
|
|||
|
|
if (typeof v === 'boolean') return v ? 'true' : 'false';
|
|||
|
|
if (typeof v === 'number') return String(v);
|
|||
|
|
|
|||
|
|
const s = String(v);
|
|||
|
|
if (s === '') return "''";
|
|||
|
|
// 这些形态不引起来会被 YAML 当成别的类型
|
|||
|
|
if (/^(true|false|null|~|yes|no|on|off)$/i.test(s)) return "'" + s + "'";
|
|||
|
|
if (/^-?\d+(\.\d+)?$/.test(s)) return s; // 纯数字:保持裸写,与现有语料一致
|
|||
|
|
if (s.includes('\n')) return null; // 交给调用方走块标量
|
|||
|
|
if (PLAIN_OK.test(s) && !s.startsWith(' ') && !s.endsWith(' ')) return s;
|
|||
|
|
return "'" + s.replace(/'/g, "''") + "'";
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
export function stringify(obj) {
|
|||
|
|
const out = [];
|
|||
|
|
for (const key of Object.keys(obj)) {
|
|||
|
|
const v = obj[key];
|
|||
|
|
|
|||
|
|
if (v === null || v === undefined) {
|
|||
|
|
out.push(key + ':');
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
if (Array.isArray(v)) {
|
|||
|
|
if (v.length === 0) {
|
|||
|
|
out.push(key + ': []');
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
out.push(key + ':');
|
|||
|
|
for (const item of v) {
|
|||
|
|
const one = toYamlScalar(item);
|
|||
|
|
if (one === null) throw new Error('列表项不支持多行内容: ' + key);
|
|||
|
|
out.push(' - ' + one);
|
|||
|
|
}
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
if (typeof v === 'object') {
|
|||
|
|
out.push(key + ':');
|
|||
|
|
for (const sub of Object.keys(v)) {
|
|||
|
|
const one = toYamlScalar(v[sub]);
|
|||
|
|
if (one === null) throw new Error('嵌套对象不支持多行内容: ' + key + '.' + sub);
|
|||
|
|
out.push(' ' + sub + ': ' + one);
|
|||
|
|
}
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
if (typeof v === 'string' && v.includes('\n')) {
|
|||
|
|
// 块标量必须带上正确的 chomping 记号,否则值会变:
|
|||
|
|
// | clip —— 保留结尾的一个换行(YAML 默认,Hugo 语料里的 >- 折叠块就是这个)
|
|||
|
|
// |- strip —— 结尾不要换行
|
|||
|
|
// |+ keep —— 保留全部结尾换行
|
|||
|
|
// 漏了这一步,「值 = "xxx\n"」会被写成「值 = "xxx"」,是实打实的语义改动。
|
|||
|
|
const trailing = (/\n+$/.exec(v) || [''])[0].length;
|
|||
|
|
const header = trailing === 0 ? '|-' : trailing === 1 ? '|' : '|+';
|
|||
|
|
out.push(key + ': ' + header);
|
|||
|
|
for (const line of v.replace(/\n+$/, '').split('\n')) {
|
|||
|
|
out.push(' ' + line);
|
|||
|
|
}
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
const one = toYamlScalar(v);
|
|||
|
|
if (one === null) throw new Error('无法序列化: ' + key);
|
|||
|
|
out.push(key + ': ' + one);
|
|||
|
|
}
|
|||
|
|
return out.join('\n');
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
/** 深比较:用来判断「用户到底动没动 front matter」 */
|
|||
|
|
export function deepEqual(a, b) {
|
|||
|
|
if (a === b) return true;
|
|||
|
|
if (a === null || b === null || typeof a !== 'object' || typeof b !== 'object') {
|
|||
|
|
return String(a) === String(b);
|
|||
|
|
}
|
|||
|
|
if (Array.isArray(a) !== Array.isArray(b)) return false;
|
|||
|
|
const ka = Object.keys(a);
|
|||
|
|
const kb = Object.keys(b);
|
|||
|
|
if (ka.length !== kb.length) return false;
|
|||
|
|
for (const k of ka) {
|
|||
|
|
if (!Object.prototype.hasOwnProperty.call(b, k)) return false;
|
|||
|
|
if (!deepEqual(a[k], b[k])) return false;
|
|||
|
|
}
|
|||
|
|
return true;
|
|||
|
|
}
|