/** * 极简 YAML front matter 子集:解析 + 序列化(零依赖)。 * * 为什么不用 gray-matter / js-yaml: * 这个后端只服务「编辑自己的 Hugo 文章」这一件事,语料是固定的 100 多篇 * Markdown。为此拖进一棵依赖树不值得,镜像也能小一圈。 * * 覆盖的形态(Hugo front matter 实际会用到的全部): * key: 标量 key: '单引号' key: "双引号" * key: 123 / true key: [a, b] key: [] * key: (空值 → null) * - a * - b (列表,缩进两格) * key: | / |- / |+ / > / >- / >+ (块标量) * key: * sub: 1 (一层嵌套 map,Hugo 的 params 会用到) * * 遇到覆盖不到的结构一律 **抛错**,由调用方降级为「原文照存」—— * 宁可少解析,绝不猜错后把用户文章写坏。 */ const OPEN = /^---[ \t]*\n/; /** * 把整篇文本切成 front matter 原文 / 正文 / 原文件换行风格。 * * ★ 换行符必须原样保留:仓库 `.gitattributes` 是 `* text=auto`, * 仓库里存的是 LF,但 Windows 工作区检出是 CRLF。如果写回时统一成 LF, * 在 Windows 侧就会产生「整文件换行变更」的巨型 diff。所以这里记住原风格, * 交给 joinFrontMatter 还原。 * * body **保留**它前面的空行,这样 join 出来的结果与原文件逐字节相同。 */ export function splitFrontMatter(text) { const eol = text.includes('\r\n') ? '\r\n' : '\n'; const norm = text.replace(/\r\n/g, '\n'); const open = OPEN.exec(norm); if (!open) return { raw: null, body: norm, eol }; const rest = norm.slice(open[0].length); const lines = rest.split('\n'); for (let i = 0; i < lines.length; i++) { if (lines[i].trim() === '---') { return { raw: lines.slice(0, i).join('\n'), body: lines.slice(i + 1).join('\n'), eol, }; } } return { raw: null, body: norm, eol }; } /** 组装回整篇文本(raw 为 null 表示本来就没有 front matter) */ export function joinFrontMatter(raw, body, eol = '\n') { const s = raw == null ? body : '---\n' + raw.replace(/[ \t]+$/, '') + '\n---\n' + body; return eol === '\n' ? s : s.replace(/\n/g, eol); } // ---------------------------------------------------------------- 解析 export function parse(raw) { const lines = raw.split('\n'); const out = {}; let i = 0; while (i < lines.length) { const line = lines[i]; if (line.trim() === '' || line.trimStart().startsWith('#')) { i++; continue; } if (/^[ \t]/.test(line)) { throw new Error('顶层出现意外缩进: ' + JSON.stringify(line)); } const m = /^([^:\s][^:]*):(.*)$/.exec(line); if (!m) throw new Error('无法解析的行: ' + JSON.stringify(line)); const key = m[1].trim(); const rest = m[2].trim(); i++; // 空值:看后面有没有缩进块 if (rest === '') { const block = []; while (i < lines.length && (lines[i].trim() === '' || /^[ \t]/.test(lines[i]))) { block.push(lines[i]); i++; } while (block.length && block[block.length - 1].trim() === '') block.pop(); if (!block.length) { out[key] = null; } else if (/^-[ \t]?/.test(block[0].trimStart())) { out[key] = parseList(block); } else { out[key] = parseMap(block); } continue; } // 块标量 if (/^[|>][+-]?$/.test(rest)) { const block = []; while (i < lines.length && (lines[i].trim() === '' || /^[ \t]/.test(lines[i]))) { block.push(lines[i]); i++; } out[key] = parseBlockScalar(block, rest); continue; } out[key] = parseScalar(rest); } return out; } function dedent(block) { let min = Infinity; for (const l of block) { if (l.trim() === '') continue; const n = l.match(/^[ \t]*/)[0].replace(/\t/g, ' ').length; if (n < min) min = n; } if (!isFinite(min)) min = 0; return block.map((l) => (l.trim() === '' ? '' : l.replace(/\t/g, ' ').slice(min))); } function parseBlockScalar(block, header) { const lines = dedent(block); while (lines.length && lines[lines.length - 1] === '') lines.pop(); let text; if (header[0] === '|') { text = lines.join('\n'); } else { // 折叠:单个换行变空格,空行保留为换行 let acc = ''; for (const l of lines) { if (l === '') { acc += '\n'; continue; } if (acc !== '' && !acc.endsWith('\n')) acc += ' '; acc += l; } text = acc; } const chomp = header[1]; if (chomp === '-') return text.replace(/\n+$/, ''); if (chomp === '+') return text + '\n'; return text.replace(/\n+$/, '') + '\n'; } function parseList(block) { const lines = dedent(block); const out = []; for (let i = 0; i < lines.length; i++) { const l = lines[i]; if (l.trim() === '') continue; const m = /^-[ \t]?(.*)$/.exec(l); if (!m) throw new Error('列表中出现了非列表项: ' + JSON.stringify(l)); if (/^[ \t]/.test(l)) throw new Error('不支持多级列表'); out.push(parseScalar(m[1])); } return out; } function parseMap(block) { const lines = dedent(block); const out = {}; for (const l of lines) { if (l.trim() === '') continue; const m = /^([^:\s][^:]*):(.*)$/.exec(l); if (!m) throw new Error('嵌套 map 中出现无法解析的行: ' + JSON.stringify(l)); out[m[1].trim()] = parseScalar(m[2].trim()); } return out; } function parseScalar(s) { s = s.trim(); if (s === '' || s === '~' || s === 'null') return null; if (s === 'true') return true; if (s === 'false') return false; if (/^'.*'$/.test(s)) return s.slice(1, -1).replace(/''/g, "'"); if (/^".*"$/.test(s)) { try { return JSON.parse(s); } catch { // JSON.parse 不认 YAML 专属的 \Uxxxxxxxx(8 位)与 \xXX 转义 —— // 仓库里就有这种标题(Telegram bot 写入的 emoji,如 \U0001F605)。 // 先把它们解成真实字符,再交给 JSON.parse 处理其余转义(\n、\" 等)。 const pre = s .slice(1, -1) .replace(/\\U([0-9a-fA-F]{8})/g, (_, h) => String.fromCodePoint(parseInt(h, 16))) .replace(/\\x([0-9a-fA-F]{2})/g, (_, h) => String.fromCodePoint(parseInt(h, 16))); try { return JSON.parse('"' + pre + '"'); } catch { return pre; } } } if (s.startsWith('[') && s.endsWith(']')) return parseInlineList(s); if (/^-?\d+$/.test(s)) return parseInt(s, 10); if (/^-?\d+\.\d+$/.test(s)) return parseFloat(s); return s; } function parseInlineList(s) { const inner = s.slice(1, -1).trim(); if (!inner) return []; const out = []; let cur = ''; let quote = null; for (const ch of inner) { if (quote) { cur += ch; if (ch === quote) quote = null; continue; } if (ch === "'" || ch === '"') { quote = ch; cur += ch; continue; } if (ch === ',') { out.push(parseScalar(cur)); cur = ''; continue; } cur += ch; } if (cur.trim() !== '') out.push(parseScalar(cur)); return out; } // ---------------------------------------------------------------- 序列化 const PLAIN_OK = /^[^\s\-?:,[\]{}#&*!|>'"%@`][^:#\n]*$/; function toYamlScalar(v) { if (v === null || v === undefined) return ''; if (typeof v === 'boolean') return v ? 'true' : 'false'; if (typeof v === 'number') return String(v); const s = String(v); if (s === '') return "''"; // 这些形态不引起来会被 YAML 当成别的类型 if (/^(true|false|null|~|yes|no|on|off)$/i.test(s)) return "'" + s + "'"; if (/^-?\d+(\.\d+)?$/.test(s)) return s; // 纯数字:保持裸写,与现有语料一致 if (s.includes('\n')) return null; // 交给调用方走块标量 if (PLAIN_OK.test(s) && !s.startsWith(' ') && !s.endsWith(' ')) return s; return "'" + s.replace(/'/g, "''") + "'"; } export function stringify(obj) { const out = []; for (const key of Object.keys(obj)) { const v = obj[key]; if (v === null || v === undefined) { out.push(key + ':'); continue; } if (Array.isArray(v)) { if (v.length === 0) { out.push(key + ': []'); continue; } out.push(key + ':'); for (const item of v) { const one = toYamlScalar(item); if (one === null) throw new Error('列表项不支持多行内容: ' + key); out.push(' - ' + one); } continue; } if (typeof v === 'object') { out.push(key + ':'); for (const sub of Object.keys(v)) { const one = toYamlScalar(v[sub]); if (one === null) throw new Error('嵌套对象不支持多行内容: ' + key + '.' + sub); out.push(' ' + sub + ': ' + one); } continue; } if (typeof v === 'string' && v.includes('\n')) { // 块标量必须带上正确的 chomping 记号,否则值会变: // | clip —— 保留结尾的一个换行(YAML 默认,Hugo 语料里的 >- 折叠块就是这个) // |- strip —— 结尾不要换行 // |+ keep —— 保留全部结尾换行 // 漏了这一步,「值 = "xxx\n"」会被写成「值 = "xxx"」,是实打实的语义改动。 const trailing = (/\n+$/.exec(v) || [''])[0].length; const header = trailing === 0 ? '|-' : trailing === 1 ? '|' : '|+'; out.push(key + ': ' + header); for (const line of v.replace(/\n+$/, '').split('\n')) { out.push(' ' + line); } continue; } const one = toYamlScalar(v); if (one === null) throw new Error('无法序列化: ' + key); out.push(key + ': ' + one); } return out.join('\n'); } /** 深比较:用来判断「用户到底动没动 front matter」 */ export function deepEqual(a, b) { if (a === b) return true; if (a === null || b === null || typeof a !== 'object' || typeof b !== 'object') { return String(a) === String(b); } if (Array.isArray(a) !== Array.isArray(b)) return false; const ka = Object.keys(a); const kb = Object.keys(b); if (ka.length !== kb.length) return false; for (const k of ka) { if (!Object.prototype.hasOwnProperty.call(b, k)) return false; if (!deepEqual(a[k], b[k])) return false; } return true; }