import { execFile } from "node:child_process"; import { mkdir, readFile, writeFile } from "node:fs/promises"; import path from "node:path"; import { promisify } from "node:util"; import { kingWenEntries } from "./content-contract.mjs"; import { repositoryRoot } from "./repository-files.mjs"; const execFileAsync = promisify(execFile); const proxyUrl = process.env.HTTPS_PROXY || process.env.HTTP_PROXY || "http://127.0.0.1:1080"; const limit = Number.parseInt(process.env.IMPORT_LIMIT ?? "64", 10); const userAgent = "LingjiContentImport/1.0 (local Zhouyi pipeline; public-domain jingwen only)"; const outputDirectory = path.join(repositoryRoot, "content", "raw"); const outputPath = path.join(outputDirectory, "zhouyi-wikisource-jing.json"); const titles = [ "乾", "坤", "屯", "蒙", "需", "訟", "師", "比", "小畜", "履", "泰", "否", "同人", "大有", "謙", "豫", "隨", "蠱", "臨", "觀", "噬嗑", "賁", "剝", "復", "无妄", "大畜", "頤", "大過", "坎", "離", "咸", "恆", "遯", "大壯", "晉", "明夷", "家人", "睽", "蹇", "解", "損", "益", "夬", "姤", "萃", "升", "困", "井", "革", "鼎", "震", "艮", "漸", "歸妹", "豐", "旅", "巽", "兌", "渙", "節", "中孚", "小過", "既濟", "未濟", ]; const linePattern = /^(初九|九二|九三|九四|九五|上九|初六|六二|六三|六四|六五|上六|用九|用六)[::,,](.*)$/u; function stripMarkup(wikitext) { return wikitext .replace(/-\{([^}|]+)-\}/gu, "$1") .replace(/\{\{[^}]*\}\}/gu, "") .replace(/<[^>]+>/gu, "") .replace(/'{2,}/gu, "") .replace(/\[\[File:[^\]]*\]\]/gu, "") .replace(/\[\[(?:[^\|\]]*\|)?([^\]]+)\]\]/gu, "$1") .replace(/ /gu, " ") .replace(/\r/gu, ""); } function parseClassicFields(section) { const rawLines = section .split("\n") .map((line) => line.replace(/^[ *#:]+/u, "").trim()) .filter(Boolean); let judgment = ""; const lines = []; let special = null; for (const rawLine of rawLines) { const compact = rawLine.replace(/\s+/gu, ""); const hit = linePattern.exec(compact) ?? linePattern.exec(rawLine); if (hit) { const label = hit[1]; const text = `${label}:${hit[2].trim()}`; if (label === "用九" || label === "用六") { special = text; } else { lines.push(text); } } else if (lines.length === 0 && special == null) { const fragment = rawLine.replace(/\s+/gu, ""); judgment = judgment ? `${judgment}${fragment}` : fragment; } } judgment = judgment .replace(/^(?:周易)?[\u4e00-\u9fff]{1,3}[::]/u, "") .trim(); if (!judgment) throw new Error("missing judgment"); if (lines.length !== 6) { throw new Error(`expected 6 lines, got ${lines.length}: ${lines.join(" | ")}`); } return { judgmentOriginal: judgment, lineTextsBottomUp: lines, specialUsageText: special }; } function extractClassicSection(wikitext) { const normalized = stripMarkup(wikitext).replace(/\u3000/gu, " "); const classicIndex = normalized.search(/易[經经][::]/u); if (classicIndex < 0) throw new Error("missing 易經 section"); const afterClassic = normalized.slice(classicIndex); const stop = afterClassic.search(/\n[ *#]*彖曰[::]/u); return (stop >= 0 ? afterClassic.slice(0, stop) : afterClassic) .replace(/^易[經经][::]\s*/u, "") .trim(); } function stripHtml(value) { return value .replace(/<[^>]+>/gu, "") .replace(/"/gu, "\"") .replace(/&/gu, "&") .replace(/</gu, "<") .replace(/>/gu, ">") .replace(/\n+/gu, "\n") .trim(); } async function curlJson(args) { const { stdout } = await execFileAsync("curl.exe", [ "-sS", "--fail", "--retry", "12", "--retry-delay", "20", "--retry-all-errors", "--max-time", "45", "-x", proxyUrl, "-H", `User-Agent: ${userAgent}`, "-H", "Accept: application/json", ...args, ], { encoding: "utf8", maxBuffer: 8 * 1024 * 1024 }); return JSON.parse(stdout); } const titleFallbacks = { 恆: ["恒"], 遯: ["遁"], 晉: ["晋"], 大壯: ["大壮"], 歸妹: ["归妹"], 豐: ["丰"], 兌: ["兑"], 渙: ["涣"], 節: ["节"], 既濟: ["既济"], 未濟: ["未济"], }; async function fetchWikitext(title) { const candidates = [title, ...(titleFallbacks[title] ?? [])]; for (const candidate of candidates) { const url = new URL("https://zh.wikisource.org/w/api.php"); url.searchParams.set("action", "parse"); url.searchParams.set("page", `周易/${candidate}`); url.searchParams.set("prop", "wikitext"); url.searchParams.set("format", "json"); const payload = await curlJson([url.toString()]); const wikitext = payload?.parse?.wikitext?.["*"]; if (typeof wikitext === "string") return wikitext; process.stdout.write(`missing ${candidate}, trying fallback\n`); } throw new Error(`no wikitext for ${title}`); } async function toHansFields(classic) { const pieces = [ classic.judgmentOriginal, ...classic.lineTextsBottomUp, ...(classic.specialUsageText == null ? [] : [classic.specialUsageText]), ]; const converted = await convertToHans(pieces.join("\n¶\n")); const parts = converted .split("¶") .map((part) => part.replace(/\s+/gu, "").trim()) .filter(Boolean); if (parts.length !== pieces.length) { throw new Error(`zh-hans field count ${parts.length} != ${pieces.length}: ${converted}`); } return { judgmentOriginal: parts[0], lineTextsBottomUp: parts.slice(1, 7), specialUsageText: classic.specialUsageText == null ? null : parts[7], }; } async function convertToHans(text) { const payload = await curlJson([ "--data-urlencode", "action=parse", "--data-urlencode", `text=${text}`, "--data-urlencode", "prop=text", "--data-urlencode", "variant=zh-hans", "--data-urlencode", "disablelimitreport=1", "--data-urlencode", "wrapoutputclass=", "--data-urlencode", "contentmodel=wikitext", "--data-urlencode", "format=json", "https://zh.wikisource.org/w/api.php", ]); const html = payload?.parse?.text?.["*"]; if (typeof html !== "string") throw new Error("zh-hans conversion returned no text"); return stripHtml(html); } const entries = kingWenEntries(); await mkdir(outputDirectory, { recursive: true }); let imported = []; try { imported = JSON.parse(await readFile(outputPath, "utf8")); if (!Array.isArray(imported)) imported = []; } catch { imported = []; } const done = new Set(imported.map((item) => item.kingWenNumber)); for (const [index, title] of titles.entries()) { if (index >= limit) break; const kingWenNumber = index + 1; if (done.has(kingWenNumber)) { process.stdout.write(`skip ${kingWenNumber}/64 ${title}\n`); continue; } const wikitext = await fetchWikitext(title); const traditional = extractClassicSection(wikitext); const classic = await toHansFields(parseClassicFields(traditional)); imported.push({ kingWenNumber, wikisourceTitle: title, sourcePage: `https://zh.wikisource.org/wiki/周易/${title}`, ...entries[index], ...classic, }); imported.sort((left, right) => left.kingWenNumber - right.kingWenNumber); await writeFile(outputPath, `${JSON.stringify(imported, null, 2)}\n`); process.stdout.write(`imported ${kingWenNumber}/64 ${title}\n`); await new Promise((resolve) => setTimeout(resolve, 1200)); } process.stdout.write(`wrote ${outputPath} (${imported.length} hexagrams)\n`);