feat(llm-wiki): 迁入 llm-wiki 技能包及依赖(85 文件,含 baoyu-url-to-markdown 适配器)
This commit is contained in:
625
llm-wiki/scripts/lib/graph-warning-bundle.js
Normal file
625
llm-wiki/scripts/lib/graph-warning-bundle.js
Normal file
@@ -0,0 +1,625 @@
|
||||
#!/usr/bin/env node
|
||||
"use strict";
|
||||
|
||||
const crypto = require("node:crypto");
|
||||
const fs = require("node:fs");
|
||||
const fsp = require("node:fs/promises");
|
||||
const path = require("node:path");
|
||||
const zlib = require("node:zlib");
|
||||
const { normalizeRelativePosixPath } = require("./wiki-file-discovery");
|
||||
|
||||
const DEFAULT_WARNING_DETAILS_REF = "wiki/graph-warnings.json";
|
||||
const OFFLINE_WARNING_LIMIT_BYTES = 2 * 1024 * 1024;
|
||||
function sha256(bytes) {
|
||||
return crypto.createHash("sha256").update(bytes).digest("hex");
|
||||
}
|
||||
|
||||
function canonicalize(value) {
|
||||
if (Array.isArray(value)) return value.map(canonicalize);
|
||||
if (!value || typeof value !== "object") return value;
|
||||
const result = {};
|
||||
for (const key of Object.keys(value).sort()) {
|
||||
if (value[key] !== undefined) result[key] = canonicalize(value[key]);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
function canonicalBytes(value) {
|
||||
return Buffer.from(JSON.stringify(canonicalize(value)), "utf8");
|
||||
}
|
||||
|
||||
function serializeJsonForHtmlScript(value) {
|
||||
const json = JSON.stringify(canonicalize(value));
|
||||
return Buffer.from(json.replace(/[<>&\u2028\u2029]/g, (character) => (
|
||||
`\\u${character.codePointAt(0).toString(16).padStart(4, "0")}`
|
||||
)), "utf8");
|
||||
}
|
||||
|
||||
function compareText(left, right) {
|
||||
return String(left).localeCompare(String(right), "en");
|
||||
}
|
||||
|
||||
function assertRelativeContentPath(value, fieldName) {
|
||||
if (typeof value !== "string" || !value || value.includes("\\")) {
|
||||
throw new Error(`${fieldName} must be a POSIX knowledge-base-relative path`);
|
||||
}
|
||||
let normalized;
|
||||
try {
|
||||
normalized = normalizeRelativePosixPath(value);
|
||||
} catch (_) {
|
||||
throw new Error(`${fieldName} must be a POSIX knowledge-base-relative path`);
|
||||
}
|
||||
if (normalized !== value || path.posix.isAbsolute(value)) {
|
||||
throw new Error(`${fieldName} must be a POSIX knowledge-base-relative path`);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
function validateDetailsRef(detailsRef) {
|
||||
assertRelativeContentPath(detailsRef, "details_ref");
|
||||
if (path.posix.basename(detailsRef) !== "graph-warnings.json") {
|
||||
throw new Error("details_ref must name graph-warnings.json");
|
||||
}
|
||||
return detailsRef;
|
||||
}
|
||||
|
||||
function normalizeOccurrence(occurrence) {
|
||||
if (!occurrence || typeof occurrence !== "object") {
|
||||
throw new Error("warning occurrence must be an object");
|
||||
}
|
||||
assertRelativeContentPath(occurrence.source_path, "source_path");
|
||||
return canonicalize(occurrence);
|
||||
}
|
||||
|
||||
function normalizeCandidateSets(candidateSets) {
|
||||
if (!Array.isArray(candidateSets)) throw new Error("candidateSets must be an array");
|
||||
const seen = new Set();
|
||||
return candidateSets.map((candidateSet) => {
|
||||
if (!candidateSet || typeof candidateSet !== "object" || !candidateSet.candidate_set_id) {
|
||||
throw new Error("candidate_set_id is required");
|
||||
}
|
||||
if (seen.has(candidateSet.candidate_set_id)) {
|
||||
throw new Error(`duplicate candidate_set_id: ${candidateSet.candidate_set_id}`);
|
||||
}
|
||||
seen.add(candidateSet.candidate_set_id);
|
||||
const candidates = Array.from(new Set((candidateSet.candidates || []).map((candidate) => (
|
||||
assertRelativeContentPath(candidate, "candidate path")
|
||||
)))).sort(compareText);
|
||||
if (candidateSet.candidate_count !== candidates.length) {
|
||||
throw new Error(`candidate_count does not match candidates for ${candidateSet.candidate_set_id}`);
|
||||
}
|
||||
return canonicalize({ ...candidateSet, candidates });
|
||||
}).sort((left, right) => compareText(left.candidate_set_id, right.candidate_set_id));
|
||||
}
|
||||
|
||||
function normalizeGroups(groups, candidateSetIds) {
|
||||
if (!Array.isArray(groups)) throw new Error("groups must be an array");
|
||||
const seen = new Set();
|
||||
const seenOccurrences = new Set();
|
||||
return groups.map((group) => {
|
||||
if (!group || typeof group !== "object" || !group.warning_id) {
|
||||
throw new Error("warning_id is required");
|
||||
}
|
||||
if (seen.has(group.warning_id)) throw new Error(`duplicate warning_id: ${group.warning_id}`);
|
||||
seen.add(group.warning_id);
|
||||
if (group.candidate_set_id && !candidateSetIds.has(group.candidate_set_id)) {
|
||||
throw new Error(`warning references missing candidate set: ${group.candidate_set_id}`);
|
||||
}
|
||||
if (!Number.isSafeInteger(group.occurrence_count) || group.occurrence_count < 0) {
|
||||
throw new Error(`invalid occurrence_count for ${group.warning_id}`);
|
||||
}
|
||||
const occurrences = (group.occurrences || []).map(normalizeOccurrence)
|
||||
.sort((left, right) => compareText(left.occurrence_id, right.occurrence_id));
|
||||
for (const occurrence of occurrences) {
|
||||
if (seenOccurrences.has(occurrence.occurrence_id)) {
|
||||
throw new Error(`duplicate occurrence_id: ${occurrence.occurrence_id}`);
|
||||
}
|
||||
seenOccurrences.add(occurrence.occurrence_id);
|
||||
}
|
||||
if (group.occurrence_count !== occurrences.length) {
|
||||
throw new Error(`occurrence_count does not match occurrences for ${group.warning_id}`);
|
||||
}
|
||||
return canonicalize({ ...group, occurrences });
|
||||
}).sort((left, right) => compareText(left.warning_id, right.warning_id));
|
||||
}
|
||||
|
||||
function sortGraphCollections(graphData) {
|
||||
const graph = structuredClone(graphData || {});
|
||||
if (graph.meta && typeof graph.meta === "object") delete graph.meta.warning_summary;
|
||||
if (Array.isArray(graph.nodes)) graph.nodes.sort((left, right) => compareText(left.id, right.id));
|
||||
if (Array.isArray(graph.edges)) {
|
||||
graph.edges.sort((left, right) => compareText(
|
||||
`${left.id || ""}\0${left.from || ""}\0${left.to || ""}`,
|
||||
`${right.id || ""}\0${right.from || ""}\0${right.to || ""}`
|
||||
));
|
||||
}
|
||||
if (graph.learning && Array.isArray(graph.learning.communities)) {
|
||||
graph.learning.communities.sort((left, right) => compareText(left.id, right.id));
|
||||
}
|
||||
return canonicalize(graph);
|
||||
}
|
||||
|
||||
function graphBuildIdentityProjection(graphData) {
|
||||
const graph = sortGraphCollections(graphData);
|
||||
if (graph.meta && typeof graph.meta === "object") delete graph.meta.build_date;
|
||||
return canonicalize(graph);
|
||||
}
|
||||
|
||||
function canonicalWarningDetailBytes(bundle) {
|
||||
return canonicalBytes({
|
||||
version: bundle.version,
|
||||
build_id: bundle.build_id,
|
||||
candidate_sets: bundle.candidate_sets,
|
||||
groups: bundle.groups
|
||||
});
|
||||
}
|
||||
|
||||
function summarizeWarningGroups(groups) {
|
||||
const byCode = {};
|
||||
let errorOccurrences = 0;
|
||||
let warningOccurrences = 0;
|
||||
for (const group of groups) {
|
||||
byCode[group.code] = (byCode[group.code] || 0) + group.occurrence_count;
|
||||
if (group.severity === "error") errorOccurrences += group.occurrence_count;
|
||||
else warningOccurrences += group.occurrence_count;
|
||||
}
|
||||
return canonicalize({
|
||||
total_groups: groups.length,
|
||||
total_occurrences: errorOccurrences + warningOccurrences,
|
||||
error_occurrences: errorOccurrences,
|
||||
warning_occurrences: warningOccurrences,
|
||||
by_code: byCode
|
||||
});
|
||||
}
|
||||
|
||||
function summaryCountsMatch(summary, groups) {
|
||||
const actual = canonicalize({
|
||||
total_groups: summary.total_groups,
|
||||
total_occurrences: summary.total_occurrences,
|
||||
error_occurrences: summary.error_occurrences,
|
||||
warning_occurrences: summary.warning_occurrences,
|
||||
by_code: summary.by_code
|
||||
});
|
||||
return canonicalBytes(actual).equals(canonicalBytes(summarizeWarningGroups(groups)));
|
||||
}
|
||||
|
||||
function assembleGraphArtifactPair({
|
||||
graphData,
|
||||
groups,
|
||||
candidateSets,
|
||||
detailsRef = DEFAULT_WARNING_DETAILS_REF
|
||||
}) {
|
||||
const validatedDetailsRef = validateDetailsRef(detailsRef);
|
||||
const candidate_sets = normalizeCandidateSets(candidateSets);
|
||||
const normalizedGroups = normalizeGroups(groups, new Set(candidate_sets.map((item) => item.candidate_set_id)));
|
||||
const graphWithoutSummary = sortGraphCollections(graphData);
|
||||
const build_id = sha256(canonicalBytes({
|
||||
graph_without_warning_summary: graphBuildIdentityProjection(graphWithoutSummary),
|
||||
warning_details: { candidate_sets, groups: normalizedGroups }
|
||||
}));
|
||||
const detailProjection = {
|
||||
version: 1,
|
||||
build_id,
|
||||
candidate_sets,
|
||||
groups: normalizedGroups
|
||||
};
|
||||
const details_sha256 = sha256(canonicalBytes(detailProjection));
|
||||
const counts = summarizeWarningGroups(normalizedGroups);
|
||||
const summary = canonicalize({
|
||||
build_id,
|
||||
...counts,
|
||||
details_ref: validatedDetailsRef,
|
||||
details_sha256
|
||||
});
|
||||
const normalizedGraph = canonicalize({
|
||||
...graphWithoutSummary,
|
||||
meta: { ...(graphWithoutSummary.meta || {}), warning_summary: summary }
|
||||
});
|
||||
const warningBundle = canonicalize({
|
||||
version: 1,
|
||||
build_id,
|
||||
summary,
|
||||
candidate_sets,
|
||||
groups: normalizedGroups
|
||||
});
|
||||
return { graphData: normalizedGraph, warningBundle };
|
||||
}
|
||||
|
||||
function summariesMatch(left, right) {
|
||||
return Boolean(left && right && canonicalBytes(left).equals(canonicalBytes(right)));
|
||||
}
|
||||
|
||||
function graphWithoutWarningSummary(graphData) {
|
||||
return sortGraphCollections(graphData);
|
||||
}
|
||||
|
||||
function recalculateBuildId(graphData, warningBundle) {
|
||||
return sha256(canonicalBytes({
|
||||
graph_without_warning_summary: graphBuildIdentityProjection(graphData),
|
||||
warning_details: {
|
||||
candidate_sets: warningBundle.candidate_sets,
|
||||
groups: warningBundle.groups
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
function parseArtifactBytes(bytes) {
|
||||
return JSON.parse(Buffer.isBuffer(bytes) ? bytes.toString("utf8") : String(bytes));
|
||||
}
|
||||
|
||||
function validateArtifactObjects({ graphData, warningBundle, expectedDetailsRef }) {
|
||||
const summary = graphData && graphData.meta && graphData.meta.warning_summary;
|
||||
if (!summary || typeof summary !== "object") {
|
||||
return { status: "unavailable", reason: "invalid", summary: summary || null };
|
||||
}
|
||||
|
||||
try {
|
||||
validateDetailsRef(summary.details_ref);
|
||||
} catch (_) {
|
||||
return { status: "unavailable", reason: "invalid", summary };
|
||||
}
|
||||
if (summary.details_ref !== expectedDetailsRef) {
|
||||
return { status: "unavailable", reason: "details_ref_mismatch", summary };
|
||||
}
|
||||
if (!warningBundle || typeof warningBundle !== "object" || warningBundle.version !== 1) {
|
||||
return { status: "unavailable", reason: "invalid", summary };
|
||||
}
|
||||
if (summary.build_id !== warningBundle.build_id || !summariesMatch(summary, warningBundle.summary)) {
|
||||
return { status: "unavailable", reason: "build_id_mismatch", summary };
|
||||
}
|
||||
|
||||
let normalizedSets;
|
||||
let normalizedGroups;
|
||||
try {
|
||||
normalizedSets = normalizeCandidateSets(warningBundle.candidate_sets);
|
||||
normalizedGroups = normalizeGroups(
|
||||
warningBundle.groups,
|
||||
new Set(normalizedSets.map((item) => item.candidate_set_id))
|
||||
);
|
||||
} catch (_) {
|
||||
return { status: "unavailable", reason: "invalid", summary };
|
||||
}
|
||||
const canonicalBundle = canonicalize({
|
||||
...warningBundle,
|
||||
candidate_sets: normalizedSets,
|
||||
groups: normalizedGroups
|
||||
});
|
||||
if (!summaryCountsMatch(summary, normalizedGroups)) {
|
||||
return { status: "unavailable", reason: "invalid", summary };
|
||||
}
|
||||
const actualDetailsSha256 = sha256(canonicalWarningDetailBytes(canonicalBundle));
|
||||
if (summary.details_sha256 !== actualDetailsSha256) {
|
||||
return { status: "unavailable", reason: "details_sha256_mismatch", summary };
|
||||
}
|
||||
if (recalculateBuildId(graphData, canonicalBundle) !== summary.build_id) {
|
||||
return { status: "unavailable", reason: "build_id_mismatch", summary };
|
||||
}
|
||||
|
||||
return { status: "available", graphData, warningBundle: canonicalBundle };
|
||||
}
|
||||
|
||||
function isWithinRoot(rootPath, candidatePath) {
|
||||
return candidatePath === rootPath || candidatePath.startsWith(`${rootPath}${path.sep}`);
|
||||
}
|
||||
|
||||
async function validateArtifactDestinations({ kbRoot, graphPath, warningPath, detailsRef }) {
|
||||
const rootReal = await fsp.realpath(kbRoot);
|
||||
const graphAbsolute = path.resolve(graphPath);
|
||||
const warningAbsolute = path.resolve(warningPath);
|
||||
if (path.basename(graphAbsolute) !== "graph-data.json") {
|
||||
throw new Error("graph destination basename must be graph-data.json");
|
||||
}
|
||||
if (path.basename(warningAbsolute) !== "graph-warnings.json") {
|
||||
throw new Error("warning destination basename must be graph-warnings.json");
|
||||
}
|
||||
|
||||
const graphParentReal = await fsp.realpath(path.dirname(graphAbsolute));
|
||||
const warningParentReal = await fsp.realpath(path.dirname(warningAbsolute));
|
||||
if (!isWithinRoot(rootReal, graphParentReal) || !isWithinRoot(rootReal, warningParentReal)) {
|
||||
throw new Error("artifact destination must remain inside the knowledge base");
|
||||
}
|
||||
if (graphParentReal !== warningParentReal) {
|
||||
throw new Error("graph-data.json and graph-warnings.json must be sibling artifacts");
|
||||
}
|
||||
const graphFinal = path.join(graphParentReal, "graph-data.json");
|
||||
const warningFinal = path.join(warningParentReal, "graph-warnings.json");
|
||||
for (const finalPath of [graphFinal, warningFinal]) {
|
||||
try {
|
||||
const stat = await fsp.lstat(finalPath);
|
||||
if (stat.isSymbolicLink() || !stat.isFile()) {
|
||||
throw new Error(`artifact destination is not a regular file: ${finalPath}`);
|
||||
}
|
||||
} catch (error) {
|
||||
if (error.code !== "ENOENT") throw error;
|
||||
}
|
||||
}
|
||||
|
||||
const normalizedDetailsRef = validateDetailsRef(detailsRef);
|
||||
const detailsAbsolute = path.resolve(rootReal, ...normalizedDetailsRef.split("/"));
|
||||
const detailsParentReal = await fsp.realpath(path.dirname(detailsAbsolute));
|
||||
const resolvedDetails = path.join(detailsParentReal, path.basename(detailsAbsolute));
|
||||
if (!isWithinRoot(rootReal, detailsParentReal) || resolvedDetails !== warningFinal) {
|
||||
throw new Error("details_ref does not resolve to the final sibling graph-warnings.json");
|
||||
}
|
||||
const expectedDetailsRef = path.relative(rootReal, warningFinal).split(path.sep).join("/");
|
||||
if (normalizedDetailsRef !== expectedDetailsRef) {
|
||||
throw new Error("details_ref does not match the final warning destination");
|
||||
}
|
||||
return { rootReal, graphFinal, warningFinal, outputParent: graphParentReal, expectedDetailsRef };
|
||||
}
|
||||
|
||||
async function writeSyncedFile(filePath, bytes) {
|
||||
const handle = await fsp.open(filePath, "wx", 0o600);
|
||||
try {
|
||||
await handle.writeFile(bytes);
|
||||
await handle.sync();
|
||||
} finally {
|
||||
await handle.close();
|
||||
}
|
||||
}
|
||||
|
||||
async function fsyncDirectory(directoryPath) {
|
||||
let handle;
|
||||
try {
|
||||
handle = await fsp.open(directoryPath, fs.constants.O_RDONLY);
|
||||
await handle.sync();
|
||||
} catch (error) {
|
||||
if (!["EINVAL", "ENOTSUP", "EBADF", "EPERM", "EISDIR"].includes(error.code)) throw error;
|
||||
} finally {
|
||||
if (handle) await handle.close();
|
||||
}
|
||||
}
|
||||
|
||||
function operationDirectoryName(name) {
|
||||
return /^[a-f0-9]{64}-[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i.test(name);
|
||||
}
|
||||
|
||||
async function pruneOldOperationDirectories(buildRoot, currentDirectory, now) {
|
||||
const entries = await fsp.readdir(buildRoot, { withFileTypes: true });
|
||||
for (const entry of entries) {
|
||||
if (!entry.isDirectory() || !operationDirectoryName(entry.name)) continue;
|
||||
const candidate = path.join(buildRoot, entry.name);
|
||||
if (candidate === currentDirectory) continue;
|
||||
const stat = await fsp.stat(candidate);
|
||||
if (now - stat.mtimeMs > 24 * 60 * 60 * 1000) {
|
||||
await fsp.rm(candidate, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function commitGraphArtifactPair({ kbRoot, graphPath, warningPath, pair, hooks = {} }) {
|
||||
if (!pair || !pair.graphData || !pair.warningBundle) throw new Error("artifact pair is required");
|
||||
const summary = pair.graphData.meta && pair.graphData.meta.warning_summary;
|
||||
if (!summary) throw new Error("graph warning summary is required");
|
||||
const destinations = await validateArtifactDestinations({
|
||||
kbRoot,
|
||||
graphPath,
|
||||
warningPath,
|
||||
detailsRef: summary.details_ref
|
||||
});
|
||||
const stat = hooks.stat || ((target) => fsp.stat(target));
|
||||
const tempParent = path.join(destinations.rootReal, ".wiki-tmp");
|
||||
await fsp.mkdir(tempParent, { recursive: true, mode: 0o700 });
|
||||
const tempParentReal = await fsp.realpath(tempParent);
|
||||
if (!isWithinRoot(destinations.rootReal, tempParentReal)) {
|
||||
throw new Error("temporary graph build directory escapes knowledge base");
|
||||
}
|
||||
const buildRoot = path.join(tempParentReal, "graph-build");
|
||||
await fsp.mkdir(buildRoot, { recursive: true, mode: 0o700 });
|
||||
|
||||
const [tempDevice, graphDevice, warningDevice] = await Promise.all([
|
||||
stat(tempParentReal),
|
||||
stat(path.dirname(destinations.graphFinal)),
|
||||
stat(path.dirname(destinations.warningFinal))
|
||||
]);
|
||||
if (tempDevice.dev !== graphDevice.dev || tempDevice.dev !== warningDevice.dev) {
|
||||
throw new Error("graph artifact destinations must use the same filesystem device as .wiki-tmp");
|
||||
}
|
||||
|
||||
const graphBytes = Buffer.from(`${JSON.stringify(pair.graphData, null, 2)}\n`, "utf8");
|
||||
const warningBytes = Buffer.from(`${JSON.stringify(pair.warningBundle, null, 2)}\n`, "utf8");
|
||||
let graphObject;
|
||||
let warningObject;
|
||||
try {
|
||||
graphObject = parseArtifactBytes(graphBytes);
|
||||
warningObject = parseArtifactBytes(warningBytes);
|
||||
} catch (error) {
|
||||
throw new Error(`invalid artifact pair JSON: ${error.message}`);
|
||||
}
|
||||
const preflight = validateArtifactObjects({
|
||||
graphData: graphObject,
|
||||
warningBundle: warningObject,
|
||||
expectedDetailsRef: destinations.expectedDetailsRef
|
||||
});
|
||||
if (preflight.status !== "available") {
|
||||
throw new Error(`invalid artifact pair: ${preflight.reason}`);
|
||||
}
|
||||
|
||||
const operationDirectory = path.join(
|
||||
buildRoot,
|
||||
`${summary.build_id}-${crypto.randomUUID()}`
|
||||
);
|
||||
await fsp.mkdir(operationDirectory, { recursive: false, mode: 0o700 });
|
||||
const tempGraph = path.join(operationDirectory, "graph-data.json");
|
||||
const tempWarning = path.join(operationDirectory, "graph-warnings.json");
|
||||
|
||||
await writeSyncedFile(tempGraph, graphBytes);
|
||||
await writeSyncedFile(tempWarning, warningBytes);
|
||||
const verifiedTemporary = validateArtifactObjects({
|
||||
graphData: parseArtifactBytes(await fsp.readFile(tempGraph)),
|
||||
warningBundle: parseArtifactBytes(await fsp.readFile(tempWarning)),
|
||||
expectedDetailsRef: destinations.expectedDetailsRef
|
||||
});
|
||||
if (verifiedTemporary.status !== "available") {
|
||||
throw new Error(`temporary artifact verification failed: ${verifiedTemporary.reason}`);
|
||||
}
|
||||
|
||||
await fsp.rename(tempWarning, destinations.warningFinal);
|
||||
await fsyncDirectory(destinations.outputParent);
|
||||
if (hooks.afterWarningReplace) await hooks.afterWarningReplace();
|
||||
await fsp.rename(tempGraph, destinations.graphFinal);
|
||||
await fsyncDirectory(destinations.outputParent);
|
||||
await fsp.rm(operationDirectory, { recursive: true, force: true });
|
||||
await pruneOldOperationDirectories(
|
||||
buildRoot,
|
||||
operationDirectory,
|
||||
hooks.now ? hooks.now() : Date.now()
|
||||
);
|
||||
}
|
||||
|
||||
async function verifyGraphArtifactPair({ kbRoot, graphPath, warningPath }) {
|
||||
let graphData;
|
||||
try {
|
||||
graphData = parseArtifactBytes(await fsp.readFile(graphPath));
|
||||
} catch (_) {
|
||||
return { status: "unavailable", reason: "invalid", summary: null };
|
||||
}
|
||||
const summary = graphData && graphData.meta && graphData.meta.warning_summary;
|
||||
if (!summary || typeof summary !== "object") {
|
||||
return { status: "unavailable", reason: "invalid", summary: summary || null };
|
||||
}
|
||||
|
||||
let destinations;
|
||||
try {
|
||||
destinations = await validateArtifactDestinations({
|
||||
kbRoot,
|
||||
graphPath,
|
||||
warningPath,
|
||||
detailsRef: summary.details_ref
|
||||
});
|
||||
} catch (error) {
|
||||
return {
|
||||
status: "unavailable",
|
||||
reason: error.message.includes("details_ref") ? "details_ref_mismatch" : "invalid",
|
||||
summary
|
||||
};
|
||||
}
|
||||
|
||||
let warningBytes;
|
||||
try {
|
||||
warningBytes = await fsp.readFile(destinations.warningFinal);
|
||||
} catch (error) {
|
||||
return {
|
||||
status: "unavailable",
|
||||
reason: error.code === "ENOENT" ? "missing" : "invalid",
|
||||
summary
|
||||
};
|
||||
}
|
||||
let warningBundle;
|
||||
try {
|
||||
warningBundle = parseArtifactBytes(warningBytes);
|
||||
} catch (_) {
|
||||
return { status: "unavailable", reason: "invalid", summary };
|
||||
}
|
||||
return validateArtifactObjects({
|
||||
graphData,
|
||||
warningBundle,
|
||||
expectedDetailsRef: destinations.expectedDetailsRef
|
||||
});
|
||||
}
|
||||
|
||||
function canonicalOfflineBundle(bundle) {
|
||||
const candidate_sets = (bundle.candidate_sets || []).map((candidateSet) => canonicalize({
|
||||
...candidateSet,
|
||||
candidates: (candidateSet.candidates || []).slice().sort(compareText)
|
||||
})).sort((left, right) => compareText(left.candidate_set_id, right.candidate_set_id));
|
||||
const groups = (bundle.groups || []).map((group) => canonicalize({
|
||||
...group,
|
||||
occurrences: (group.occurrences || []).slice()
|
||||
.sort((left, right) => compareText(left.occurrence_id, right.occurrence_id))
|
||||
})).sort((left, right) => compareText(left.warning_id, right.warning_id));
|
||||
return canonicalize({ ...bundle, candidate_sets, groups });
|
||||
}
|
||||
|
||||
function offlinePayload(summary, bundle, truncated, omittedGroupCount, omittedCandidateSetCount) {
|
||||
return canonicalize({
|
||||
summary,
|
||||
details_status: "available",
|
||||
details_unavailable_reason: null,
|
||||
warning_details_truncated: truncated,
|
||||
omitted_group_count: omittedGroupCount,
|
||||
omitted_candidate_set_count: omittedCandidateSetCount,
|
||||
bundle
|
||||
});
|
||||
}
|
||||
|
||||
function compressedPayloadBytes(payload) {
|
||||
const scriptBytes = serializeJsonForHtmlScript(payload);
|
||||
return {
|
||||
scriptBytes,
|
||||
compressedBytes: zlib.gzipSync(scriptBytes, { level: 9 }).length
|
||||
};
|
||||
}
|
||||
|
||||
function prepareOfflineWarningPayload({
|
||||
summary,
|
||||
bundle,
|
||||
maxCompressedBytes = OFFLINE_WARNING_LIMIT_BYTES
|
||||
}) {
|
||||
if (!Number.isSafeInteger(maxCompressedBytes) || maxCompressedBytes <= 0) {
|
||||
throw new Error("maxCompressedBytes must be a positive integer");
|
||||
}
|
||||
const completeBundle = canonicalOfflineBundle(bundle);
|
||||
let payload = offlinePayload(summary, completeBundle, false, 0, 0);
|
||||
let { scriptBytes, compressedBytes } = compressedPayloadBytes(payload);
|
||||
if (compressedBytes <= maxCompressedBytes) return { payload, scriptBytes, compressedBytes };
|
||||
|
||||
const compactBundle = canonicalOfflineBundle({
|
||||
...completeBundle,
|
||||
groups: completeBundle.groups.map((group) => ({
|
||||
...group,
|
||||
occurrences: group.occurrences.slice(0, 20)
|
||||
})),
|
||||
candidate_sets: completeBundle.candidate_sets.map((candidateSet) => ({
|
||||
...candidateSet,
|
||||
candidates: candidateSet.candidates.slice(0, 20)
|
||||
}))
|
||||
});
|
||||
let omittedGroupCount = 0;
|
||||
let omittedCandidateSetCount = 0;
|
||||
const refresh = () => {
|
||||
payload = offlinePayload(summary, compactBundle, true, omittedGroupCount, omittedCandidateSetCount);
|
||||
({ scriptBytes, compressedBytes } = compressedPayloadBytes(payload));
|
||||
return compressedBytes <= maxCompressedBytes;
|
||||
};
|
||||
if (refresh()) return { payload, scriptBytes, compressedBytes };
|
||||
|
||||
for (let index = compactBundle.groups.length - 1; index >= 0; index -= 1) {
|
||||
while (compactBundle.groups[index].occurrences.length > 0) {
|
||||
compactBundle.groups[index].occurrences.pop();
|
||||
if (refresh()) return { payload, scriptBytes, compressedBytes };
|
||||
}
|
||||
}
|
||||
for (let index = compactBundle.candidate_sets.length - 1; index >= 0; index -= 1) {
|
||||
while (compactBundle.candidate_sets[index].candidates.length > 0) {
|
||||
compactBundle.candidate_sets[index].candidates.pop();
|
||||
if (refresh()) return { payload, scriptBytes, compressedBytes };
|
||||
}
|
||||
}
|
||||
|
||||
while (compactBundle.groups.length > 0) {
|
||||
compactBundle.groups.pop();
|
||||
omittedGroupCount += 1;
|
||||
if (refresh()) return { payload, scriptBytes, compressedBytes };
|
||||
}
|
||||
while (compactBundle.candidate_sets.length > 0) {
|
||||
compactBundle.candidate_sets.pop();
|
||||
omittedCandidateSetCount += 1;
|
||||
if (refresh()) return { payload, scriptBytes, compressedBytes };
|
||||
}
|
||||
if (!refresh()) {
|
||||
throw new Error("offline warning summary exceeds the compressed payload limit");
|
||||
}
|
||||
return { payload, scriptBytes, compressedBytes };
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
DEFAULT_WARNING_DETAILS_REF,
|
||||
OFFLINE_WARNING_LIMIT_BYTES,
|
||||
assembleGraphArtifactPair,
|
||||
canonicalWarningDetailBytes,
|
||||
commitGraphArtifactPair,
|
||||
prepareOfflineWarningPayload,
|
||||
serializeJsonForHtmlScript,
|
||||
verifyGraphArtifactPair
|
||||
};
|
||||
158
llm-wiki/scripts/lib/source-signal-eligibility.js
Normal file
158
llm-wiki/scripts/lib/source-signal-eligibility.js
Normal file
@@ -0,0 +1,158 @@
|
||||
#!/usr/bin/env node
|
||||
"use strict";
|
||||
|
||||
const SCAN_KINDS = [
|
||||
{ subdir: "entities", pageType: "entity", applicable: true },
|
||||
{ subdir: "topics", pageType: "topic", applicable: true },
|
||||
{ subdir: "sources", pageType: "source", applicable: true },
|
||||
{ subdir: "comparisons", pageType: "comparison", applicable: true },
|
||||
{ subdir: "queries", pageType: "query", applicable: false },
|
||||
{ subdir: "synthesis", pageType: "synthesis", applicable: false }
|
||||
];
|
||||
|
||||
function sortedUnique(values) {
|
||||
return Array.from(new Set(values)).sort();
|
||||
}
|
||||
|
||||
function extractFrontmatter(text) {
|
||||
if (!text.startsWith("---\n") && !text.startsWith("---\r\n")) {
|
||||
return { hasFrontmatter: false, frontmatter: "", body: text };
|
||||
}
|
||||
|
||||
const match = text.match(/^---\r?\n([\s\S]*?)\r?\n---(?:\r?\n|$)([\s\S]*)$/);
|
||||
if (!match) {
|
||||
return { hasFrontmatter: false, frontmatter: "", body: text };
|
||||
}
|
||||
|
||||
return {
|
||||
hasFrontmatter: true,
|
||||
frontmatter: match[1],
|
||||
body: match[2]
|
||||
};
|
||||
}
|
||||
|
||||
function normalizeSourceToken(token) {
|
||||
const trimmed = String(token || "").trim();
|
||||
if (!trimmed) return null;
|
||||
|
||||
let value = trimmed;
|
||||
if ((value.startsWith('"') && value.endsWith('"')) || (value.startsWith("'") && value.endsWith("'"))) {
|
||||
value = value.slice(1, -1).trim();
|
||||
}
|
||||
|
||||
return value || null;
|
||||
}
|
||||
|
||||
function parseInlineSources(raw) {
|
||||
const trimmed = raw.trim();
|
||||
if (trimmed === "[]") return { ok: true, values: [] };
|
||||
if (!(trimmed.startsWith("[") && trimmed.endsWith("]"))) {
|
||||
return { ok: false, values: [] };
|
||||
}
|
||||
|
||||
const inner = trimmed.slice(1, -1).trim();
|
||||
if (!inner) return { ok: true, values: [] };
|
||||
|
||||
const values = inner
|
||||
.split(",")
|
||||
.map(normalizeSourceToken)
|
||||
.filter(Boolean);
|
||||
|
||||
return { ok: true, values };
|
||||
}
|
||||
|
||||
function parseSourcesFrontmatter(frontmatter) {
|
||||
if (!frontmatter) {
|
||||
return { hasField: false, parsed: false, sources: [], signalAvailable: false };
|
||||
}
|
||||
|
||||
const lines = frontmatter.split(/\r?\n/);
|
||||
|
||||
for (let index = 0; index < lines.length; index += 1) {
|
||||
const match = lines[index].match(/^sources:\s*(.*)$/);
|
||||
if (!match) continue;
|
||||
|
||||
const rest = match[1].trim();
|
||||
if (rest) {
|
||||
if (!rest.startsWith("[")) {
|
||||
const single = normalizeSourceToken(rest);
|
||||
return {
|
||||
hasField: true,
|
||||
parsed: Boolean(single),
|
||||
sources: single ? [single] : [],
|
||||
signalAvailable: Boolean(single)
|
||||
};
|
||||
}
|
||||
|
||||
const parsedInline = parseInlineSources(rest);
|
||||
return {
|
||||
hasField: true,
|
||||
parsed: parsedInline.ok,
|
||||
sources: parsedInline.ok ? sortedUnique(parsedInline.values) : [],
|
||||
signalAvailable: parsedInline.ok && parsedInline.values.length > 0
|
||||
};
|
||||
}
|
||||
|
||||
const collected = [];
|
||||
let parsed = true;
|
||||
let consumed = 0;
|
||||
|
||||
for (let cursor = index + 1; cursor < lines.length; cursor += 1) {
|
||||
const line = lines[cursor];
|
||||
if (!line.trim()) {
|
||||
consumed += 1;
|
||||
continue;
|
||||
}
|
||||
if (/^[^\s-]/.test(line)) break;
|
||||
const itemMatch = line.match(/^\s*-\s*(.+)$/);
|
||||
if (!itemMatch) {
|
||||
parsed = false;
|
||||
consumed += 1;
|
||||
continue;
|
||||
}
|
||||
const token = normalizeSourceToken(itemMatch[1]);
|
||||
if (token) collected.push(token);
|
||||
consumed += 1;
|
||||
}
|
||||
|
||||
index += consumed;
|
||||
return {
|
||||
hasField: true,
|
||||
parsed,
|
||||
sources: parsed ? sortedUnique(collected) : [],
|
||||
signalAvailable: parsed && collected.length > 0
|
||||
};
|
||||
}
|
||||
|
||||
return { hasField: false, parsed: false, sources: [], signalAvailable: false };
|
||||
}
|
||||
|
||||
function evaluateSourceSignalEligibility({ pageType, frontmatter }) {
|
||||
const kind = SCAN_KINDS.find((k) => k.pageType === pageType);
|
||||
if (!kind || !kind.applicable) {
|
||||
return { eligible: false, reason: "not_applicable", sources: [] };
|
||||
}
|
||||
|
||||
const parsed = parseSourcesFrontmatter(frontmatter);
|
||||
|
||||
if (!parsed.hasField) {
|
||||
return { eligible: false, reason: "missing_sources", sources: [] };
|
||||
}
|
||||
if (!parsed.parsed) {
|
||||
return { eligible: false, reason: "invalid_sources", sources: [] };
|
||||
}
|
||||
if (parsed.sources.length === 0) {
|
||||
return { eligible: false, reason: "empty_sources", sources: [] };
|
||||
}
|
||||
|
||||
return { eligible: true, reason: "ok", sources: parsed.sources };
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
SCAN_KINDS,
|
||||
extractFrontmatter,
|
||||
evaluateSourceSignalEligibility,
|
||||
normalizeSourceToken,
|
||||
parseSourcesFrontmatter,
|
||||
sortedUnique
|
||||
};
|
||||
68
llm-wiki/scripts/lib/unicode-case-folding.js
Normal file
68
llm-wiki/scripts/lib/unicode-case-folding.js
Normal file
@@ -0,0 +1,68 @@
|
||||
#!/usr/bin/env node
|
||||
"use strict";
|
||||
|
||||
const crypto = require("node:crypto");
|
||||
const fs = require("node:fs");
|
||||
const path = require("node:path");
|
||||
const { loadUnicode17NfcNormalizer } = require("./unicode-normalization");
|
||||
|
||||
const TABLE_PATH = path.join(__dirname, "../../deps/unicode/CaseFolding-17.0.0.txt");
|
||||
const EXPECTED_HASH = "ff8d8fefbf123574205085d6714c36149eb946d717a0c585c27f0f4ef58c4183";
|
||||
|
||||
let cached = null;
|
||||
|
||||
function sha256(buffer) {
|
||||
return crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
}
|
||||
|
||||
function parseUnicode17CaseFolding(text, normalizeNfc = loadUnicode17NfcNormalizer()) {
|
||||
const mappings = new Map();
|
||||
|
||||
for (const rawLine of text.split(/\r?\n/)) {
|
||||
const line = rawLine.replace(/#.*/, "").trim();
|
||||
if (!line) continue;
|
||||
|
||||
const [sourceHex, status, targetHex] = line.split(";").map((part) => part.trim());
|
||||
if (status !== "C" && status !== "F") continue;
|
||||
|
||||
mappings.set(
|
||||
Number.parseInt(sourceHex, 16),
|
||||
targetHex
|
||||
.split(/\s+/)
|
||||
.filter(Boolean)
|
||||
.map((value) => String.fromCodePoint(Number.parseInt(value, 16)))
|
||||
.join("")
|
||||
);
|
||||
}
|
||||
|
||||
return (value) => {
|
||||
let folded = "";
|
||||
for (const character of normalizeNfc(String(value))) {
|
||||
folded += mappings.get(character.codePointAt(0)) || character;
|
||||
}
|
||||
return normalizeNfc(folded);
|
||||
};
|
||||
}
|
||||
|
||||
function loadUnicode17CaseFolder() {
|
||||
if (cached) return cached;
|
||||
|
||||
const content = fs.readFileSync(TABLE_PATH);
|
||||
if (sha256(content) !== EXPECTED_HASH) {
|
||||
throw new Error(`Unicode runtime data hash mismatch for ${path.basename(TABLE_PATH)}`);
|
||||
}
|
||||
|
||||
cached = parseUnicode17CaseFolding(content.toString("utf8"));
|
||||
return cached;
|
||||
}
|
||||
|
||||
function defaultCaseFoldUnicode17(value) {
|
||||
return loadUnicode17CaseFolder()(value);
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
TABLE_PATH,
|
||||
defaultCaseFoldUnicode17,
|
||||
loadUnicode17CaseFolder,
|
||||
parseUnicode17CaseFolding
|
||||
};
|
||||
362
llm-wiki/scripts/lib/unicode-normalization.js
Normal file
362
llm-wiki/scripts/lib/unicode-normalization.js
Normal file
@@ -0,0 +1,362 @@
|
||||
#!/usr/bin/env node
|
||||
"use strict";
|
||||
|
||||
const crypto = require("node:crypto");
|
||||
const fs = require("node:fs");
|
||||
const path = require("node:path");
|
||||
|
||||
const UNICODE_DATA_PATH = path.join(__dirname, "../../deps/unicode/UnicodeData-17.0.0.txt");
|
||||
const DERIVED_NORMALIZATION_PROPS_PATH = path.join(
|
||||
__dirname,
|
||||
"../../deps/unicode/DerivedNormalizationProps-17.0.0.txt"
|
||||
);
|
||||
|
||||
const EXPECTED_HASHES = {
|
||||
[UNICODE_DATA_PATH]: "2e1efc1dcb59c575eedf5ccae60f95229f706ee6d031835247d843c11d96470c",
|
||||
[DERIVED_NORMALIZATION_PROPS_PATH]: "71fd6a206a2c0cdd41feb6b7f656aa31091db45e9cedc926985d718397f9e488"
|
||||
};
|
||||
|
||||
const HANGUL = {
|
||||
SBase: 0xac00,
|
||||
LBase: 0x1100,
|
||||
VBase: 0x1161,
|
||||
TBase: 0x11a7,
|
||||
LCount: 19,
|
||||
VCount: 21,
|
||||
TCount: 28
|
||||
};
|
||||
HANGUL.NCount = HANGUL.VCount * HANGUL.TCount;
|
||||
HANGUL.SCount = HANGUL.LCount * HANGUL.NCount;
|
||||
|
||||
let cachedTables = null;
|
||||
let cachedNormalizer = null;
|
||||
|
||||
function sha256(buffer) {
|
||||
return crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
}
|
||||
|
||||
function verifyRuntimeFile(filePath) {
|
||||
const actualHash = sha256(fs.readFileSync(filePath));
|
||||
const expectedHash = EXPECTED_HASHES[filePath];
|
||||
|
||||
if (actualHash !== expectedHash) {
|
||||
throw new Error(`Unicode runtime data hash mismatch for ${path.basename(filePath)}`);
|
||||
}
|
||||
}
|
||||
|
||||
function parseCodePointRange(rangeText) {
|
||||
const [startHex, endHex] = rangeText.split("..");
|
||||
return {
|
||||
start: Number.parseInt(startHex, 16),
|
||||
end: Number.parseInt(endHex || startHex, 16)
|
||||
};
|
||||
}
|
||||
|
||||
function expandRange(start, end, callback) {
|
||||
for (let codePoint = start; codePoint <= end; codePoint += 1) {
|
||||
callback(codePoint);
|
||||
}
|
||||
}
|
||||
|
||||
function parseDecomposition(rawField) {
|
||||
if (!rawField) return null;
|
||||
if (rawField.startsWith("<")) return null;
|
||||
return rawField.split(/\s+/).filter(Boolean).map((value) => Number.parseInt(value, 16));
|
||||
}
|
||||
|
||||
function pairKey(left, right) {
|
||||
return `${left}:${right}`;
|
||||
}
|
||||
|
||||
function lookupRangeValue(ranges, codePoint) {
|
||||
let low = 0;
|
||||
let high = ranges.length - 1;
|
||||
|
||||
while (low <= high) {
|
||||
const middle = Math.floor((low + high) / 2);
|
||||
const entry = ranges[middle];
|
||||
|
||||
if (codePoint < entry.start) {
|
||||
high = middle - 1;
|
||||
} else if (codePoint > entry.end) {
|
||||
low = middle + 1;
|
||||
} else {
|
||||
return entry.value;
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
function getCanonicalCombiningClass(tables, codePoint) {
|
||||
return tables.combiningClasses.get(codePoint) || lookupRangeValue(tables.combiningClassRanges, codePoint);
|
||||
}
|
||||
|
||||
function isHangulSyllable(codePoint) {
|
||||
return codePoint >= HANGUL.SBase && codePoint < HANGUL.SBase + HANGUL.SCount;
|
||||
}
|
||||
|
||||
function isHangulL(codePoint) {
|
||||
return codePoint >= HANGUL.LBase && codePoint < HANGUL.LBase + HANGUL.LCount;
|
||||
}
|
||||
|
||||
function isHangulV(codePoint) {
|
||||
return codePoint >= HANGUL.VBase && codePoint < HANGUL.VBase + HANGUL.VCount;
|
||||
}
|
||||
|
||||
function isHangulT(codePoint) {
|
||||
return codePoint > HANGUL.TBase && codePoint < HANGUL.TBase + HANGUL.TCount;
|
||||
}
|
||||
|
||||
function decomposeHangul(codePoint) {
|
||||
const sIndex = codePoint - HANGUL.SBase;
|
||||
const lIndex = Math.floor(sIndex / HANGUL.NCount);
|
||||
const vIndex = Math.floor((sIndex % HANGUL.NCount) / HANGUL.TCount);
|
||||
const tIndex = sIndex % HANGUL.TCount;
|
||||
|
||||
const result = [
|
||||
HANGUL.LBase + lIndex,
|
||||
HANGUL.VBase + vIndex
|
||||
];
|
||||
|
||||
if (tIndex !== 0) {
|
||||
result.push(HANGUL.TBase + tIndex);
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
function composeHangul(left, right) {
|
||||
if (isHangulL(left) && isHangulV(right)) {
|
||||
const lIndex = left - HANGUL.LBase;
|
||||
const vIndex = right - HANGUL.VBase;
|
||||
return HANGUL.SBase + (lIndex * HANGUL.NCount) + (vIndex * HANGUL.TCount);
|
||||
}
|
||||
|
||||
if (
|
||||
isHangulSyllable(left)
|
||||
&& ((left - HANGUL.SBase) % HANGUL.TCount === 0)
|
||||
&& isHangulT(right)
|
||||
) {
|
||||
return left + (right - HANGUL.TBase);
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function parseUnicode17NormalizationData(unicodeDataText, derivedPropsText) {
|
||||
const combiningClasses = new Map();
|
||||
const combiningClassRanges = [];
|
||||
const canonicalDecompositions = new Map();
|
||||
const compositionExclusions = new Set();
|
||||
const compositionMap = new Map();
|
||||
|
||||
let pendingRange = null;
|
||||
|
||||
for (const rawLine of unicodeDataText.split(/\r?\n/)) {
|
||||
if (!rawLine) continue;
|
||||
const fields = rawLine.split(";");
|
||||
if (fields.length < 6) continue;
|
||||
|
||||
const codePoint = Number.parseInt(fields[0], 16);
|
||||
const name = fields[1];
|
||||
const canonicalCombiningClass = Number.parseInt(fields[3], 10) || 0;
|
||||
const decomposition = parseDecomposition(fields[5]);
|
||||
|
||||
if (name.endsWith(", First>")) {
|
||||
pendingRange = {
|
||||
start: codePoint,
|
||||
combiningClass: canonicalCombiningClass,
|
||||
decomposition
|
||||
};
|
||||
continue;
|
||||
}
|
||||
|
||||
if (name.endsWith(", Last>") && pendingRange) {
|
||||
if (pendingRange.combiningClass !== 0) {
|
||||
combiningClassRanges.push({
|
||||
start: pendingRange.start,
|
||||
end: codePoint,
|
||||
value: pendingRange.combiningClass
|
||||
});
|
||||
}
|
||||
|
||||
if (pendingRange.decomposition) {
|
||||
expandRange(pendingRange.start, codePoint, (rangeCodePoint) => {
|
||||
canonicalDecompositions.set(rangeCodePoint, pendingRange.decomposition);
|
||||
});
|
||||
}
|
||||
|
||||
pendingRange = null;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (canonicalCombiningClass !== 0) {
|
||||
combiningClasses.set(codePoint, canonicalCombiningClass);
|
||||
}
|
||||
if (decomposition) {
|
||||
canonicalDecompositions.set(codePoint, decomposition);
|
||||
}
|
||||
}
|
||||
|
||||
combiningClassRanges.sort((left, right) => left.start - right.start);
|
||||
|
||||
for (const rawLine of derivedPropsText.split(/\r?\n/)) {
|
||||
const line = rawLine.replace(/#.*/, "").trim();
|
||||
if (!line) continue;
|
||||
|
||||
const [rangeText, property] = line.split(";").map((part) => part.trim());
|
||||
if (property !== "Full_Composition_Exclusion") continue;
|
||||
|
||||
const { start, end } = parseCodePointRange(rangeText);
|
||||
expandRange(start, end, (codePoint) => {
|
||||
compositionExclusions.add(codePoint);
|
||||
});
|
||||
}
|
||||
|
||||
for (const [composite, decomposition] of canonicalDecompositions.entries()) {
|
||||
if (compositionExclusions.has(composite)) continue;
|
||||
if (decomposition.length !== 2) continue;
|
||||
compositionMap.set(pairKey(decomposition[0], decomposition[1]), composite);
|
||||
}
|
||||
|
||||
return Object.freeze({
|
||||
combiningClasses,
|
||||
combiningClassRanges,
|
||||
canonicalDecompositions,
|
||||
compositionExclusions,
|
||||
compositionMap
|
||||
});
|
||||
}
|
||||
|
||||
function recursivelyDecompose(codePoint, tables, output) {
|
||||
if (isHangulSyllable(codePoint)) {
|
||||
for (const part of decomposeHangul(codePoint)) {
|
||||
recursivelyDecompose(part, tables, output);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
const decomposition = tables.canonicalDecompositions.get(codePoint);
|
||||
if (!decomposition) {
|
||||
output.push(codePoint);
|
||||
return;
|
||||
}
|
||||
|
||||
for (const part of decomposition) {
|
||||
recursivelyDecompose(part, tables, output);
|
||||
}
|
||||
}
|
||||
|
||||
function reorderSegment(segment, combiningClasses) {
|
||||
if (segment.length <= 1) return segment;
|
||||
|
||||
const starterCount = combiningClasses[0] === 0 ? 1 : 0;
|
||||
const head = segment.slice(0, starterCount);
|
||||
const marks = segment.slice(starterCount).map((codePoint, index) => ({
|
||||
codePoint,
|
||||
ccc: combiningClasses[starterCount + index],
|
||||
index
|
||||
}));
|
||||
|
||||
marks.sort((left, right) => {
|
||||
if (left.ccc !== right.ccc) return left.ccc - right.ccc;
|
||||
return left.index - right.index;
|
||||
});
|
||||
|
||||
return head.concat(marks.map((item) => item.codePoint));
|
||||
}
|
||||
|
||||
function canonicalOrder(codePoints, tables) {
|
||||
const ordered = [];
|
||||
let segment = [];
|
||||
let classes = [];
|
||||
|
||||
function flush() {
|
||||
if (segment.length === 0) return;
|
||||
ordered.push(...reorderSegment(segment, classes));
|
||||
segment = [];
|
||||
classes = [];
|
||||
}
|
||||
|
||||
for (const codePoint of codePoints) {
|
||||
const ccc = getCanonicalCombiningClass(tables, codePoint);
|
||||
if (ccc === 0 && segment.length > 0) {
|
||||
flush();
|
||||
}
|
||||
segment.push(codePoint);
|
||||
classes.push(ccc);
|
||||
}
|
||||
|
||||
flush();
|
||||
return ordered;
|
||||
}
|
||||
|
||||
function recompose(codePoints, tables) {
|
||||
if (codePoints.length === 0) return [];
|
||||
|
||||
const result = [codePoints[0]];
|
||||
let starterIndex = getCanonicalCombiningClass(tables, codePoints[0]) === 0 ? 0 : -1;
|
||||
let starter = starterIndex === 0 ? codePoints[0] : null;
|
||||
let lastCombiningClass = getCanonicalCombiningClass(tables, codePoints[0]);
|
||||
|
||||
for (let index = 1; index < codePoints.length; index += 1) {
|
||||
const codePoint = codePoints[index];
|
||||
const combiningClass = getCanonicalCombiningClass(tables, codePoint);
|
||||
let composite = null;
|
||||
|
||||
if (starter !== null) {
|
||||
composite = composeHangul(starter, codePoint) || tables.compositionMap.get(pairKey(starter, codePoint)) || null;
|
||||
}
|
||||
|
||||
if (composite !== null && (lastCombiningClass < combiningClass || lastCombiningClass === 0)) {
|
||||
result[starterIndex] = composite;
|
||||
starter = composite;
|
||||
continue;
|
||||
}
|
||||
|
||||
result.push(codePoint);
|
||||
lastCombiningClass = combiningClass;
|
||||
|
||||
if (combiningClass === 0) {
|
||||
starterIndex = result.length - 1;
|
||||
starter = codePoint;
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
function normalizeNfcUnicode17(value, tables) {
|
||||
const input = String(value);
|
||||
const decomposed = [];
|
||||
|
||||
for (const character of input) {
|
||||
recursivelyDecompose(character.codePointAt(0), tables, decomposed);
|
||||
}
|
||||
|
||||
const ordered = canonicalOrder(decomposed, tables);
|
||||
return String.fromCodePoint(...recompose(ordered, tables));
|
||||
}
|
||||
|
||||
function loadUnicode17NfcNormalizer() {
|
||||
if (cachedNormalizer) return cachedNormalizer;
|
||||
|
||||
verifyRuntimeFile(UNICODE_DATA_PATH);
|
||||
verifyRuntimeFile(DERIVED_NORMALIZATION_PROPS_PATH);
|
||||
|
||||
cachedTables ||= parseUnicode17NormalizationData(
|
||||
fs.readFileSync(UNICODE_DATA_PATH, "utf8"),
|
||||
fs.readFileSync(DERIVED_NORMALIZATION_PROPS_PATH, "utf8")
|
||||
);
|
||||
cachedNormalizer = (value) => normalizeNfcUnicode17(value, cachedTables);
|
||||
return cachedNormalizer;
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
DERIVED_NORMALIZATION_PROPS_PATH,
|
||||
UNICODE_DATA_PATH,
|
||||
loadUnicode17NfcNormalizer,
|
||||
normalizeNfcUnicode17,
|
||||
parseUnicode17NormalizationData
|
||||
};
|
||||
204
llm-wiki/scripts/lib/wiki-file-discovery.js
Normal file
204
llm-wiki/scripts/lib/wiki-file-discovery.js
Normal file
@@ -0,0 +1,204 @@
|
||||
#!/usr/bin/env node
|
||||
"use strict";
|
||||
|
||||
const crypto = require("node:crypto");
|
||||
const fs = require("node:fs");
|
||||
const path = require("node:path");
|
||||
|
||||
const GRAPH_PAGE_TYPES = Object.freeze({
|
||||
entities: "entity",
|
||||
topics: "topic",
|
||||
sources: "source",
|
||||
comparisons: "comparison",
|
||||
synthesis: "synthesis",
|
||||
queries: "query"
|
||||
});
|
||||
|
||||
const ROOT_EDITABLE_MARKDOWN = new Set(["index.md", "log.md", "purpose.md"]);
|
||||
const EXCLUDED_DIRECTORY_NAMES = new Set([".obsidian", ".git", ".wiki-tmp", "node_modules"]);
|
||||
const EXCLUDED_BASENAMES = new Set(["graph-data.json", "graph-warnings.json"]);
|
||||
|
||||
function sha256(text) {
|
||||
return crypto.createHash("sha256").update(text).digest("hex");
|
||||
}
|
||||
|
||||
function normalizeRelativePosixPath(pathValue) {
|
||||
const value = String(pathValue || "").replaceAll("\\", "/");
|
||||
if (!value) {
|
||||
throw new Error("Path must not be empty");
|
||||
}
|
||||
if (value.startsWith("/")) {
|
||||
throw new Error("Path must be relative");
|
||||
}
|
||||
|
||||
const normalized = path.posix.normalize(value);
|
||||
if (
|
||||
normalized === "."
|
||||
|| normalized === ".."
|
||||
|| normalized.startsWith("../")
|
||||
|| normalized.includes("/../")
|
||||
) {
|
||||
throw new Error("Path escapes knowledge-base root");
|
||||
}
|
||||
|
||||
return normalized.replace(/^\.\//, "");
|
||||
}
|
||||
|
||||
function isWithinRoot(rootRealPath, candidateAbsolutePath) {
|
||||
return candidateAbsolutePath === rootRealPath || candidateAbsolutePath.startsWith(`${rootRealPath}${path.sep}`);
|
||||
}
|
||||
|
||||
function resolveInsideKnowledgeBase(kbRoot, relativePath) {
|
||||
const rootRealPath = fs.realpathSync.native(kbRoot);
|
||||
const normalized = normalizeRelativePosixPath(relativePath);
|
||||
const absolutePath = path.resolve(rootRealPath, ...normalized.split("/"));
|
||||
|
||||
if (!isWithinRoot(rootRealPath, absolutePath)) {
|
||||
throw new Error(`Path escapes knowledge-base root: ${relativePath}`);
|
||||
}
|
||||
|
||||
return absolutePath;
|
||||
}
|
||||
|
||||
function isGeneratedArtifact(relativePath) {
|
||||
const basename = path.posix.basename(relativePath);
|
||||
if (EXCLUDED_BASENAMES.has(basename)) return true;
|
||||
return /^knowledge-graph(?:[^/]*)\.html$/i.test(basename);
|
||||
}
|
||||
|
||||
function isRenameStagingFile(relativePath) {
|
||||
return path.posix.basename(relativePath).startsWith(".llm-wiki-rename-");
|
||||
}
|
||||
|
||||
function isMarkdown(relativePath) {
|
||||
return relativePath.toLowerCase().endsWith(".md");
|
||||
}
|
||||
|
||||
function isAttachment(relativePath) {
|
||||
return !isMarkdown(relativePath);
|
||||
}
|
||||
|
||||
function graphTypeFor(relativePath) {
|
||||
const parts = relativePath.split("/");
|
||||
return parts[0] === "wiki" ? (GRAPH_PAGE_TYPES[parts[1]] || null) : null;
|
||||
}
|
||||
|
||||
function isGraphSource(relativePath) {
|
||||
return isMarkdown(relativePath) && graphTypeFor(relativePath) !== null;
|
||||
}
|
||||
|
||||
function isLintSource(relativePath) {
|
||||
return relativePath === "index.md" || (relativePath.startsWith("wiki/") && isMarkdown(relativePath));
|
||||
}
|
||||
|
||||
function isRenameEditableSource(relativePath) {
|
||||
return ROOT_EDITABLE_MARKDOWN.has(relativePath) || (relativePath.startsWith("wiki/") && isMarkdown(relativePath));
|
||||
}
|
||||
|
||||
function isRenameReadOnlySource(relativePath) {
|
||||
return isMarkdown(relativePath) && !isRenameEditableSource(relativePath);
|
||||
}
|
||||
|
||||
function fileSetSignature(items) {
|
||||
return sha256(items.map((item) => `${item.path}\0${item.kind}\0${item.size}\0${item.mtimeNs}`).join("\n"));
|
||||
}
|
||||
|
||||
function fileKindFor(relativePath) {
|
||||
return isMarkdown(relativePath) ? "markdown" : "attachment";
|
||||
}
|
||||
|
||||
function walkKnowledgeBase(rootRealPath, directoryAbsolutePath, results) {
|
||||
const entries = fs.readdirSync(directoryAbsolutePath, { withFileTypes: true })
|
||||
.sort((left, right) => left.name.localeCompare(right.name, "en"));
|
||||
|
||||
for (const entry of entries) {
|
||||
const absolutePath = path.join(directoryAbsolutePath, entry.name);
|
||||
const relativePath = path.relative(rootRealPath, absolutePath).split(path.sep).join("/");
|
||||
const topLevelName = relativePath.split("/")[0];
|
||||
|
||||
if (entry.isDirectory()) {
|
||||
if (
|
||||
entry.name.startsWith(".")
|
||||
|| EXCLUDED_DIRECTORY_NAMES.has(entry.name)
|
||||
|| EXCLUDED_DIRECTORY_NAMES.has(topLevelName)
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
if (entry.isSymbolicLink && entry.isSymbolicLink()) {
|
||||
continue;
|
||||
}
|
||||
const resolvedDirectoryPath = fs.realpathSync.native(absolutePath);
|
||||
if (!isWithinRoot(rootRealPath, resolvedDirectoryPath)) {
|
||||
continue;
|
||||
}
|
||||
walkKnowledgeBase(rootRealPath, absolutePath, results);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (entry.isSymbolicLink && entry.isSymbolicLink()) {
|
||||
continue;
|
||||
}
|
||||
if (!entry.isFile()) {
|
||||
continue;
|
||||
}
|
||||
if (EXCLUDED_DIRECTORY_NAMES.has(topLevelName)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const normalizedPath = normalizeRelativePosixPath(relativePath);
|
||||
if (entry.name.startsWith(".") && normalizedPath !== ".wiki-schema.md") {
|
||||
continue;
|
||||
}
|
||||
if (isGeneratedArtifact(normalizedPath) || isRenameStagingFile(normalizedPath)) {
|
||||
continue;
|
||||
}
|
||||
if (!isMarkdown(normalizedPath) && !isAttachment(normalizedPath)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const resolvedFilePath = fs.realpathSync.native(absolutePath);
|
||||
if (!isWithinRoot(rootRealPath, resolvedFilePath)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const stat = fs.statSync(absolutePath, { bigint: true });
|
||||
results.push({
|
||||
path: normalizedPath,
|
||||
absolutePath,
|
||||
kind: fileKindFor(normalizedPath),
|
||||
editable: isRenameEditableSource(normalizedPath),
|
||||
graphType: graphTypeFor(normalizedPath),
|
||||
size: Number(stat.size),
|
||||
mtimeNs: String(stat.mtimeNs)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
function discoverKnowledgeBaseFiles(kbRoot) {
|
||||
const rootRealPath = fs.realpathSync.native(kbRoot);
|
||||
const inventory = [];
|
||||
walkKnowledgeBase(rootRealPath, rootRealPath, inventory);
|
||||
inventory.sort((left, right) => left.path.localeCompare(right.path, "en"));
|
||||
|
||||
const graphSources = inventory.filter((item) => item.kind === "markdown" && isGraphSource(item.path));
|
||||
const lintSources = inventory.filter((item) => item.kind === "markdown" && isLintSource(item.path));
|
||||
const renameEditableSources = inventory.filter((item) => item.kind === "markdown" && isRenameEditableSource(item.path));
|
||||
const renameReadOnlySources = inventory.filter((item) => item.kind === "markdown" && isRenameReadOnlySource(item.path));
|
||||
const targets = inventory.map(({ size, mtimeNs, ...item }) => item);
|
||||
|
||||
return {
|
||||
graphSources,
|
||||
lintSources,
|
||||
renameEditableSources,
|
||||
renameReadOnlySources,
|
||||
targets,
|
||||
fileSetSha256: fileSetSignature(inventory)
|
||||
};
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
GRAPH_PAGE_TYPES,
|
||||
discoverKnowledgeBaseFiles,
|
||||
normalizeRelativePosixPath,
|
||||
resolveInsideKnowledgeBase
|
||||
};
|
||||
452
llm-wiki/scripts/lib/wiki-link-index.js
Normal file
452
llm-wiki/scripts/lib/wiki-link-index.js
Normal file
@@ -0,0 +1,452 @@
|
||||
#!/usr/bin/env node
|
||||
"use strict";
|
||||
|
||||
const crypto = require("node:crypto");
|
||||
const fs = require("node:fs");
|
||||
const path = require("node:path");
|
||||
const { loadUnicode17CaseFolder } = require("./unicode-case-folding");
|
||||
const {
|
||||
validateGraphRenameFilenameSyntax
|
||||
} = require("../../packages/workbench-contracts/src/graph-rename-filename.js");
|
||||
const {
|
||||
discoverKnowledgeBaseFiles,
|
||||
normalizeRelativePosixPath
|
||||
} = require("./wiki-file-discovery");
|
||||
const { extractFrontmatter, parseSourcesFrontmatter } = require("./source-signal-eligibility");
|
||||
const { parseWikilinks, renderWikilinkReplacement } = require("./wikilink-parser");
|
||||
|
||||
function sha256(text) {
|
||||
return crypto.createHash("sha256").update(text).digest("hex");
|
||||
}
|
||||
|
||||
function stableId(prefix, value) {
|
||||
return `${prefix}-${sha256(value).slice(0, 16)}`;
|
||||
}
|
||||
|
||||
function portablePathKey(pathValue, fold = loadUnicode17CaseFolder()) {
|
||||
return fold(normalizeRelativePosixPath(pathValue));
|
||||
}
|
||||
|
||||
function buildWikiTargetIndex(inventory) {
|
||||
const exactPaths = new Map();
|
||||
const portablePaths = new Map();
|
||||
const portableBasenames = new Map();
|
||||
|
||||
for (const item of inventory) {
|
||||
exactPaths.set(item.path, item);
|
||||
|
||||
const pathKey = portablePathKey(item.path);
|
||||
const pathItems = portablePaths.get(pathKey) || [];
|
||||
pathItems.push(item);
|
||||
portablePaths.set(pathKey, pathItems);
|
||||
|
||||
if (item.kind === "markdown") {
|
||||
const basenameKey = portablePathKey(path.posix.basename(item.path, ".md"));
|
||||
const basenameItems = portableBasenames.get(basenameKey) || [];
|
||||
basenameItems.push(item);
|
||||
portableBasenames.set(basenameKey, basenameItems);
|
||||
}
|
||||
}
|
||||
|
||||
const portableCollisions = Array.from(portablePaths.values())
|
||||
.filter((items) => items.length > 1)
|
||||
.map((items) => {
|
||||
const candidates = items.map((item) => item.path).sort();
|
||||
return {
|
||||
collision_id: stableId("portable-collision", candidates.join("\n")),
|
||||
candidate_set_id: stableId("candidate-set", candidates.join("\n")),
|
||||
candidates
|
||||
};
|
||||
})
|
||||
.sort((left, right) => left.candidates.join("\n").localeCompare(right.candidates.join("\n"), "en"));
|
||||
|
||||
return { exactPaths, portablePaths, portableBasenames, portableCollisions };
|
||||
}
|
||||
|
||||
function normalizeExplicitTarget(target) {
|
||||
const normalized = normalizeRelativePosixPath(target);
|
||||
const extension = path.posix.extname(normalized);
|
||||
if (!extension) {
|
||||
return `${normalized}.md`;
|
||||
}
|
||||
return normalized;
|
||||
}
|
||||
|
||||
function warningSeverity(code) {
|
||||
if (code === "pending_wikilink" || code === "noncanonical_wikilink") {
|
||||
return "warning";
|
||||
}
|
||||
return "error";
|
||||
}
|
||||
|
||||
function resolveWikilink(occurrence, sourcePath, index) {
|
||||
if (occurrence.link_kind === "same_page_anchor") {
|
||||
return {
|
||||
status: "resolved",
|
||||
target_path: sourcePath,
|
||||
creates_edge: false,
|
||||
warning_code: null,
|
||||
candidate_paths: [],
|
||||
target_key: sourcePath
|
||||
};
|
||||
}
|
||||
|
||||
const rawTarget = occurrence.page_target;
|
||||
const explicitPath = rawTarget.includes("/") || occurrence.link_kind === "attachment_wikilink" || rawTarget.endsWith(".md");
|
||||
const targetKey = rawTarget ? normalizeRelativePosixPath(rawTarget) : sourcePath;
|
||||
|
||||
let candidates = [];
|
||||
let warningCode = null;
|
||||
|
||||
if (explicitPath) {
|
||||
const normalizedTarget = normalizeExplicitTarget(rawTarget);
|
||||
const exactMatch = index.exactPaths.get(normalizedTarget);
|
||||
if (exactMatch) {
|
||||
candidates = [exactMatch];
|
||||
} else {
|
||||
candidates = index.portablePaths.get(portablePathKey(normalizedTarget)) || [];
|
||||
if (candidates.length === 1) {
|
||||
warningCode = "noncanonical_wikilink";
|
||||
}
|
||||
}
|
||||
} else {
|
||||
const basename = rawTarget.replace(/\.md$/i, "");
|
||||
candidates = index.portableBasenames.get(portablePathKey(basename)) || [];
|
||||
}
|
||||
|
||||
if (candidates.length === 0) {
|
||||
return {
|
||||
status: "missing",
|
||||
target_path: null,
|
||||
creates_edge: false,
|
||||
warning_code: occurrence.pending ? "pending_wikilink" : (occurrence.link_kind === "attachment_wikilink" ? null : "broken_wikilink"),
|
||||
candidate_paths: [],
|
||||
target_key: targetKey
|
||||
};
|
||||
}
|
||||
|
||||
if (candidates.length > 1) {
|
||||
return {
|
||||
status: "ambiguous",
|
||||
target_path: null,
|
||||
creates_edge: false,
|
||||
warning_code: "ambiguous_wikilink",
|
||||
candidate_paths: candidates.map((item) => item.path).sort(),
|
||||
target_key: targetKey
|
||||
};
|
||||
}
|
||||
|
||||
const [target] = candidates;
|
||||
const createsEdge = Boolean(
|
||||
target.kind === "markdown"
|
||||
&& target.graphType
|
||||
&& target.path !== sourcePath
|
||||
&& occurrence.link_kind !== "attachment_wikilink"
|
||||
);
|
||||
|
||||
return {
|
||||
status: "resolved",
|
||||
target_path: target.path,
|
||||
creates_edge: createsEdge,
|
||||
warning_code: warningCode,
|
||||
candidate_paths: [],
|
||||
target_key: targetKey
|
||||
};
|
||||
}
|
||||
|
||||
function warningMessage(code, targetKey) {
|
||||
switch (code) {
|
||||
case "ambiguous_wikilink":
|
||||
return `Ambiguous wikilink: ${targetKey}`;
|
||||
case "broken_wikilink":
|
||||
return `Broken wikilink: ${targetKey}`;
|
||||
case "pending_wikilink":
|
||||
return `Pending wikilink: ${targetKey}`;
|
||||
case "noncanonical_wikilink":
|
||||
return `Noncanonical wikilink: ${targetKey}`;
|
||||
case "portable_path_collision":
|
||||
return "Portable path collision";
|
||||
default:
|
||||
return code;
|
||||
}
|
||||
}
|
||||
|
||||
function addWarningGroup(groupMap, candidateSetMap, code, resolution, occurrenceRecord) {
|
||||
const candidatePaths = resolution.candidate_paths || [];
|
||||
const candidateSetId = candidatePaths.length > 0
|
||||
? stableId("candidate-set", candidatePaths.join("\n"))
|
||||
: null;
|
||||
|
||||
if (candidateSetId && !candidateSetMap.has(candidateSetId)) {
|
||||
candidateSetMap.set(candidateSetId, {
|
||||
candidate_set_id: candidateSetId,
|
||||
candidate_count: candidatePaths.length,
|
||||
candidates: candidatePaths
|
||||
});
|
||||
}
|
||||
|
||||
const warningKey = `${code}\0${resolution.target_key || ""}\0${candidateSetId || ""}`;
|
||||
const warningId = stableId("warning", warningKey);
|
||||
if (!groupMap.has(warningId)) {
|
||||
groupMap.set(warningId, {
|
||||
warning_id: warningId,
|
||||
code,
|
||||
severity: warningSeverity(code),
|
||||
message: warningMessage(code, resolution.target_key || ""),
|
||||
target_key: resolution.target_key || undefined,
|
||||
candidate_set_id: candidateSetId || undefined,
|
||||
occurrence_count: 0,
|
||||
occurrences: []
|
||||
});
|
||||
}
|
||||
|
||||
const group = groupMap.get(warningId);
|
||||
group.occurrence_count += 1;
|
||||
group.occurrences.push(occurrenceRecord);
|
||||
}
|
||||
|
||||
function scanPolicySources(inventory, policy) {
|
||||
if (policy === "graph") return inventory.graphSources;
|
||||
if (policy === "lint") return inventory.lintSources;
|
||||
if (policy === "rename") {
|
||||
return inventory.renameEditableSources.concat(inventory.renameReadOnlySources)
|
||||
.sort((left, right) => left.path.localeCompare(right.path, "en"));
|
||||
}
|
||||
throw new Error(`Unknown scan policy: ${policy}`);
|
||||
}
|
||||
|
||||
function parseImagePaths(frontmatter) {
|
||||
if (!frontmatter) return [];
|
||||
const lines = frontmatter.split(/\r?\n/);
|
||||
for (let index = 0; index < lines.length; index += 1) {
|
||||
const match = lines[index].match(/^image_paths:\s*(.*)$/);
|
||||
if (!match) continue;
|
||||
const inline = match[1].trim();
|
||||
if (inline) {
|
||||
if (inline === "[]") return [];
|
||||
if (!inline.startsWith("[") || !inline.endsWith("]")) return [];
|
||||
return inline.slice(1, -1).split(",")
|
||||
.map((value) => value.trim().replace(/^['"]|['"]$/g, ""))
|
||||
.filter(Boolean);
|
||||
}
|
||||
const values = [];
|
||||
for (let cursor = index + 1; cursor < lines.length; cursor += 1) {
|
||||
if (!lines[cursor].trim()) continue;
|
||||
const item = lines[cursor].match(/^\s*-\s*(.+?)\s*$/);
|
||||
if (!item) break;
|
||||
values.push(item[1].trim().replace(/^['"]|['"]$/g, ""));
|
||||
}
|
||||
return values.filter(Boolean);
|
||||
}
|
||||
return [];
|
||||
}
|
||||
|
||||
function scanKnowledgeBaseLinks(kbRoot, policy) {
|
||||
const inventory = discoverKnowledgeBaseFiles(kbRoot);
|
||||
const index = buildWikiTargetIndex(inventory.targets);
|
||||
const sources = scanPolicySources(inventory, policy);
|
||||
const candidateSetMap = new Map();
|
||||
const groupMap = new Map();
|
||||
const occurrences = [];
|
||||
const edges = [];
|
||||
const sourceDocuments = [];
|
||||
const stalePendingWrappers = [];
|
||||
const edgeByEndpoints = new Map();
|
||||
const metrics = {
|
||||
inventory_walks: 1,
|
||||
target_index_builds: 1,
|
||||
source_files_parsed: 0,
|
||||
files_read: 0,
|
||||
files_parsed: 0,
|
||||
graph_source_bytes: 0,
|
||||
utf8_bytes_scanned: 0,
|
||||
position_bytes_advanced: 0
|
||||
};
|
||||
|
||||
for (const collision of index.portableCollisions) {
|
||||
candidateSetMap.set(collision.candidate_set_id, {
|
||||
candidate_set_id: collision.candidate_set_id,
|
||||
candidate_count: collision.candidates.length,
|
||||
candidates: collision.candidates
|
||||
});
|
||||
groupMap.set(collision.collision_id, {
|
||||
warning_id: collision.collision_id,
|
||||
code: "portable_path_collision",
|
||||
severity: "error",
|
||||
message: "Portable path collision",
|
||||
id: collision.collision_id,
|
||||
candidate_set_id: collision.candidate_set_id,
|
||||
occurrence_count: 0,
|
||||
occurrences: []
|
||||
});
|
||||
}
|
||||
|
||||
for (const source of sources) {
|
||||
const buffer = fs.readFileSync(source.absolutePath);
|
||||
const rawContent = buffer.toString("utf8");
|
||||
const frontmatter = extractFrontmatter(rawContent);
|
||||
const parsedSources = parseSourcesFrontmatter(frontmatter.frontmatter);
|
||||
const heading = frontmatter.body.match(/^#\s+(.+?)\s*$/m);
|
||||
sourceDocuments.push({
|
||||
source_path: source.path,
|
||||
graph_type: source.graphType,
|
||||
label: heading ? heading[1].trim() : path.posix.basename(source.path, ".md"),
|
||||
_content: rawContent,
|
||||
_signals: {
|
||||
sources: parsedSources.sources,
|
||||
sourceSignalAvailable: parsedSources.signalAvailable,
|
||||
sourceFieldPresent: parsedSources.hasField,
|
||||
sourceFieldParsed: parsedSources.parsed,
|
||||
imagePaths: parseImagePaths(frontmatter.frontmatter)
|
||||
}
|
||||
});
|
||||
metrics.files_read += 1;
|
||||
metrics.files_parsed += 1;
|
||||
metrics.source_files_parsed += 1;
|
||||
if (source.graphType) metrics.graph_source_bytes += buffer.length;
|
||||
|
||||
const parsed = parseWikilinks(buffer, source.path);
|
||||
metrics.utf8_bytes_scanned += parsed.metrics.utf8_bytes_scanned;
|
||||
metrics.position_bytes_advanced += parsed.metrics.position_bytes_advanced;
|
||||
for (const occurrence of parsed.occurrences) {
|
||||
const resolution = resolveWikilink(occurrence, source.path, index);
|
||||
const resolutionCandidateSetId = resolution.candidate_paths.length > 0
|
||||
? stableId("candidate-set", resolution.candidate_paths.join("\n"))
|
||||
: null;
|
||||
const occurrenceRecord = {
|
||||
occurrence_id: stableId(
|
||||
"occurrence",
|
||||
`${occurrence.source_path}\0${occurrence.file_sha256}\0${occurrence.start_byte}\0${occurrence.end_byte}\0${occurrence.raw_link}`
|
||||
),
|
||||
source_path: occurrence.source_path,
|
||||
line: occurrence.line,
|
||||
column: occurrence.column,
|
||||
start_byte: occurrence.start_byte,
|
||||
end_byte: occurrence.end_byte,
|
||||
raw_link: occurrence.raw_link,
|
||||
file_sha256: occurrence.file_sha256,
|
||||
link_kind: occurrence.link_kind,
|
||||
read_only: source.editable === false
|
||||
};
|
||||
|
||||
occurrences.push({
|
||||
...occurrence,
|
||||
read_only: source.editable === false,
|
||||
resolution: {
|
||||
...resolution,
|
||||
candidate_paths: undefined,
|
||||
candidate_set_id: resolutionCandidateSetId || undefined
|
||||
}
|
||||
});
|
||||
|
||||
if (occurrence.pending && resolution.status === "resolved") {
|
||||
stalePendingWrappers.push({
|
||||
source_path: source.path,
|
||||
raw_link: occurrence.raw_link,
|
||||
replacement: renderWikilinkReplacement(occurrence, occurrence.page_target)
|
||||
});
|
||||
} else if (resolution.warning_code) {
|
||||
addWarningGroup(groupMap, candidateSetMap, resolution.warning_code, resolution, occurrenceRecord);
|
||||
}
|
||||
|
||||
if (resolution.creates_edge && resolution.target_path) {
|
||||
const edgeKey = `${source.path}\0${resolution.target_path}`;
|
||||
if (!edgeByEndpoints.has(edgeKey)) {
|
||||
const edge = {
|
||||
from: source.path,
|
||||
to: resolution.target_path,
|
||||
relation_type: occurrence.relation_type || "依赖",
|
||||
confidence: occurrence.confidence || "EXTRACTED",
|
||||
_relation_explicit: Boolean(occurrence.relation_type),
|
||||
_confidence_explicit: Boolean(occurrence.confidence)
|
||||
};
|
||||
edgeByEndpoints.set(edgeKey, edge);
|
||||
edges.push(edge);
|
||||
} else {
|
||||
const edge = edgeByEndpoints.get(edgeKey);
|
||||
if (!edge._confidence_explicit && occurrence.confidence) {
|
||||
edge.confidence = occurrence.confidence;
|
||||
edge._confidence_explicit = true;
|
||||
}
|
||||
if (!edge._relation_explicit && occurrence.relation_type) {
|
||||
edge.relation_type = occurrence.relation_type;
|
||||
edge._relation_explicit = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const candidate_sets = Array.from(candidateSetMap.values())
|
||||
.sort((left, right) => left.candidate_set_id.localeCompare(right.candidate_set_id, "en"));
|
||||
const groups = Array.from(groupMap.values())
|
||||
.map((group) => ({
|
||||
...group,
|
||||
occurrences: group.occurrences.slice().sort((left, right) => {
|
||||
if (left.source_path !== right.source_path) {
|
||||
return left.source_path.localeCompare(right.source_path, "en");
|
||||
}
|
||||
return left.start_byte - right.start_byte;
|
||||
})
|
||||
}))
|
||||
.sort((left, right) => left.warning_id.localeCompare(right.warning_id, "en"));
|
||||
|
||||
edges.sort((left, right) => {
|
||||
const leftKey = `${left.from}\0${left.to}\0${left.relation_type}`;
|
||||
const rightKey = `${right.from}\0${right.to}\0${right.relation_type}`;
|
||||
return leftKey.localeCompare(rightKey, "en");
|
||||
});
|
||||
for (const edge of edges) {
|
||||
delete edge._confidence_explicit;
|
||||
delete edge._relation_explicit;
|
||||
}
|
||||
|
||||
stalePendingWrappers.sort((left, right) => left.source_path.localeCompare(right.source_path, "en"));
|
||||
|
||||
return {
|
||||
inventory,
|
||||
edges,
|
||||
candidate_sets,
|
||||
groups,
|
||||
occurrences: policy === "graph" ? [] : occurrences,
|
||||
source_documents: sourceDocuments,
|
||||
stale_pending_wrappers: stalePendingWrappers,
|
||||
metrics
|
||||
};
|
||||
}
|
||||
|
||||
function validatePortableMarkdownFilename(sourcePath, newName, inventoryOrTargets) {
|
||||
const targets = Array.isArray(inventoryOrTargets)
|
||||
? inventoryOrTargets
|
||||
: inventoryOrTargets.targets;
|
||||
const syntax = validateGraphRenameFilenameSyntax(String(newName ?? ""));
|
||||
if (!syntax.ok) return syntax;
|
||||
const normalizedName = syntax.normalized_name;
|
||||
|
||||
const sourceDir = path.posix.dirname(sourcePath);
|
||||
const targetPath = sourceDir === "." ? normalizedName : `${sourceDir}/${normalizedName}`;
|
||||
const targetPortableKey = portablePathKey(targetPath);
|
||||
const collisions = targets
|
||||
.filter((item) => item.path !== sourcePath && portablePathKey(item.path) === targetPortableKey)
|
||||
.map((item) => item.path)
|
||||
.sort();
|
||||
|
||||
if (collisions.length > 0) {
|
||||
return { ok: false, reason: "portable_path_collision", collision_paths: collisions };
|
||||
}
|
||||
|
||||
return {
|
||||
ok: true,
|
||||
normalized_name: normalizedName,
|
||||
target_path: targetPath,
|
||||
requires_transit: portablePathKey(sourcePath) === targetPortableKey && sourcePath !== targetPath
|
||||
};
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
buildWikiTargetIndex,
|
||||
portablePathKey,
|
||||
resolveWikilink,
|
||||
scanKnowledgeBaseLinks,
|
||||
validatePortableMarkdownFilename
|
||||
};
|
||||
305
llm-wiki/scripts/lib/wikilink-parser.js
Normal file
305
llm-wiki/scripts/lib/wikilink-parser.js
Normal file
@@ -0,0 +1,305 @@
|
||||
#!/usr/bin/env node
|
||||
"use strict";
|
||||
|
||||
const crypto = require("node:crypto");
|
||||
const path = require("node:path");
|
||||
|
||||
function sha256(buffer) {
|
||||
return crypto.createHash("sha256").update(buffer).digest("hex");
|
||||
}
|
||||
|
||||
function countCodePoints(value) {
|
||||
return Array.from(value).length;
|
||||
}
|
||||
|
||||
function extractLineAnnotations(lineText) {
|
||||
const confidenceMatch = lineText.match(/<!--\s*confidence:\s*([A-Z]+)\s*-->/);
|
||||
const relationTypeMatch = lineText.match(/<!--\s*relation(?:_type)?:\s*([^>]+?)\s*-->/);
|
||||
|
||||
return {
|
||||
confidence: confidenceMatch ? confidenceMatch[1] : null,
|
||||
relation_type: relationTypeMatch ? relationTypeMatch[1].trim() : null
|
||||
};
|
||||
}
|
||||
|
||||
function parseFenceCandidate(lineText) {
|
||||
const match = lineText.match(/^( {0,3})(`{3,}|~{3,})(.*)$/);
|
||||
if (!match) return null;
|
||||
return {
|
||||
marker: match[2][0],
|
||||
length: match[2].length,
|
||||
rest: match[3]
|
||||
};
|
||||
}
|
||||
|
||||
function collectInlineBacktickRuns(text) {
|
||||
const positionsByLength = new Map();
|
||||
let fence = null;
|
||||
let lineStartIndex = 0;
|
||||
|
||||
while (lineStartIndex <= text.length) {
|
||||
const nextNewlineIndex = text.indexOf("\n", lineStartIndex);
|
||||
const lineEndIndex = nextNewlineIndex === -1 ? text.length : nextNewlineIndex;
|
||||
const lineText = text.slice(lineStartIndex, lineEndIndex);
|
||||
const fenceCandidate = parseFenceCandidate(lineText);
|
||||
let handledAsFence = false;
|
||||
|
||||
if (fence) {
|
||||
handledAsFence = true;
|
||||
if (
|
||||
fenceCandidate
|
||||
&& fence.marker === fenceCandidate.marker
|
||||
&& fenceCandidate.length >= fence.length
|
||||
&& /^[ \t]*$/.test(fenceCandidate.rest)
|
||||
) {
|
||||
fence = null;
|
||||
}
|
||||
} else if (
|
||||
fenceCandidate
|
||||
&& !(fenceCandidate.marker === "`" && fenceCandidate.rest.includes("`"))
|
||||
) {
|
||||
fence = { marker: fenceCandidate.marker, length: fenceCandidate.length };
|
||||
handledAsFence = true;
|
||||
}
|
||||
|
||||
if (!handledAsFence) {
|
||||
for (let index = 0; index < lineText.length;) {
|
||||
if (lineText[index] !== "`") {
|
||||
index += String.fromCodePoint(lineText.codePointAt(index)).length;
|
||||
continue;
|
||||
}
|
||||
|
||||
let runLength = 1;
|
||||
while (lineText[index + runLength] === "`") {
|
||||
runLength += 1;
|
||||
}
|
||||
const positions = positionsByLength.get(runLength) || [];
|
||||
positions.push(lineStartIndex + index);
|
||||
positionsByLength.set(runLength, positions);
|
||||
index += runLength;
|
||||
}
|
||||
}
|
||||
|
||||
if (nextNewlineIndex === -1) break;
|
||||
lineStartIndex = nextNewlineIndex + 1;
|
||||
}
|
||||
|
||||
return { positionsByLength, cursorsByLength: new Map() };
|
||||
}
|
||||
|
||||
function consumeInlineBacktickRun(runIndex, runLength, absoluteIndex) {
|
||||
const positions = runIndex.positionsByLength.get(runLength) || [];
|
||||
let cursor = runIndex.cursorsByLength.get(runLength) || 0;
|
||||
while (cursor < positions.length && positions[cursor] <= absoluteIndex) {
|
||||
cursor += 1;
|
||||
}
|
||||
runIndex.cursorsByLength.set(runLength, cursor);
|
||||
return cursor < positions.length;
|
||||
}
|
||||
|
||||
function parseOccurrence(rawLink, sourcePath, fileSha256, line, column, startByte, annotations) {
|
||||
let innerStart = 0;
|
||||
let innerEnd = rawLink.length;
|
||||
let embedded = false;
|
||||
let pending = false;
|
||||
|
||||
if (rawLink.startsWith("[待创建: [[")) {
|
||||
innerStart = "[待创建: [[".length;
|
||||
innerEnd = rawLink.length - "]]]".length;
|
||||
pending = true;
|
||||
} else if (rawLink.startsWith("[To create: [[")) {
|
||||
innerStart = "[To create: [[".length;
|
||||
innerEnd = rawLink.length - "]]]".length;
|
||||
pending = true;
|
||||
} else if (rawLink.startsWith("![[")) {
|
||||
innerStart = "![[".length;
|
||||
innerEnd = rawLink.length - "]]".length;
|
||||
embedded = true;
|
||||
} else {
|
||||
innerStart = "[[".length;
|
||||
innerEnd = rawLink.length - "]]".length;
|
||||
}
|
||||
|
||||
const inner = rawLink.slice(innerStart, innerEnd);
|
||||
const pipeIndex = inner.indexOf("|");
|
||||
const targetAndAnchorRaw = pipeIndex >= 0 ? inner.slice(0, pipeIndex) : inner;
|
||||
const displayRaw = pipeIndex >= 0 ? inner.slice(pipeIndex + 1) : null;
|
||||
const anchorIndex = targetAndAnchorRaw.indexOf("#");
|
||||
const targetRaw = anchorIndex >= 0 ? targetAndAnchorRaw.slice(0, anchorIndex) : targetAndAnchorRaw;
|
||||
const anchor = anchorIndex >= 0 ? targetAndAnchorRaw.slice(anchorIndex + 1).trim() : null;
|
||||
const display = displayRaw === null ? null : displayRaw.trim();
|
||||
|
||||
const leadingWhitespace = (targetRaw.match(/^\s*/) || [""])[0].length;
|
||||
const trailingWhitespace = (targetRaw.match(/\s*$/) || [""])[0].length;
|
||||
const targetStartInRaw = innerStart + leadingWhitespace;
|
||||
const targetEndInRaw = innerStart + targetRaw.length - trailingWhitespace;
|
||||
const pageTarget = targetRaw.trim();
|
||||
const extension = path.posix.extname(pageTarget);
|
||||
|
||||
let linkKind = "page_wikilink";
|
||||
if (pageTarget === "" && anchor) {
|
||||
linkKind = "same_page_anchor";
|
||||
} else if (extension && extension.toLowerCase() !== ".md") {
|
||||
linkKind = "attachment_wikilink";
|
||||
}
|
||||
|
||||
return {
|
||||
occurrence_id: `${sourcePath}\0${fileSha256}\0${startByte}\0${startByte + Buffer.byteLength(rawLink, "utf8")}\0${rawLink}`,
|
||||
source_path: sourcePath,
|
||||
file_sha256: fileSha256,
|
||||
raw_link: rawLink,
|
||||
line,
|
||||
column,
|
||||
start_byte: startByte,
|
||||
end_byte: startByte + Buffer.byteLength(rawLink, "utf8"),
|
||||
link_kind: linkKind,
|
||||
embedded,
|
||||
pending,
|
||||
page_target: pageTarget,
|
||||
anchor,
|
||||
display,
|
||||
confidence: annotations.confidence,
|
||||
relation_type: annotations.relation_type,
|
||||
target_start_in_raw: targetStartInRaw,
|
||||
target_end_in_raw: targetEndInRaw
|
||||
};
|
||||
}
|
||||
|
||||
function parseWikilinks(buffer, sourcePath) {
|
||||
const text = buffer.toString("utf8");
|
||||
const fileSha256 = sha256(buffer);
|
||||
const occurrences = [];
|
||||
const inlineBacktickRuns = collectInlineBacktickRuns(text);
|
||||
|
||||
let fence = null;
|
||||
let inlineDelimiter = 0;
|
||||
let lineNumber = 1;
|
||||
let lineStartIndex = 0;
|
||||
let positionBytesAdvanced = 0;
|
||||
|
||||
while (lineStartIndex <= text.length) {
|
||||
const nextNewlineIndex = text.indexOf("\n", lineStartIndex);
|
||||
const lineEndIndex = nextNewlineIndex === -1 ? text.length : nextNewlineIndex;
|
||||
const lineText = text.slice(lineStartIndex, lineEndIndex);
|
||||
const annotations = extractLineAnnotations(lineText);
|
||||
const fenceCandidate = parseFenceCandidate(lineText);
|
||||
let handledAsFence = false;
|
||||
|
||||
if (fence) {
|
||||
handledAsFence = true;
|
||||
if (
|
||||
fenceCandidate
|
||||
&& fence.marker === fenceCandidate.marker
|
||||
&& fenceCandidate.length >= fence.length
|
||||
&& /^[ \t]*$/.test(fenceCandidate.rest)
|
||||
) {
|
||||
fence = null;
|
||||
}
|
||||
} else if (
|
||||
inlineDelimiter === 0
|
||||
&& fenceCandidate
|
||||
&& !(fenceCandidate.marker === "`" && fenceCandidate.rest.includes("`"))
|
||||
) {
|
||||
fence = { marker: fenceCandidate.marker, length: fenceCandidate.length };
|
||||
handledAsFence = true;
|
||||
}
|
||||
|
||||
if (handledAsFence) {
|
||||
positionBytesAdvanced += Buffer.byteLength(lineText, "utf8");
|
||||
} else {
|
||||
let column = 1;
|
||||
|
||||
for (let index = 0; index < lineText.length;) {
|
||||
if (lineText[index] === "`") {
|
||||
let runLength = 1;
|
||||
while (lineText[index + runLength] === "`") {
|
||||
runLength += 1;
|
||||
}
|
||||
const hasEqualLengthCloser = consumeInlineBacktickRun(
|
||||
inlineBacktickRuns,
|
||||
runLength,
|
||||
lineStartIndex + index
|
||||
);
|
||||
if (inlineDelimiter === 0) {
|
||||
if (hasEqualLengthCloser) {
|
||||
inlineDelimiter = runLength;
|
||||
}
|
||||
} else if (runLength === inlineDelimiter) {
|
||||
inlineDelimiter = 0;
|
||||
}
|
||||
index += runLength;
|
||||
column += runLength;
|
||||
positionBytesAdvanced += runLength;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (inlineDelimiter > 0) {
|
||||
const symbol = String.fromCodePoint(lineText.codePointAt(index));
|
||||
const symbolBytes = Buffer.byteLength(symbol, "utf8");
|
||||
index += symbol.length;
|
||||
column += 1;
|
||||
positionBytesAdvanced += symbolBytes;
|
||||
continue;
|
||||
}
|
||||
|
||||
let rawLink = null;
|
||||
let endIndex = null;
|
||||
|
||||
if (lineText.startsWith("[待创建: [[", index) || lineText.startsWith("[To create: [[", index)) {
|
||||
const wrapperPrefix = lineText.startsWith("[待创建: [[", index) ? "[待创建: [[" : "[To create: [[";
|
||||
const closeInner = lineText.indexOf("]]]", index + wrapperPrefix.length);
|
||||
if (closeInner >= 0) {
|
||||
rawLink = lineText.slice(index, closeInner + 3);
|
||||
endIndex = closeInner + 3;
|
||||
}
|
||||
} else if (lineText.startsWith("![[", index) || lineText.startsWith("[[", index)) {
|
||||
const closeInner = lineText.indexOf("]]", index + 2);
|
||||
if (closeInner >= 0) {
|
||||
rawLink = lineText.slice(index, closeInner + 2);
|
||||
endIndex = closeInner + 2;
|
||||
}
|
||||
}
|
||||
|
||||
if (rawLink && endIndex !== null) {
|
||||
const startByte = positionBytesAdvanced;
|
||||
const rawLinkBytes = Buffer.byteLength(rawLink, "utf8");
|
||||
occurrences.push(parseOccurrence(rawLink, sourcePath, fileSha256, lineNumber, column, startByte, annotations));
|
||||
index = endIndex;
|
||||
column += countCodePoints(rawLink);
|
||||
positionBytesAdvanced += rawLinkBytes;
|
||||
continue;
|
||||
}
|
||||
|
||||
const symbol = String.fromCodePoint(lineText.codePointAt(index));
|
||||
const symbolBytes = Buffer.byteLength(symbol, "utf8");
|
||||
index += symbol.length;
|
||||
column += 1;
|
||||
positionBytesAdvanced += symbolBytes;
|
||||
}
|
||||
}
|
||||
|
||||
if (nextNewlineIndex === -1) {
|
||||
break;
|
||||
}
|
||||
|
||||
lineStartIndex = nextNewlineIndex + 1;
|
||||
positionBytesAdvanced += 1;
|
||||
lineNumber += 1;
|
||||
}
|
||||
|
||||
return {
|
||||
source_path: sourcePath,
|
||||
file_sha256: fileSha256,
|
||||
occurrences,
|
||||
metrics: {
|
||||
utf8_bytes_scanned: buffer.length,
|
||||
position_bytes_advanced: positionBytesAdvanced
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
function renderWikilinkReplacement(occurrence, replacementTarget) {
|
||||
return `${occurrence.raw_link.slice(0, occurrence.target_start_in_raw)}${replacementTarget}${occurrence.raw_link.slice(occurrence.target_end_in_raw)}`;
|
||||
}
|
||||
|
||||
module.exports = { parseWikilinks, renderWikilinkReplacement };
|
||||
Reference in New Issue
Block a user