dedup.ts 22 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561
  1. /**
  2. * Duplicate-entity / -concept detection and merge for wiki maintenance.
  3. *
  4. * Problem: across re-ingests, the LLM names the same underlying
  5. * topic differently — `paos` vs `聚磷菌`, `dpao` vs `dpaos` (plural)
  6. * vs `反硝化除磷菌`, `vfa` vs `volatile-fatty-acids`. Each becomes a
  7. * separate page even though they're the same entity. The page-merge
  8. * layer only catches *exact* slug collisions; this module catches
  9. * the soft-collision case via an LLM-driven self-check.
  10. *
  11. * Three stages, each independently testable:
  12. *
  13. * 1. extractEntitySummaries: walk wiki/entities and wiki/concepts,
  14. * pull (slug, title, description, tags) per page. Pure-data;
  15. * no LLM.
  16. * 2. detectDuplicateGroups: hand the summary list to an LLM, ask
  17. * it to identify groups of slugs likely to refer to the same
  18. * thing. Returns parsed JSON groups with reason + confidence.
  19. * The LLM call is injected so unit tests don't hit a model.
  20. * 3. mergeDuplicateGroup: given a confirmed group + chosen
  21. * canonical slug, merge bodies (LLM call), union frontmatter
  22. * array fields (deterministic), rewrite every wikilink /
  23. * `related:` reference / index.md entry across the wiki, and
  24. * package up a result the caller writes to disk + backs up.
  25. *
  26. * The caller (UI) is responsible for filesystem reads/writes and
  27. * for showing the user the candidate groups. This module only
  28. * transforms data.
  29. */
  30. import { parseFrontmatter } from "./frontmatter"
  31. import {
  32. parseFrontmatterArray,
  33. mergeArrayFieldsIntoContent,
  34. writeFrontmatterArray,
  35. } from "./sources-merge"
  36. // ──────────────────────────────────────────────────────────────────
  37. // Types
  38. // ──────────────────────────────────────────────────────────────────
  39. export interface EntitySummary {
  40. /** kebab-case slug (basename without `.md`). */
  41. slug: string
  42. /** Path relative to project root, e.g. `wiki/entities/foo.md`. */
  43. path: string
  44. /** entity | concept | source | ... — frontmatter `type` field. */
  45. type: string
  46. title: string
  47. /** Optional one-line description from frontmatter `description`,
  48. * or the first non-empty body paragraph as a fallback. Truncated
  49. * to ~200 chars to keep the detector prompt small. */
  50. description?: string
  51. tags: string[]
  52. }
  53. export interface DuplicateGroup {
  54. /** Two or more slugs from the input list. */
  55. slugs: string[]
  56. /** Why the model believes these are duplicates. Short prose. */
  57. reason: string
  58. confidence: "high" | "medium" | "low"
  59. }
  60. export interface MergeRequest {
  61. /** Pages in the duplicate group, with their full content loaded. */
  62. group: { slug: string; path: string; content: string }[]
  63. /** Slug to keep. Must be one of group[].slug. The other pages
  64. * are deleted; their wikilinks/related entries get rewritten
  65. * to point here. */
  66. canonicalSlug: string
  67. /** Every other .md under the project's wiki/ tree. Used to
  68. * rewrite cross-references when the merge replaces multiple
  69. * pages with one. */
  70. otherWikiPages: { path: string; content: string }[]
  71. }
  72. export interface MergeResult {
  73. /** Final content of the canonical page (frontmatter + body),
  74. * after LLM body merge + deterministic frontmatter unification. */
  75. canonicalContent: string
  76. /** Path of the canonical page on disk (one of the group's). */
  77. canonicalPath: string
  78. /** Cross-reference rewrites in other wiki pages. Caller writes
  79. * each (path → newContent) back to disk. */
  80. rewrites: { path: string; newContent: string }[]
  81. /** Paths to delete after canonical + rewrites are written.
  82. * Excludes the canonical path. */
  83. pagesToDelete: string[]
  84. /** Snapshot of every file the merge touches BEFORE the merge
  85. * was computed. Caller persists this to .qmai/page-history/
  86. * before writing changes so a bad merge can be rolled back. */
  87. backup: { path: string; content: string }[]
  88. }
  89. /**
  90. * Generic two-prompt LLM call. Both detector and merger use it.
  91. * Production wraps `streamChat`; tests use mocks.
  92. */
  93. export type DedupLlmCall = (
  94. systemPrompt: string,
  95. userMessage: string,
  96. signal?: AbortSignal,
  97. ) => Promise<string>
  98. // ──────────────────────────────────────────────────────────────────
  99. // Stage 1: extract summaries (no LLM)
  100. // ──────────────────────────────────────────────────────────────────
  101. /**
  102. * Build an EntitySummary from a single page's path + content.
  103. * `pathRelativeToProject` should be the canonical wiki-relative
  104. * form (`wiki/entities/foo.md`) so callers downstream can derive
  105. * slugs consistently.
  106. */
  107. export function extractEntitySummary(
  108. pathRelativeToProject: string,
  109. content: string,
  110. ): EntitySummary | null {
  111. const { frontmatter, body } = parseFrontmatter(content)
  112. if (!frontmatter) return null
  113. const type = stringField(frontmatter.type) ?? "unknown"
  114. const title = stringField(frontmatter.title) ?? slugFromPath(pathRelativeToProject)
  115. const description = stringField(frontmatter.description) ?? firstBodyParagraph(body)
  116. const tags = arrayField(frontmatter.tags)
  117. return {
  118. slug: slugFromPath(pathRelativeToProject),
  119. path: pathRelativeToProject,
  120. type,
  121. title,
  122. description: description ? truncate(description, 200) : undefined,
  123. tags,
  124. }
  125. }
  126. function slugFromPath(path: string): string {
  127. const base = path.split("/").pop() ?? path
  128. return base.replace(/\.md$/, "")
  129. }
  130. function stringField(v: unknown): string | undefined {
  131. if (typeof v === "string" && v.trim() !== "") return v.trim()
  132. return undefined
  133. }
  134. function arrayField(v: unknown): string[] {
  135. if (!Array.isArray(v)) return []
  136. return v.filter((x): x is string => typeof x === "string" && x.trim() !== "")
  137. }
  138. function firstBodyParagraph(body: string): string | undefined {
  139. const lines = body.split("\n").map((l) => l.trim()).filter(Boolean)
  140. // Skip leading h1/h2 lines so the description isn't just the title again.
  141. for (const line of lines) {
  142. if (line.startsWith("#")) continue
  143. if (line.startsWith("|")) continue // table — too noisy
  144. return line
  145. }
  146. return undefined
  147. }
  148. function truncate(s: string, max: number): string {
  149. if (s.length <= max) return s
  150. return s.slice(0, max - 1) + "…"
  151. }
  152. // ──────────────────────────────────────────────────────────────────
  153. // Stage 2: LLM-driven duplicate detection
  154. // ──────────────────────────────────────────────────────────────────
  155. const DETECTOR_SYSTEM_PROMPT = `你是一个维基维护助手。你将收到一个维基中的实体/概念页面列表。请找出那些很可能指向同一主题但名称不同的 slug 分组——例如:
  156. - 同一名称的不同语言版本(中英文等)
  157. - 单复数形式(如 "dpao" 和 "dpaos")
  158. - 缩写与全称(如 "vfa" 和 "volatile-fatty-acids")
  159. - 同义词
  160. - 同一专有名词的不同拼写
  161. 只输出有效的 JSON。不要输出散文、markdown 代码块或 JSON 之外的任何解释。JSON 结构如下:
  162. {
  163. "groups": [
  164. {
  165. "slugs": ["slug-a", "slug-b"],
  166. "reason": "两个页面都指向 X;第一个是英文,第二个是中文。",
  167. "confidence": "high"
  168. }
  169. ]
  170. }
  171. 规则:
  172. - 只包含输入列表中 2 个或更多 slug 的分组。
  173. - "high" = 明显是同一实体,只是命名不同。
  174. - "medium" = 可能是同一实体,但需要结合上下文判断。
  175. - "low" = 不确定,需要用户仔细审查。
  176. - 不要编造输入列表中不存在的 slug。
  177. - 如果没有重复项,输出 {"groups": []}。
  178. - 不同 \`type\`(如 entity 和 concept)的页面通常不应分在一组——只有在明确是同一事物时才跨类型分组。
  179. 重要:reason 字段必须使用中文描述。`
  180. /**
  181. * Run the LLM duplicate-detector. The caller hands in summaries
  182. * (typically every entity + concept page in the wiki) and a
  183. * function that wraps an LLM call. Returns parsed, validated
  184. * groups — invalid entries (slugs not in the input, single-element
  185. * groups) are filtered out so the caller never sees garbage.
  186. *
  187. * Already-confirmed-not-duplicate groups passed in `notDuplicates`
  188. * are filtered out before returning so the same false positive
  189. * doesn't keep appearing on every run.
  190. */
  191. export async function detectDuplicateGroups(
  192. summaries: EntitySummary[],
  193. llmCall: DedupLlmCall,
  194. options: { signal?: AbortSignal; notDuplicates?: string[][] } = {},
  195. ): Promise<DuplicateGroup[]> {
  196. if (summaries.length < 2) return []
  197. const userMessage = buildDetectorUserMessage(summaries)
  198. const response = await llmCall(DETECTOR_SYSTEM_PROMPT, userMessage, options.signal)
  199. const parsed = parseDetectorResponse(response)
  200. const validSlugs = new Set(summaries.map((s) => s.slug))
  201. const notDupSet = new Set(
  202. (options.notDuplicates ?? []).map((g) => normalizeGroupKey(g)),
  203. )
  204. return parsed
  205. .map((g) => ({ ...g, slugs: g.slugs.filter((s) => validSlugs.has(s)) }))
  206. .filter((g) => g.slugs.length >= 2)
  207. .filter((g) => !notDupSet.has(normalizeGroupKey(g.slugs)))
  208. }
  209. function buildDetectorUserMessage(summaries: EntitySummary[]): string {
  210. const lines = summaries.map((s) => {
  211. const tagPart = s.tags.length > 0 ? ` [${s.tags.join(", ")}]` : ""
  212. const descPart = s.description ? ` — ${s.description}` : ""
  213. return `- type=${s.type}, slug=${s.slug}, title=${JSON.stringify(s.title)}${tagPart}${descPart}`
  214. })
  215. return `## Wiki pages to scan (${summaries.length} entries)\n\n${lines.join("\n")}\n\nReturn duplicate groups as JSON only.`
  216. }
  217. /**
  218. * Tolerant JSON extraction. The LLM might wrap output in code
  219. * fences (\`\`\`json), prepend "Sure, here you go:", or trail
  220. * with a polite "Let me know if...". Pull the first {…} block
  221. * with balanced braces and parse it. Returns [] for any failure
  222. * — the caller treats "no duplicates found" identically to "LLM
  223. * output garbled".
  224. */
  225. export function parseDetectorResponse(raw: string): DuplicateGroup[] {
  226. const jsonText = extractFirstJsonObject(raw)
  227. if (!jsonText) return []
  228. let parsed: unknown
  229. try {
  230. parsed = JSON.parse(jsonText)
  231. } catch {
  232. return []
  233. }
  234. if (!parsed || typeof parsed !== "object") return []
  235. const groupsRaw = (parsed as { groups?: unknown }).groups
  236. if (!Array.isArray(groupsRaw)) return []
  237. const out: DuplicateGroup[] = []
  238. for (const g of groupsRaw) {
  239. if (!g || typeof g !== "object") continue
  240. const obj = g as Record<string, unknown>
  241. const slugs = Array.isArray(obj.slugs)
  242. ? obj.slugs.filter((s): s is string => typeof s === "string")
  243. : []
  244. if (slugs.length < 2) continue
  245. const reason = typeof obj.reason === "string" ? obj.reason : ""
  246. const confidence: DuplicateGroup["confidence"] =
  247. obj.confidence === "high" || obj.confidence === "medium"
  248. ? obj.confidence
  249. : "low"
  250. out.push({ slugs, reason, confidence })
  251. }
  252. return out
  253. }
  254. /** Extract the first balanced `{...}` substring from arbitrary text. */
  255. function extractFirstJsonObject(text: string): string | null {
  256. const start = text.indexOf("{")
  257. if (start < 0) return null
  258. let depth = 0
  259. let inString = false
  260. let escape = false
  261. for (let i = start; i < text.length; i++) {
  262. const ch = text[i]
  263. if (escape) {
  264. escape = false
  265. continue
  266. }
  267. if (ch === "\\") {
  268. escape = true
  269. continue
  270. }
  271. if (ch === '"') {
  272. inString = !inString
  273. continue
  274. }
  275. if (inString) continue
  276. if (ch === "{") depth++
  277. else if (ch === "}") {
  278. depth--
  279. if (depth === 0) return text.slice(start, i + 1)
  280. }
  281. }
  282. return null
  283. }
  284. /** Canonical key for a group — lowercased, sorted, comma-joined. */
  285. function normalizeGroupKey(slugs: string[]): string {
  286. return [...slugs].map((s) => s.toLowerCase()).sort().join(",")
  287. }
  288. // ──────────────────────────────────────────────────────────────────
  289. // Stage 3: merge a confirmed duplicate group
  290. // ──────────────────────────────────────────────────────────────────
  291. const MERGER_SYSTEM_PROMPT = `你是一个维基维护助手。你将收到几个描述同一实体或概念但名称不同的维基页面。请将它们合并为一个连贯的维基页面。
  292. 输出完整的合并文件(frontmatter + 正文)。你回复的第一个字符必须是 "-"("---" 的开头)。不要输出前言或文件之外的任何解释。
  293. 规则:
  294. - 保留每个输入页面中所有不同的事实性陈述。
  295. - 消除冗余(不要在多个章节中重复相同的内容)。
  296. - 重新组织章节结构,使其对统一后的主题具有逻辑性,而不是简单拼接输入内容。
  297. - 在正文中使用 [[wikilink]] 语法(如果输入中使用了的话)。
  298. - Frontmatter:保留标准字段(type, title, created, updated, tags, related, sources)。调用方会在之后用确定性合并覆盖 sources / tags / related / updated 字段——你的任务是生成合理的正文和合理的 frontmatter 结构。
  299. - 选择最具描述性的标题。如果输入使用了不同语言,优先选择与正文内容多数语言匹配的语言。`
  300. const FIELDS_TO_UNION = ["sources", "tags", "related"] as const
  301. /**
  302. * Compute everything needed to merge a confirmed duplicate group:
  303. * - LLM call to produce the merged canonical body
  304. * - Deterministic frontmatter union (sources, tags, related)
  305. * - Canonical slug enforcement on title path
  306. * - Cross-reference rewrites across every other wiki page
  307. * - Backup snapshot of all touched files
  308. *
  309. * Returns a MergeResult; the CALLER is responsible for actually
  310. * writing canonicalContent + each rewrite + deleting the merged-
  311. * away files + storing the backup. Splitting compute from I/O
  312. * keeps this testable.
  313. */
  314. export async function mergeDuplicateGroup(
  315. req: MergeRequest,
  316. llmCall: DedupLlmCall,
  317. options: { signal?: AbortSignal; today?: () => string } = {},
  318. ): Promise<MergeResult> {
  319. const canonical = req.group.find((p) => p.slug === req.canonicalSlug)
  320. if (!canonical) {
  321. throw new Error(
  322. `canonicalSlug "${req.canonicalSlug}" is not in the group: ${req.group.map((p) => p.slug).join(", ")}`,
  323. )
  324. }
  325. if (req.group.length < 2) {
  326. throw new Error("mergeDuplicateGroup requires at least 2 pages in the group")
  327. }
  328. // 1. LLM body merge
  329. const userMessage = buildMergerUserMessage(req.group)
  330. const llmOutput = await llmCall(MERGER_SYSTEM_PROMPT, userMessage, options.signal)
  331. // 2. Frontmatter union (deterministic post-processing of LLM output).
  332. // For each unioned field, fold every input page's values into
  333. // the LLM output via mergeArrayFieldsIntoContent.
  334. let merged = llmOutput
  335. for (const page of req.group) {
  336. merged = mergeArrayFieldsIntoContent(merged, page.content, [...FIELDS_TO_UNION])
  337. }
  338. // 3. Stamp updated to today and force a sensible title.
  339. const today = (options.today ?? defaultToday)()
  340. merged = setFrontmatterScalar(merged, "updated", today)
  341. // If LLM output's frontmatter parses cleanly we leave its title;
  342. // if not, the application layer doesn't try to manufacture one.
  343. // 4. Cross-reference rewrites: every other wiki page that mentions
  344. // a non-canonical slug needs its wikilinks / related entries
  345. // rewritten to the canonical.
  346. const slugRedirects = new Map<string, string>()
  347. for (const page of req.group) {
  348. if (page.slug !== req.canonicalSlug) {
  349. slugRedirects.set(page.slug, req.canonicalSlug)
  350. }
  351. }
  352. const rewrites: MergeResult["rewrites"] = []
  353. for (const page of req.otherWikiPages) {
  354. const rewritten = rewriteCrossReferences(page.content, slugRedirects)
  355. if (rewritten !== page.content) {
  356. rewrites.push({ path: page.path, newContent: rewritten })
  357. }
  358. }
  359. // 5. Backup: every touched file's PRE-merge content.
  360. const backup: MergeResult["backup"] = []
  361. for (const page of req.group) {
  362. backup.push({ path: page.path, content: page.content })
  363. }
  364. for (const r of rewrites) {
  365. const orig = req.otherWikiPages.find((p) => p.path === r.path)
  366. if (orig) backup.push({ path: orig.path, content: orig.content })
  367. }
  368. // 6. Pages to delete: every group member except the canonical.
  369. const pagesToDelete = req.group
  370. .filter((p) => p.slug !== req.canonicalSlug)
  371. .map((p) => p.path)
  372. return {
  373. canonicalContent: merged,
  374. canonicalPath: canonical.path,
  375. rewrites,
  376. pagesToDelete,
  377. backup,
  378. }
  379. }
  380. function buildMergerUserMessage(
  381. group: { slug: string; content: string }[],
  382. ): string {
  383. const sections = group.map((p, i) => {
  384. return [
  385. `## Page ${i + 1} (slug: ${p.slug})`,
  386. "",
  387. p.content,
  388. "",
  389. ].join("\n")
  390. })
  391. return [
  392. `These ${group.length} wiki pages have been confirmed by the user to describe the same topic.`,
  393. `Merge them into a single coherent page (the canonical slug will be "${group[0].slug}" or whichever the caller chose).`,
  394. "",
  395. sections.join("\n---\n\n"),
  396. "",
  397. "Now output the merged file. First character must be `-`.",
  398. ].join("\n")
  399. }
  400. /**
  401. * Rewrite cross-references to merged-away slugs throughout one
  402. * page's content. Three forms get rewritten:
  403. *
  404. * 1. `[[old-slug]]` and `[[old-slug|alias]]` in the body
  405. * — replace just the target portion, keep alias if present.
  406. * 2. `related: [..., old-slug, ...]` (inline form) — substitute
  407. * old-slug with canonical inside the array, then dedup.
  408. * 3. `related:\n - old-slug` (block form) — same substitution.
  409. *
  410. * `wiki/index.md`-style listings of files are out of scope here —
  411. * the caller handles index regeneration separately.
  412. */
  413. export function rewriteCrossReferences(
  414. content: string,
  415. slugRedirects: Map<string, string>,
  416. ): string {
  417. let out = content
  418. // 1. Wikilinks in the body — both [[slug]] and [[slug|alias]].
  419. for (const [oldSlug, newSlug] of slugRedirects) {
  420. const escaped = escapeRegex(oldSlug)
  421. const re = new RegExp(`\\[\\[${escaped}(\\|[^\\]]+)?\\]\\]`, "g")
  422. out = out.replace(re, (_match, alias) => `[[${newSlug}${alias ?? ""}]]`)
  423. }
  424. // 2. & 3. `related` field — re-parse and rewrite.
  425. const existing = parseFrontmatterArray(out, "related")
  426. if (existing.length > 0) {
  427. const rewritten = existing.map((s) => slugRedirects.get(s) ?? s)
  428. // Deduplicate (case-insensitive, first-seen casing wins)
  429. const seen = new Set<string>()
  430. const unique: string[] = []
  431. for (const s of rewritten) {
  432. const k = s.toLowerCase()
  433. if (seen.has(k)) continue
  434. seen.add(k)
  435. unique.push(s)
  436. }
  437. if (
  438. unique.length !== existing.length ||
  439. unique.some((s, i) => s !== existing[i])
  440. ) {
  441. out = writeFrontmatterArray(out, "related", unique)
  442. }
  443. }
  444. return out
  445. }
  446. function escapeRegex(s: string): string {
  447. return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")
  448. }
  449. function setFrontmatterScalar(
  450. content: string,
  451. field: string,
  452. value: string,
  453. ): string {
  454. const m = content.match(/^(---\n)([\s\S]*?)(\n---)/)
  455. if (!m) return content
  456. const [, open, body, close] = m
  457. const escaped = field.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")
  458. const newLine = `${field}: ${value}`
  459. const lineRe = new RegExp(`^${escaped}:\\s*(?!\\[)([^\\n]*)`, "m")
  460. if (lineRe.test(body)) {
  461. const rewritten = body.replace(lineRe, newLine)
  462. return `${open}${rewritten}${close}${content.slice(m[0].length)}`
  463. }
  464. return `${open}${body}\n${newLine}${close}${content.slice(m[0].length)}`
  465. }
  466. function defaultToday(): string {
  467. return new Date().toISOString().slice(0, 10)
  468. }
  469. // ──────────────────────────────────────────────────────────────────
  470. // Index rewriter — wiki/index.md-specific
  471. // ──────────────────────────────────────────────────────────────────
  472. /**
  473. * Remove entries for merged-away slugs from `wiki/index.md`.
  474. * Index files are typically formatted as bullet / link lists
  475. * grouped by section. This is a CONSERVATIVE rewriter:
  476. * - Removes any whole line that contains a markdown link or
  477. * wikilink to a merged-away slug.
  478. * - Preserves all other content verbatim (other sections,
  479. * intros, the canonical entry).
  480. * The caller (UI) shows the user a diff before writing so any
  481. * over-removal is visible.
  482. */
  483. export function rewriteIndexMd(
  484. content: string,
  485. removedSlugs: Set<string>,
  486. ): string {
  487. if (removedSlugs.size === 0) return content
  488. const lines = content.split("\n")
  489. const out: string[] = []
  490. for (const line of lines) {
  491. if (lineRefersToSlug(line, removedSlugs)) continue
  492. out.push(line)
  493. }
  494. return out.join("\n")
  495. }
  496. function lineRefersToSlug(line: string, slugs: Set<string>): boolean {
  497. for (const slug of slugs) {
  498. const escaped = escapeRegex(slug)
  499. // Wikilink form: [[slug]] or [[slug|alias]]
  500. if (new RegExp(`\\[\\[${escaped}(\\|[^\\]]*)?\\]\\]`).test(line)) return true
  501. // Markdown link form: [...](slug.md) or [...](path/slug.md)
  502. if (new RegExp(`\\(([^)]*\\/)?${escaped}\\.md\\)`).test(line)) return true
  503. // Bare slug.md mention (rare but seen in raw lists)
  504. if (new RegExp(`\\b${escaped}\\.md\\b`).test(line)) return true
  505. }
  506. return false
  507. }