From 953688d96ecbb65a3f6bc0d150d1ed2707df7a0d Mon Sep 17 00:00:00 2001 From: Kodai Nakamura Date: Mon, 22 Jun 2026 01:22:38 +0900 Subject: [PATCH] refactor(replace-links): centralize candidate scanning Why: - Candidate detection was duplicated between link replacement and AI ambiguity resolution, which made matching behavior hard to reason about. - A single scanner improves locality for namespace, protected-span, case, and CJK matching rules. What: - Add a shared candidate scanner module for unlinked text and existing wikilinks. - Route AI ambiguity request construction through the scanner. - Keep link rendering behavior unchanged and covered by existing replacement tests. --- .../__tests__/candidate-scanner.test.ts | 131 ++++ src/replace-links/candidate-scanner.ts | 724 ++++++++++++++++++ src/replace-links/replace-links.ts | 359 +-------- src/utils/resolve-ambiguities.ts | 82 +- 4 files changed, 894 insertions(+), 402 deletions(-) create mode 100644 src/replace-links/__tests__/candidate-scanner.test.ts create mode 100644 src/replace-links/candidate-scanner.ts diff --git a/src/replace-links/__tests__/candidate-scanner.test.ts b/src/replace-links/__tests__/candidate-scanner.test.ts new file mode 100644 index 0000000..5aa386f --- /dev/null +++ b/src/replace-links/__tests__/candidate-scanner.test.ts @@ -0,0 +1,131 @@ +import { describe, expect, it } from "vitest" +import { buildCandidateTrieForTest } from "./test-helpers" +import { + getOccurrenceContext, + scanCandidateOccurrences, +} from "../candidate-scanner" + +describe("scanCandidateOccurrences", () => { + it("reports ambiguous prose candidates and skips inline code", () => { + const { candidateMap, trie } = buildCandidateTrieForTest({ + files: [ + { path: "work/meeting" }, + { path: "private/meeting" }, + ], + settings: { + scoped: false, + baseDir: undefined, + ignoreCase: true, + }, + }) + + const occurrences = scanCandidateOccurrences({ + text: "`meeting` meeting", + filePath: "notes/today", + trie, + candidateMap, + settings: { ignoreCase: true, proximityBasedLinking: true }, + }) + + expect(occurrences.map(o => ({ + kind: o.kind, + text: o.text, + start: o.start, + end: o.end, + candidates: o.candidateData.candidates.map(c => c.canonical), + }))).toEqual([ + { + kind: "unlinked", + text: "meeting", + start: 10, + end: 17, + candidates: ["work/meeting", "private/meeting"], + }, + ]) + }) + + it("reports existing wikilinks by their display alias", () => { + const { candidateMap, trie } = buildCandidateTrieForTest({ + files: [ + { path: "work/meeting" }, + { path: "private/meeting" }, + ], + settings: { + scoped: false, + baseDir: undefined, + ignoreCase: true, + }, + }) + + const occurrences = scanCandidateOccurrences({ + text: "Check [[private/meeting|meeting]] notes.", + filePath: "notes/today", + trie, + candidateMap, + settings: { ignoreCase: true, proximityBasedLinking: true }, + }) + + expect(occurrences.map(o => ({ + kind: o.kind, + text: o.text, + candidateKey: o.candidateKey, + candidates: o.candidateData.candidates.map(c => c.canonical), + }))).toEqual([ + { + kind: "existing-wikilink", + text: "[[private/meeting|meeting]]", + candidateKey: "meeting", + candidates: ["work/meeting", "private/meeting"], + }, + ]) + }) + + it("filters scoped candidates outside the current namespace", () => { + const { candidateMap, trie } = buildCandidateTrieForTest({ + files: [ + { path: "pages/team-a/internal" }, + { path: "pages/team-b/internal" }, + ], + settings: { + scoped: true, + baseDir: "pages", + ignoreCase: true, + }, + }) + + const occurrences = scanCandidateOccurrences({ + text: "internal", + filePath: "pages/team-a/today", + trie, + candidateMap, + settings: { + baseDir: "pages", + ignoreCase: true, + proximityBasedLinking: true, + }, + }) + + expect(occurrences).toHaveLength(1) + expect(occurrences[0].candidateData.candidates.map(c => c.canonical)).toEqual([ + "pages/team-a/internal", + ]) + }) +}) + +describe("getOccurrenceContext", () => { + it("returns bounded surrounding text for AI requests", () => { + const occurrence = { + kind: "unlinked" as const, + start: 10, + end: 17, + text: "meeting", + candidateKey: "meeting", + candidateData: { candidates: [] }, + isInTable: false, + } + + expect(getOccurrenceContext("before -- meeting -- after", occurrence, 4)).toBe( + " -- meeting -- ", + ) + }) +}) diff --git a/src/replace-links/candidate-scanner.ts b/src/replace-links/candidate-scanner.ts new file mode 100644 index 0000000..2da65a3 --- /dev/null +++ b/src/replace-links/candidate-scanner.ts @@ -0,0 +1,724 @@ +import { CandidateData, getTopLevelDirectoryName, TrieNode } from "../trie" +import type { ReplaceLinksSettings } from "./replace-links" + +export type CandidateOccurrenceKind = "unlinked" | "existing-wikilink" + +export interface CandidateOccurrence { + kind: CandidateOccurrenceKind + start: number + end: number + text: string + candidateKey: string + candidateData: CandidateData + isInTable: boolean +} + +export interface ScanCandidateOccurrencesOptions { + text: string + filePath: string + trie: TrieNode + candidateMap: Map + settings?: ReplaceLinksSettings +} + +export const REGEX_PATTERNS = { + PROTECTED: /(```[\s\S]*?```|`[^`]*`|\[\[([^\]]+)\]\]|\[[^\]]+\]\([^)]+\)|\[[^\]]+\]|https?:\/\/[^\s]+)/g, + DATE_FORMAT: /^\d{4}-\d{2}-\d{2}$/, + MONTH_NOTE: /^[0-9]{1,2}$/, + CJK: /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}]/u, + CJK_CANDIDATE: /^[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}\s\d]+$/u, + KOREAN: /^[\p{Script=Hangul}]+$/u, + JAPANESE: /^[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\s\d]+$/u, + URL: /^(https?:\/\/[^\s]+)/, + PROTECTED_LINK: /^\s*(\[\[[^\]]+\]\]|\[[^\]]+\]\([^)]+\))\s*$/, + KOREAN_SUFFIX: /^(이다\.?)/, + KOREAN_PARTICLES: /^(는|은)/, + KOREAN_PARTICLES_EXTENDED: /^(가|는|을|에|서|와|로부터|까지|보다|로|의|나|도|또한)/, + TABLE_SEPARATOR: /^[|:\s-]+$/, + WORD_BOUNDARY: /[\p{L}\p{N}_/-]/u, + WHITESPACE: /[\t\n\r ]/, +} as const + +export const isWordBoundary = (char: string | undefined): boolean => { + if (char === undefined) return true + if (REGEX_PATTERNS.CJK.test(char)) return true + return (!REGEX_PATTERNS.WORD_BOUNDARY.test(char) || REGEX_PATTERNS.WHITESPACE.test(char)) +} + +export const isMonthNote = (candidate: string): boolean => + REGEX_PATTERNS.MONTH_NOTE.test(candidate) + && parseInt(candidate, 10) >= 1 + && parseInt(candidate, 10) <= 12 + +export const isProtectedLink = (body: string): boolean => REGEX_PATTERNS.PROTECTED_LINK.test(body) + +export const isCjkText = (text: string): boolean => REGEX_PATTERNS.CJK.test(text) + +export const isCjkCandidate = (candidate: string): boolean => REGEX_PATTERNS.CJK_CANDIDATE.test(candidate) + +export const isKoreanText = (text: string): boolean => REGEX_PATTERNS.KOREAN.test(text) + +export const isSentenceStart = (text: string, index: number): boolean => { + if (index === 0) return true + if (text[index - 1] === "\n") return true + if (index >= 3 && text[index - 1] === " " && text[index - 2] === ".") { + const charBeforePeriod = text[index - 3] + if (/[a-zA-Z]/.test(charBeforePeriod)) { + return true + } + } + return false +} + +const fallbackIndexCache = new WeakMap< + Map, + Map>> +>() + +export const buildFallbackIndex = ( + candidateMap: Map, + ignoreCase?: boolean, +): Map> => { + let cacheForMap = fallbackIndexCache.get(candidateMap) + if (!cacheForMap) { + cacheForMap = new Map() + fallbackIndexCache.set(candidateMap, cacheForMap) + } + + const cacheKey = ignoreCase ? "ignoreCase" : "normal" + const cached = cacheForMap.get(cacheKey) + if (cached) return cached + + const fallbackIndex = new Map>() + + for (const [key, data] of candidateMap.entries()) { + const slashIndex = key.lastIndexOf("/") + if (slashIndex === -1) continue + + const shorthand = key.slice(slashIndex + 1) + const indexKey = ignoreCase ? shorthand.toLowerCase() : shorthand + + let arr = fallbackIndex.get(indexKey) + if (!arr) { + arr = [] + fallbackIndex.set(indexKey, arr) + } + arr.push([key, data]) + } + + cacheForMap.set(cacheKey, fallbackIndex) + return fallbackIndex +} + +export const getCurrentNamespace = (filePath: string, baseDir?: string): string => { + if (baseDir) { + return getTopLevelDirectoryName(filePath, baseDir) + } + + const segments = filePath.split("/") + return segments[0] || "" +} + +export const normalizeCanonicalPath = (linkPath: string, baseDir?: string): string => { + if (baseDir && linkPath.startsWith(baseDir + "/")) { + return linkPath.slice((baseDir + "/").length) + } + return linkPath +} + +export const extractLinkParts = ( + canonicalPath: string, +): { linkPath: string, alias: string, hasAlias: boolean } => { + const pipeIndex = canonicalPath.indexOf("|") + const hasAlias = pipeIndex !== -1 + + if (hasAlias) { + const linkPath = canonicalPath.slice(0, pipeIndex) + const alias = canonicalPath.slice(pipeIndex + 1) + return { linkPath, alias, hasAlias } + } + + return { linkPath: canonicalPath, alias: "", hasAlias } +} + +export const escapeRegExp = (text: string): string => + text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&") + +export const extractFencedCodeBlocks = ( + body: string, +): { body: string, codeBlocks: Array<{ placeholder: string, content: string }> } => { + if (!body.includes("```") && !body.includes("~~~")) { + return { body, codeBlocks: [] } + } + + const codeBlocks: Array<{ placeholder: string, content: string }> = [] + let result = "" + let cursor = 0 + let codeBlockIndex = 0 + + const openingFencePattern = /^ {0,3}(`{3,}|~{3,})[^\r\n]*(?:\r?\n|$)/gm + let openingMatch: RegExpExecArray | null + + while ((openingMatch = openingFencePattern.exec(body)) !== null) { + const start = openingMatch.index + if (start < cursor) { + continue + } + + const openingFence = openingMatch[1] + const fenceChar = openingFence[0] + const fenceLength = openingFence.length + const closingFencePattern = new RegExp( + `^ {0,3}${escapeRegExp(fenceChar)}{${fenceLength},}[ \\t]*(?:\\r?\\n|$)`, + "gm", + ) + closingFencePattern.lastIndex = openingMatch.index + openingMatch[0].length + const closingMatch = closingFencePattern.exec(body) + const end = closingMatch + ? closingMatch.index + closingMatch[0].length + : body.length + + const placeholder = `__CODE_BLOCK_${codeBlockIndex}__` + codeBlocks.push({ + placeholder, + content: body.slice(start, end), + }) + result += body.slice(cursor, start) + placeholder + cursor = end + codeBlockIndex++ + openingFencePattern.lastIndex = end + } + + result += body.slice(cursor) + return { body: result, codeBlocks } +} + +export const isSelfLink = ( + candidateData: CandidateData, + currentFilePath: string, + settings: ReplaceLinksSettings = {}, +): boolean => { + if (!settings.preventSelfLinking || candidateData.candidates.length === 0) { + return false + } + + const { linkPath } = extractLinkParts(candidateData.candidates[0].canonical) + const normalizedLinkPath = normalizeCanonicalPath( + linkPath, + settings.baseDir, + ) + const normalizedCurrentPath = normalizeCanonicalPath( + currentFilePath, + settings.baseDir, + ) + + return normalizedLinkPath === normalizedCurrentPath +} + +export const shouldSkipCandidate = ( + candidate: string, + settings: ReplaceLinksSettings, +): boolean => { + if ( + settings.ignoreDateFormats + && REGEX_PATTERNS.DATE_FORMAT.test(candidate) + ) { + return true + } + return isMonthNote(candidate) +} + +export const isMarkdownTableLine = (line: string): boolean => { + const trimmedLine = line.trim() + if (!trimmedLine || !trimmedLine.includes("|")) { + return false + } + + if (trimmedLine.startsWith("|") + && trimmedLine.endsWith("|") + && REGEX_PATTERNS.TABLE_SEPARATOR.test(trimmedLine) + ) { + return true + } + + return trimmedLine.startsWith("|") && trimmedLine.endsWith("|") +} + +export const isIndexInsideMarkdownTable = (text: string, index: number): boolean => { + let lineStart = text.lastIndexOf("\n", index - 1) + 1 + if (lineStart === 0 && text[0] !== "\n") { + lineStart = 0 + } + + let lineEnd = text.indexOf("\n", index) + if (lineEnd === -1) { + lineEnd = text.length + } + + const line = text.slice(lineStart, lineEnd) + return isMarkdownTableLine(line) +} + +export const findBestCandidateInSameNamespace = ( + filteredCandidates: Array<[string, CandidateData]>, + filePath: string, + settings: ReplaceLinksSettings = {}, +): [string, CandidateData] | null => { + let bestCandidate: [string, CandidateData] | null = null + let bestScore = -1 + + const filePathDir = filePath.includes("/") + ? filePath.slice(0, filePath.lastIndexOf("/")) + : "" + const filePathSegments = filePathDir ? filePathDir.split("/") : [] + + for (const [key, data] of filteredCandidates) { + const slashIndex = key.lastIndexOf("/") + const candidateDir = key.slice(0, slashIndex) + const candidateSegments = candidateDir.split("/") + let score = 0 + + for ( + let idx = 0; + idx < Math.min(candidateSegments.length, filePathSegments.length); + idx++ + ) { + if (candidateSegments[idx] === filePathSegments[idx]) { + score++ + } + else { + break + } + } + + if (score > bestScore) { + bestScore = score + bestCandidate = [key, data] + } + else if (score === bestScore && bestCandidate !== null) { + if (filePathDir === "" && settings.baseDir) { + const basePrefix = settings.baseDir + "/" + const getRelativeDepth = (k: string): number => { + if (k.startsWith(basePrefix)) { + const relativeParts = k + .slice(basePrefix.length) + .split("/") + return relativeParts.length - 1 + } + return Infinity + } + + const candidateDepth = getRelativeDepth(key) + const bestCandidateDepth = getRelativeDepth(bestCandidate[0]) + + if (candidateDepth < bestCandidateDepth + || (candidateDepth === bestCandidateDepth && key.length < bestCandidate[0].length)) { + bestCandidate = [key, data] + } + } + else { + const currentBestDir = bestCandidate[0].slice(0, bestCandidate[0].lastIndexOf("/")) + const currentBestSegments = currentBestDir.split("/") + + if ( + candidateSegments.length < currentBestSegments.length + || (candidateSegments.length === currentBestSegments.length && key.length < bestCandidate[0].length) + ) { + bestCandidate = [key, data] + } + } + } + } + + return bestCandidate +} + +const dedupeCandidates = (candidates: CandidateData["candidates"]): CandidateData => { + const unique = new Map(candidates.map(candidate => [candidate.canonical, candidate])) + return { + candidates: Array.from(unique.values()).sort((a, b) => { + if (a.canonical.length !== b.canonical.length) { + return a.canonical.length - b.canonical.length + } + return a.canonical.localeCompare(b.canonical) + }), + } +} + +const filterCandidateDataForNamespace = ( + candidateData: CandidateData, + currentNamespace: string, + settings: ReplaceLinksSettings, +): CandidateData => { + if (!settings.proximityBasedLinking) { + return dedupeCandidates(candidateData.candidates) + } + + const filteredCandidates = candidateData.candidates.filter((candidate) => { + return !(candidate.scoped && candidate.namespace !== currentNamespace) + }) + + return dedupeCandidates(filteredCandidates) +} + +const collectExistingWikilinks = ( + text: string, + candidateMap: Map, + occurrences: CandidateOccurrence[], + settings: ReplaceLinksSettings, +): void => { + const existingLinkRegex = /\[\[([^|\]]+)(?:\|([^\]]+))?\]\]/g + let match: RegExpExecArray | null + + while ((match = existingLinkRegex.exec(text)) !== null) { + const fullMatch = match[0] + const path = match[1] + const alias = match[2] || path + const candidateKey = settings.ignoreCase ? alias.toLowerCase() : alias + const candidateData = candidateMap.get(candidateKey) ?? candidateMap.get(alias) + + if (!candidateData) { + continue + } + + occurrences.push({ + kind: "existing-wikilink", + start: match.index, + end: match.index + fullMatch.length, + text: fullMatch, + candidateKey, + candidateData: dedupeCandidates(candidateData.candidates), + isInTable: isIndexInsideMarkdownTable(text, match.index), + }) + } +} + +const collectFallbackOccurrence = ({ + text, + startIndex, + fallbackIndex, + filePath, + currentNamespace, + settings, +}: { + text: string + startIndex: number + fallbackIndex: Map> + filePath: string + currentNamespace: string + settings: ReplaceLinksSettings +}): CandidateOccurrence | null => { + const prevChar = text[startIndex - 1] + if (!isWordBoundary(prevChar)) { + return null + } + + let longestMatch: { + word: string + length: number + key: string + candidateList: Array<[string, CandidateData]> + } | null = null + + const maxSearchLength = Math.min(text.length - startIndex, 100) + + let potentialMatch = "" + let searchWord = "" + + for (let length = 1; length <= maxSearchLength; length++) { + const endIndex = startIndex + length + const currentChar = text[startIndex + length - 1] + potentialMatch += currentChar + + if (settings.matchSentenceCase && !settings.ignoreCase && isSentenceStart(text, startIndex)) { + if (length === 1) { + searchWord = currentChar.toLowerCase() + } + else { + searchWord += currentChar + } + } + else { + searchWord = settings.ignoreCase + ? searchWord + currentChar.toLowerCase() + : potentialMatch + } + + const candidateList = fallbackIndex.get(searchWord) + if (!candidateList) { + continue + } + + const nextChar = text[endIndex] + if (!isWordBoundary(nextChar)) { + continue + } + + if (shouldSkipCandidate(potentialMatch, settings)) { + continue + } + + longestMatch = { + word: potentialMatch, + length, + key: searchWord, + candidateList, + } + } + + if (!longestMatch) return null + + const filteredCandidates = longestMatch.candidateList.filter(([, data]) => { + if (data.candidates.length === 0 || !settings.proximityBasedLinking) { + return true + } + const candidate = data.candidates[0] + return !(candidate.scoped && candidate.namespace !== currentNamespace) + }) + + if (filteredCandidates.length === 0) { + return null + } + + const candidateData = dedupeCandidates( + filteredCandidates.flatMap(([, data]) => data.candidates), + ) + + if (candidateData.candidates.length === 0) { + return null + } + + if (isSelfLink(candidateData, filePath, settings)) { + return null + } + + const bestCandidateResult = filteredCandidates.length > 1 + ? findBestCandidateInSameNamespace(filteredCandidates, filePath, settings) + : filteredCandidates[0] + if (!bestCandidateResult) { + return null + } + + return { + kind: "unlinked", + start: startIndex, + end: startIndex + longestMatch.length, + text: longestMatch.word, + candidateKey: longestMatch.key, + candidateData, + isInTable: isIndexInsideMarkdownTable(text, startIndex), + } +} + +const collectUnlinkedOccurrences = ({ + text, + filePath, + trie, + candidateMap, + fallbackIndex, + currentNamespace, + settings, + occurrences, +}: { + text: string + filePath: string + trie: TrieNode + candidateMap: Map + fallbackIndex: Map> + currentNamespace: string + settings: ReplaceLinksSettings + occurrences: CandidateOccurrence[] +}): void => { + let i = 0 + + outer: while (i < text.length) { + if (text[i] === "h" && text.slice(i, i + 4) === "http") { + const urlMatch = text.slice(i).match(REGEX_PATTERNS.URL) + if (urlMatch) { + i += urlMatch[0].length + continue + } + } + + let node = trie + let lastCandidate: { candidate: string, length: number } | null = null + let j = i + let candidateBuilder = "" + + while (j < text.length) { + const ch = text[j] + let chLower = settings.ignoreCase ? ch.toLowerCase() : ch + if (settings.matchSentenceCase && !settings.ignoreCase && j === i && isSentenceStart(text, i)) { + chLower = ch.toLowerCase() + } + candidateBuilder += ch + + const child = node.children.get(chLower) + if (!child) break + + node = child + if (node.candidate) { + const candidateIsCjk = isCjkCandidate(candidateBuilder) + if (candidateIsCjk || isWordBoundary(text[j + 1])) { + lastCandidate = { + candidate: node.candidate, + length: j - i + 1, + } + } + } + j++ + } + + if (lastCandidate) { + const candidate = candidateBuilder.slice(0, lastCandidate.length) + + if (shouldSkipCandidate(candidate, settings)) { + i += lastCandidate.length + continue + } + + const trieCandidateKey = lastCandidate.candidate + const candidateData = candidateMap.get(trieCandidateKey) + + if (candidateData) { + const filteredCandidateData = filterCandidateDataForNamespace( + candidateData, + currentNamespace, + settings, + ) + + if (filteredCandidateData.candidates.length === 0) { + i += candidate.length + continue + } + + if (isSelfLink(filteredCandidateData, filePath, settings)) { + i += candidate.length + continue + } + + const candidateIsCjk = isCjkCandidate(candidate) + if (!candidateIsCjk) { + const left = i > 0 ? text[i - 1] : undefined + const right = i + candidate.length < text.length + ? text[i + candidate.length] + : undefined + + if (!isWordBoundary(left) || !isWordBoundary(right)) { + i++ + continue + } + } + + occurrences.push({ + kind: "unlinked", + start: i, + end: i + candidate.length, + text: candidate, + candidateKey: trieCandidateKey, + candidateData: filteredCandidateData, + isInTable: isIndexInsideMarkdownTable(text, i), + }) + i += candidate.length + continue + } + } + + if (settings.proximityBasedLinking) { + const fallbackOccurrence = collectFallbackOccurrence({ + text, + startIndex: i, + fallbackIndex, + filePath, + currentNamespace, + settings, + }) + if (fallbackOccurrence) { + occurrences.push(fallbackOccurrence) + i = fallbackOccurrence.end + continue outer + } + } + + i++ + } +} + +export const scanCandidateOccurrences = ({ + text, + filePath, + trie, + candidateMap, + settings = {}, +}: ScanCandidateOccurrencesOptions): CandidateOccurrence[] => { + const occurrences: CandidateOccurrence[] = [] + const normalizedText = text.normalize("NFC") + const fallbackIndex = buildFallbackIndex(candidateMap, settings.ignoreCase) + const currentNamespace = getCurrentNamespace(filePath, settings.baseDir) + + collectExistingWikilinks(normalizedText, candidateMap, occurrences, settings) + + const protectedFreeOccurrences: CandidateOccurrence[] = [] + REGEX_PATTERNS.PROTECTED.lastIndex = 0 + let lastIndex = 0 + let match: RegExpExecArray | null + while ((match = REGEX_PATTERNS.PROTECTED.exec(normalizedText)) !== null) { + const segment = normalizedText.slice(lastIndex, match.index) + const segmentOccurrences: CandidateOccurrence[] = [] + collectUnlinkedOccurrences({ + text: segment, + filePath, + trie, + candidateMap, + fallbackIndex, + currentNamespace, + settings, + occurrences: segmentOccurrences, + }) + for (const occurrence of segmentOccurrences) { + protectedFreeOccurrences.push({ + ...occurrence, + start: occurrence.start + lastIndex, + end: occurrence.end + lastIndex, + }) + } + lastIndex = match.index + match[0].length + + if (match[0].length === 0) { + REGEX_PATTERNS.PROTECTED.lastIndex++ + } + } + + const tailOccurrences: CandidateOccurrence[] = [] + collectUnlinkedOccurrences({ + text: normalizedText.slice(lastIndex), + filePath, + trie, + candidateMap, + fallbackIndex, + currentNamespace, + settings, + occurrences: tailOccurrences, + }) + for (const occurrence of tailOccurrences) { + protectedFreeOccurrences.push({ + ...occurrence, + start: occurrence.start + lastIndex, + end: occurrence.end + lastIndex, + }) + } + + occurrences.push(...protectedFreeOccurrences) + + return occurrences.sort((a, b) => a.start - b.start) +} + +export const getOccurrenceContext = ( + text: string, + occurrence: CandidateOccurrence, + maxContext: number, +): string => { + const start = Math.max(0, occurrence.start - maxContext) + const end = Math.min(text.length, occurrence.end + maxContext) + return text.slice(start, end) +} diff --git a/src/replace-links/replace-links.ts b/src/replace-links/replace-links.ts index daf892b..afddcfc 100644 --- a/src/replace-links/replace-links.ts +++ b/src/replace-links/replace-links.ts @@ -1,4 +1,24 @@ -import { CandidateData, getTopLevelDirectoryName, TrieNode } from "../trie" +import { CandidateData, TrieNode } from "../trie" +import { + buildFallbackIndex, + extractFencedCodeBlocks, + extractLinkParts, + findBestCandidateInSameNamespace, + getCurrentNamespace, + isCjkCandidate, + isCjkText, + isIndexInsideMarkdownTable, + isKoreanText, + isMarkdownTableLine, + isMonthNote, + isProtectedLink, + isSelfLink, + isSentenceStart, + isWordBoundary, + normalizeCanonicalPath, + REGEX_PATTERNS, + shouldSkipCandidate, +} from "./candidate-scanner" // Types for the replaceLinks function export interface LinkResolverContext { @@ -36,214 +56,6 @@ export interface ReplaceLinksOptions { resolvedAmbiguities?: Map } -// Constants and Regular Expressions -const REGEX_PATTERNS = { - PROTECTED: /(```[\s\S]*?```|`[^`]*`|\[\[([^\]]+)\]\]|\[[^\]]+\]\([^)]+\)|\[[^\]]+\]|https?:\/\/[^\s]+)/g, - DATE_FORMAT: /^\d{4}-\d{2}-\d{2}$/, - MONTH_NOTE: /^[0-9]{1,2}$/, - CJK: /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}]/u, - CJK_CANDIDATE: /^[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}\s\d]+$/u, - KOREAN: /^[\p{Script=Hangul}]+$/u, - JAPANESE: /^[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\s\d]+$/u, - URL: /^(https?:\/\/[^\s]+)/, - PROTECTED_LINK: /^\s*(\[\[[^\]]+\]\]|\[[^\]]+\]\([^)]+\))\s*$/, - KOREAN_SUFFIX: /^(이다\.?)/, - KOREAN_PARTICLES: /^(는|은)/, - KOREAN_PARTICLES_EXTENDED: /^(가|는|을|에|서|와|로부터|까지|보다|로|의|나|도|또한)/, - TABLE_SEPARATOR: /^[|:\s-]+$/, - WORD_BOUNDARY: /[\p{L}\p{N}_/-]/u, - WHITESPACE: /[\t\n\r ]/, -} as const - -// Text Analysis Utilities -const isWordBoundary = (char: string | undefined): boolean => { - if (char === undefined) return true - if (REGEX_PATTERNS.CJK.test(char)) return true - return (!REGEX_PATTERNS.WORD_BOUNDARY.test(char) || REGEX_PATTERNS.WHITESPACE.test(char)) -} - -const isMonthNote = (candidate: string): boolean => - REGEX_PATTERNS.MONTH_NOTE.test(candidate) - && parseInt(candidate, 10) >= 1 - && parseInt(candidate, 10) <= 12 - -const isProtectedLink = (body: string): boolean => REGEX_PATTERNS.PROTECTED_LINK.test(body) - -const isCjkText = (text: string): boolean => REGEX_PATTERNS.CJK.test(text) - -const isCjkCandidate = (candidate: string): boolean => REGEX_PATTERNS.CJK_CANDIDATE.test(candidate) - -const isKoreanText = (text: string): boolean => REGEX_PATTERNS.KOREAN.test(text) - -const isSentenceStart = (text: string, index: number): boolean => { - if (index === 0) return true - if (text[index - 1] === "\n") return true - // ". " pattern (after period + space) - // Avoid acronyms by checking the char before the period is a letter - if (index >= 3 && text[index - 1] === " " && text[index - 2] === ".") { - const charBeforePeriod = text[index - 3] - if (/[a-zA-Z]/.test(charBeforePeriod)) { - return true - } - } - return false -} - -// Cache for fallback index to avoid rebuilding -const fallbackIndexCache = new WeakMap< - Map, - Map>> ->() - -const buildFallbackIndex = ( - candidateMap: Map, - ignoreCase?: boolean, -): Map> => { - // Get or create cache for this candidateMap - let cacheForMap = fallbackIndexCache.get(candidateMap) - if (!cacheForMap) { - cacheForMap = new Map() - fallbackIndexCache.set(candidateMap, cacheForMap) - } - - // Check if we have cached result for this ignoreCase setting - const cacheKey = ignoreCase ? "ignoreCase" : "normal" - const cached = cacheForMap.get(cacheKey) - if (cached) return cached - - // Build new fallback index - const fallbackIndex = new Map>() - - for (const [key, data] of candidateMap.entries()) { - const slashIndex = key.lastIndexOf("/") - if (slashIndex === -1) continue - - const shorthand = key.slice(slashIndex + 1) - const indexKey = ignoreCase ? shorthand.toLowerCase() : shorthand - - let arr = fallbackIndex.get(indexKey) - if (!arr) { - arr = [] - fallbackIndex.set(indexKey, arr) - } - arr.push([key, data]) - } - - // Cache the result - cacheForMap.set(cacheKey, fallbackIndex) - return fallbackIndex -} - -const getCurrentNamespace = (filePath: string, baseDir?: string): string => { - if (baseDir) { - return getTopLevelDirectoryName(filePath, baseDir) - } - - const segments = filePath.split("/") - return segments[0] || "" -} - -const normalizeCanonicalPath = (linkPath: string, baseDir?: string): string => { - if (baseDir && linkPath.startsWith(baseDir + "/")) { - return linkPath.slice((baseDir + "/").length) - } - return linkPath -} - -const extractLinkParts = ( - canonicalPath: string, -): { linkPath: string, alias: string, hasAlias: boolean } => { - const pipeIndex = canonicalPath.indexOf("|") - const hasAlias = pipeIndex !== -1 - - if (hasAlias) { - const linkPath = canonicalPath.slice(0, pipeIndex) - const alias = canonicalPath.slice(pipeIndex + 1) - return { linkPath, alias, hasAlias } - } - - return { linkPath: canonicalPath, alias: "", hasAlias } -} - -const escapeRegExp = (text: string): string => - text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&") - -const extractFencedCodeBlocks = ( - body: string, -): { body: string, codeBlocks: Array<{ placeholder: string, content: string }> } => { - if (!body.includes("```") && !body.includes("~~~")) { - return { body, codeBlocks: [] } - } - - const codeBlocks: Array<{ placeholder: string, content: string }> = [] - let result = "" - let cursor = 0 - let codeBlockIndex = 0 - - const openingFencePattern = /^ {0,3}(`{3,}|~{3,})[^\r\n]*(?:\r?\n|$)/gm - let openingMatch: RegExpExecArray | null - - while ((openingMatch = openingFencePattern.exec(body)) !== null) { - const start = openingMatch.index - if (start < cursor) { - continue - } - - const openingFence = openingMatch[1] - const fenceChar = openingFence[0] - const fenceLength = openingFence.length - const closingFencePattern = new RegExp( - `^ {0,3}${escapeRegExp(fenceChar)}{${fenceLength},}[ \\t]*(?:\\r?\\n|$)`, - "gm", - ) - closingFencePattern.lastIndex = openingMatch.index + openingMatch[0].length - const closingMatch = closingFencePattern.exec(body) - const end = closingMatch - ? closingMatch.index + closingMatch[0].length - : body.length - - const placeholder = `__CODE_BLOCK_${codeBlockIndex}__` - codeBlocks.push({ - placeholder, - content: body.slice(start, end), - }) - result += body.slice(cursor, start) + placeholder - cursor = end - codeBlockIndex++ - openingFencePattern.lastIndex = end - } - - result += body.slice(cursor) - return { body: result, codeBlocks } -} - -// Self-linking Prevention -const isSelfLink = ( - candidateData: CandidateData, - currentFilePath: string, - settings: ReplaceLinksSettings = {}, -): boolean => { - if (!settings.preventSelfLinking || candidateData.candidates.length === 0) { - return false - } - - // Extract the link path from the canonical path - const { linkPath } = extractLinkParts(candidateData.candidates[0].canonical) - - // Normalize paths for comparison - const normalizedLinkPath = normalizeCanonicalPath( - linkPath, - settings.baseDir, - ) - const normalizedCurrentPath = normalizeCanonicalPath( - currentFilePath, - settings.baseDir, - ) - - // Compare the paths - return normalizedLinkPath === normalizedCurrentPath -} - // Helper function to check if a path should have its alias removed const shouldRemoveAlias = ( normalizedPath: string, @@ -371,55 +183,6 @@ export const defaultLinkGenerator: LinkGenerator = ({ return escapeLinkForMarkdownTable(`[[${linkContent}]]`, isInTable) } -// Candidate Validation -const shouldSkipCandidate = ( - candidate: string, - settings: ReplaceLinksSettings, -): boolean => { - if ( - settings.ignoreDateFormats - && REGEX_PATTERNS.DATE_FORMAT.test(candidate) - ) { - return true - } - return isMonthNote(candidate) -} - -// Markdown Table Detection -const isMarkdownTableLine = (line: string): boolean => { - const trimmedLine = line.trim() - if (!trimmedLine || !trimmedLine.includes("|")) { - return false - } - - if (trimmedLine.startsWith("|") - && trimmedLine.endsWith("|") - && REGEX_PATTERNS.TABLE_SEPARATOR.test(trimmedLine) - ) { - return true - } - - return trimmedLine.startsWith("|") && trimmedLine.endsWith("|") -} - -const isIndexInsideMarkdownTable = (text: string, index: number): boolean => { - // Find the start of the line containing the index - let lineStart = text.lastIndexOf("\n", index - 1) + 1 - if (lineStart === 0 && text[0] !== "\n") { - lineStart = 0 - } - - // Find the end of the line containing the index - let lineEnd = text.indexOf("\n", index) - if (lineEnd === -1) { - lineEnd = text.length - } - - // Extract the line and check if it's a table line - const line = text.slice(lineStart, lineEnd) - return isMarkdownTableLine(line) -} - // Processing functions for different text types const processCjkText = ( text: string, @@ -630,86 +393,6 @@ const handleKoreanSpecialCases = ( return null } -const findBestCandidateInSameNamespace = ( - filteredCandidates: Array<[string, CandidateData]>, - filePath: string, - settings: ReplaceLinksSettings = {}, -): [string, CandidateData] | null => { - let bestCandidate: [string, CandidateData] | null = null - let bestScore = -1 - - // Get the directory portion of the current file (if any) - const filePathDir = filePath.includes("/") - ? filePath.slice(0, filePath.lastIndexOf("/")) - : "" - const filePathSegments = filePathDir ? filePathDir.split("/") : [] - - for (const [key, data] of filteredCandidates) { - const slashIndex = key.lastIndexOf("/") - const candidateDir = key.slice(0, slashIndex) - const candidateSegments = candidateDir.split("/") - let score = 0 - - // Calculate common prefix score - for ( - let idx = 0; - idx < Math.min(candidateSegments.length, filePathSegments.length); - idx++ - ) { - if (candidateSegments[idx] === filePathSegments[idx]) { - score++ - } - else { - break - } - } - - if (score > bestScore) { - bestScore = score - bestCandidate = [key, data] - } - else if (score === bestScore && bestCandidate !== null) { - if (filePathDir === "" && settings.baseDir) { - // When the current file is in the base directory, compare candidates by relative depth - const basePrefix = settings.baseDir + "/" - const getRelativeDepth = (k: string): number => { - if (k.startsWith(basePrefix)) { - // Remove the baseDir part and count the remaining segments - const relativeParts = k - .slice(basePrefix.length) - .split("/") - return relativeParts.length - 1 - } - return Infinity - } - - const candidateDepth = getRelativeDepth(key) - const bestCandidateDepth = getRelativeDepth(bestCandidate[0]) - - // Prefer the candidate with lower depth or shorter path - if (candidateDepth < bestCandidateDepth - || (candidateDepth === bestCandidateDepth && key.length < bestCandidate[0].length)) { - bestCandidate = [key, data] - } - } - else { - // Otherwise, choose the candidate with fewer directory segments - const currentBestDir = bestCandidate[0].slice(0, bestCandidate[0].lastIndexOf("/")) - const currentBestSegments = currentBestDir.split("/") - - if ( - candidateSegments.length < currentBestSegments.length - || (candidateSegments.length === currentBestSegments.length && key.length < bestCandidate[0].length) - ) { - bestCandidate = [key, data] - } - } - } - } - - return bestCandidate -} - const processStandardText = ( text: string, trie: TrieNode, diff --git a/src/utils/resolve-ambiguities.ts b/src/utils/resolve-ambiguities.ts index 831a947..68cae63 100644 --- a/src/utils/resolve-ambiguities.ts +++ b/src/utils/resolve-ambiguities.ts @@ -1,79 +1,33 @@ import { CandidateData, TrieNode } from "../trie" import { AutomaticLinkerSettings } from "../settings/settings-info" import { resolveAmbiguitiesBatch, AIResolveRequest } from "./ai-client" +import { + getOccurrenceContext, + scanCandidateOccurrences, +} from "../replace-links/candidate-scanner" -// A simplified scanner to find candidates that have multiple options or existing links to verify. export const resolveAmbiguities = async ( text: string, candidateMap: Map, trie: TrieNode, settings: AutomaticLinkerSettings, ): Promise> => { - const requests: AIResolveRequest[] = [] + const occurrences = scanCandidateOccurrences({ + text, + filePath: "", + trie, + candidateMap, + settings, + }) - // 1. Scan for existing links [[Path|Alias]] or [[Path]] - const existingLinkRegex = /\[\[([^|\]]+)(?:\|([^\]]+))?\]\]/g - let match: RegExpExecArray | null - while ((match = existingLinkRegex.exec(text)) !== null) { - const fullMatch = match[0] - const path = match[1] - const alias = match[2] || path + const requests: AIResolveRequest[] = occurrences + .filter(occurrence => occurrence.candidateData.candidates.length > 1) + .map(occurrence => ({ + word: occurrence.text, + text: getOccurrenceContext(text, occurrence, settings.aiMaxContext), + candidates: occurrence.candidateData.candidates.map(c => c.canonical), + })) - const candidateData = candidateMap.get(alias) - if (candidateData && candidateData.candidates.length > 1) { - const start = Math.max(0, match.index - settings.aiMaxContext) - const end = Math.min(text.length, match.index + fullMatch.length + settings.aiMaxContext) - - requests.push({ - word: fullMatch, // Using the full link as the "word" key for replacement - text: text.slice(start, end), - candidates: candidateData.candidates.map(c => c.canonical), - }) - } - } - - // 2. Scan for unlinked words with multiple candidates using the Trie - // (This is a simplified version of the trie traversal logic) - let i = 0 - while (i < text.length) { - let node = trie - let j = i - let lastMatch: { word: string, data: CandidateData, index: number } | null = null - - while (j < text.length && node.children.has(text[j].toLowerCase())) { - node = node.children.get(text[j].toLowerCase())! - if (node.candidate) { - const data = candidateMap.get(node.candidate) - if (data) { - lastMatch = { word: node.candidate, data, index: i } - } - } - j++ - } - - if (lastMatch && lastMatch.data.candidates.length > 1) { - // Check if it's already inside a link (simple check) - const isInsideLink = text.slice(Math.max(0, i - 2), i) === "[[" - || text.slice(i + lastMatch.word.length, i + lastMatch.word.length + 2) === "]]" - - if (!isInsideLink) { - const start = Math.max(0, i - settings.aiMaxContext) - const end = Math.min(text.length, i + lastMatch.word.length + settings.aiMaxContext) - - requests.push({ - word: lastMatch.word, - text: text.slice(start, end), - candidates: lastMatch.data.candidates.map(c => c.canonical), - }) - } - i += lastMatch.word.length - } - else { - i++ - } - } - - // Deduplicate requests by word to avoid redundant AI calls const uniqueRequests = Array.from(new Map(requests.map(r => [r.word, r])).values()) return await resolveAmbiguitiesBatch(settings, uniqueRequests)