From 9209356ee0d5590616a4bb74de80d25ef76baec6 Mon Sep 17 00:00:00 2001 From: Kodai Nakamura Date: Mon, 22 Jun 2026 02:57:16 +0900 Subject: [PATCH] refactor(markdown): centralize protected segment handling Why: - Link replacement and URL title code each carried their own Markdown context checks, which made protected text behavior hard to keep consistent. - A shared Markdown segment module improves locality for prose-only transformations. - The refactor needed to preserve existing wikilink correction and table behavior without bringing back placeholder collisions. What: - Add a pure Markdown segment module with prose mapping and focused tests. - Route link replacement and URL title flows through shared protected segment handling. - Preserve existing-link correction, heading/callout/table protection, and add fast paths to keep performance within the existing thresholds. --- src/__tests__/markdown-segments.test.ts | 61 +++++ src/markdown-segments.ts | 253 ++++++++++++++++++ src/replace-links/replace-links.ts | 199 ++++++-------- src/replace-url-with-title/index.ts | 128 ++++----- .../utils/list-up-all-urls.ts | 192 ++++--------- 5 files changed, 503 insertions(+), 330 deletions(-) create mode 100644 src/__tests__/markdown-segments.test.ts create mode 100644 src/markdown-segments.ts diff --git a/src/__tests__/markdown-segments.test.ts b/src/__tests__/markdown-segments.test.ts new file mode 100644 index 0000000..3d9821f --- /dev/null +++ b/src/__tests__/markdown-segments.test.ts @@ -0,0 +1,61 @@ +import { describe, expect, it } from "vitest" +import { mapMarkdownProse, segmentMarkdown } from "../markdown-segments" + +describe("segmentMarkdown", () => { + it("round-trips prose and protected inline code", () => { + const text = "Use `TypeScript` with TypeScript" + const segments = segmentMarkdown(text) + + expect(segments.map(segment => ({ + kind: segment.kind, + protectedKind: segment.protectedKind, + text: segment.text, + }))).toEqual([ + { kind: "prose", protectedKind: undefined, text: "Use " }, + { kind: "protected", protectedKind: "inline-code", text: "`TypeScript`" }, + { kind: "prose", protectedKind: undefined, text: " with TypeScript" }, + ]) + expect(segments.map(segment => segment.text).join("")).toBe(text) + }) + + it("protects fenced code blocks including unclosed blocks", () => { + const text = "before\n```ts\nTypeScript" + const segments = segmentMarkdown(text) + + expect(segments.map(segment => ({ + kind: segment.kind, + protectedKind: segment.protectedKind, + text: segment.text, + }))).toEqual([ + { kind: "prose", protectedKind: undefined, text: "before\n" }, + { kind: "protected", protectedKind: "fenced-code", text: "```ts\nTypeScript" }, + ]) + }) + + it("protects headings, tables, and callouts when requested", () => { + const text = "# TypeScript\n| TypeScript |\n> [!note]\n> TypeScript\nTypeScript" + const segments = segmentMarkdown(text, { + protectHeadings: true, + protectTableRows: true, + protectCallouts: true, + }) + + expect(segments.filter(segment => segment.kind === "protected").map(segment => segment.protectedKind)).toEqual([ + "heading", + "table-row", + "callout", + ]) + expect(segments.map(segment => segment.text).join("")).toBe(text) + }) +}) + +describe("mapMarkdownProse", () => { + it("transforms only prose segments", () => { + const result = mapMarkdownProse( + "TypeScript `TypeScript` [[TypeScript]]", + text => text.replace(/TypeScript/g, "TS"), + ) + + expect(result).toBe("TS `TypeScript` [[TypeScript]]") + }) +}) diff --git a/src/markdown-segments.ts b/src/markdown-segments.ts new file mode 100644 index 0000000..bffa45d --- /dev/null +++ b/src/markdown-segments.ts @@ -0,0 +1,253 @@ +import { isMarkdownTableLine } from "./replace-links/candidate-scanner" + +export type MarkdownSegmentKind = "prose" | "protected" + +export type MarkdownProtectedKind = "inline-code" + | "fenced-code" + | "wikilink" + | "markdown-link" + | "single-bracket" + | "url" + | "heading" + | "callout" + | "table-row" + +export interface MarkdownSegment { + kind: MarkdownSegmentKind + protectedKind?: MarkdownProtectedKind + start: number + end: number + text: string +} + +export interface SegmentMarkdownOptions { + protectHeadings?: boolean + protectCallouts?: boolean + protectTableRows?: boolean + protectUrls?: boolean +} + +interface ProtectedRange { + start: number + end: number + protectedKind: MarkdownProtectedKind +} + +const collectHeadingRanges = (text: string): ProtectedRange[] => { + const ranges: ProtectedRange[] = [] + const headingPattern = /^#{1,6}\s+.*$/gm + let match: RegExpExecArray | null + + while ((match = headingPattern.exec(text)) !== null) { + ranges.push({ + start: match.index, + end: match.index + match[0].length, + protectedKind: "heading", + }) + } + + return ranges +} + +const collectCalloutRanges = (text: string): ProtectedRange[] => { + const ranges: ProtectedRange[] = [] + const calloutPattern = /^>[ \t]*\[![\w-]+\].*?(\n>.*?)*(?=\n(?!>)|$)/gm + let match: RegExpExecArray | null + + while ((match = calloutPattern.exec(text)) !== null) { + ranges.push({ + start: match.index, + end: match.index + match[0].length, + protectedKind: "callout", + }) + } + + return ranges +} + +const collectTableRowRanges = (text: string): ProtectedRange[] => { + const ranges: ProtectedRange[] = [] + const linePattern = /[^\n]*(?:\n|$)/g + let match: RegExpExecArray | null + + while ((match = linePattern.exec(text)) !== null) { + if (match[0] === "") { + break + } + + const lineText = match[0] + const lineContent = lineText.endsWith("\n") + ? lineText.slice(0, -1).replace(/\r$/, "") + : lineText.replace(/\r$/, "") + + if (isMarkdownTableLine(lineContent)) { + ranges.push({ + start: match.index, + end: match.index + lineText.length, + protectedKind: "table-row", + }) + } + } + + return ranges +} + +const buildProtectedPattern = (protectUrls: boolean): RegExp => { + const parts = [ + "```[\\s\\S]*?(?:```|$)", + "~~~[\\s\\S]*?(?:~~~|$)", + "`[^`]*`", + "\\[\\[[^\\]]+\\]\\]", + "\\[[^\\]]+\\]\\([^)]+\\)", + "\\[[^\\]]+\\]", + ] + + if (protectUrls) { + parts.push("https?:\\/\\/[^\\s]+") + } + + return new RegExp(`(${parts.join("|")})`, "g") +} + +const getProtectedKind = (text: string): MarkdownProtectedKind => { + if (text.startsWith("```") || text.startsWith("~~~")) { + return "fenced-code" + } + + if (text.startsWith("`")) { + return "inline-code" + } + + if (text.startsWith("[[")) { + return "wikilink" + } + + if (text.startsWith("[")) { + return text.includes("](") ? "markdown-link" : "single-bracket" + } + + return "url" +} + +const sortAndMergeRanges = (ranges: ProtectedRange[]): ProtectedRange[] => { + const sortedRanges = ranges + .slice() + .sort((a, b) => a.start - b.start || a.end - b.end) + const merged: ProtectedRange[] = [] + + for (const range of sortedRanges) { + const lastRange = merged[merged.length - 1] + if (!lastRange || range.start >= lastRange.end) { + merged.push({ ...range }) + continue + } + + lastRange.end = Math.max(lastRange.end, range.end) + } + + return merged +} + +const isInsideRanges = ( + index: number, + ranges: ProtectedRange[], +): boolean => { + return ranges.some(range => index >= range.start && index < range.end) +} + +export const segmentMarkdown = ( + text: string, + options: SegmentMarkdownOptions = {}, +): MarkdownSegment[] => { + const mayContainProtectedMarkdown = text.includes("`") + || text.includes("[") + || (options.protectHeadings && text.includes("#")) + || (options.protectCallouts && text.includes(">")) + || (options.protectTableRows && text.includes("|")) + || (options.protectUrls && text.includes("http")) + + if (!mayContainProtectedMarkdown) { + return [{ + kind: "prose", + start: 0, + end: text.length, + text, + }] + } + + const ranges: ProtectedRange[] = [] + + if (options.protectHeadings) { + ranges.push(...collectHeadingRanges(text)) + } + + if (options.protectCallouts) { + ranges.push(...collectCalloutRanges(text)) + } + + if (options.protectTableRows) { + ranges.push(...collectTableRowRanges(text)) + } + + const protectedPattern = buildProtectedPattern(options.protectUrls ?? false) + let match: RegExpExecArray | null + + while ((match = protectedPattern.exec(text)) !== null) { + if (isInsideRanges(match.index, ranges)) { + continue + } + + ranges.push({ + start: match.index, + end: match.index + match[0].length, + protectedKind: getProtectedKind(match[0]), + }) + } + + const mergedRanges = sortAndMergeRanges(ranges) + const segments: MarkdownSegment[] = [] + let cursor = 0 + + for (const range of mergedRanges) { + if (cursor < range.start) { + segments.push({ + kind: "prose", + start: cursor, + end: range.start, + text: text.slice(cursor, range.start), + }) + } + + segments.push({ + kind: "protected", + protectedKind: range.protectedKind, + start: range.start, + end: range.end, + text: text.slice(range.start, range.end), + }) + cursor = range.end + } + + if (cursor < text.length || segments.length === 0) { + segments.push({ + kind: "prose", + start: cursor, + end: text.length, + text: text.slice(cursor), + }) + } + + return segments +} + +export const mapMarkdownProse = ( + text: string, + transform: (segmentText: string, segment: MarkdownSegment) => string, + options: SegmentMarkdownOptions = {}, +): string => { + return segmentMarkdown(text, options) + .map(segment => segment.kind === "prose" + ? transform(segment.text, segment) + : segment.text) + .join("") +} diff --git a/src/replace-links/replace-links.ts b/src/replace-links/replace-links.ts index a008413..4800dbf 100644 --- a/src/replace-links/replace-links.ts +++ b/src/replace-links/replace-links.ts @@ -1,7 +1,7 @@ import { CandidateData, TrieNode } from "../trie" +import { mapMarkdownProse, segmentMarkdown } from "../markdown-segments" import { buildFallbackIndex, - extractFencedCodeBlocks, extractLinkParts, findBestCandidateInSameNamespace, getCurrentNamespace, @@ -9,7 +9,6 @@ import { isCjkText, isIndexInsideMarkdownTable, isKoreanText, - isMarkdownTableLine, isMonthNote, isProtectedLink, isSelfLink, @@ -211,6 +210,7 @@ const processCjkText = ( linkGenerator: LinkGenerator, settings: ReplaceLinksSettings = {}, resolvedAmbiguities?: Map, + forceIsInTable?: boolean, ): string => { // For CJK texts that might contain non-CJK terms like "taro-san", ensure we use a consistent approach // Pass the proper filePath to maintain correct namespace resolution @@ -224,6 +224,7 @@ const processCjkText = ( linkGenerator, settings, resolvedAmbiguities, + forceIsInTable, ) } @@ -236,6 +237,7 @@ const processFallbackSearch = ( currentNamespace: string, linkGenerator: LinkGenerator, settings: ReplaceLinksSettings, + forceIsInTable?: boolean, ): { result: string, newIndex: number } | null => { // Early boundary check - if start isn't a word boundary, skip const prevChar = text[startIndex - 1] @@ -345,7 +347,7 @@ const processFallbackSearch = ( longestMatch.word, settings, ) - const isInTable = isIndexInsideMarkdownTable(text, startIndex) + const isInTable = forceIsInTable ?? isIndexInsideMarkdownTable(text, startIndex) const finalLink = linkGenerator({ linkPath, sourcePath: filePath, @@ -369,6 +371,7 @@ const handleKoreanSpecialCases = ( linkGenerator: LinkGenerator, settings: ReplaceLinksSettings = {}, resolvedAmbiguities?: Map, + forceIsInTable?: boolean, ): { result: string, newIndex: number } | null => { const remaining = text.slice(i + candidate.length) @@ -393,7 +396,7 @@ const handleKoreanSpecialCases = ( linkPath, sourcePath: filePath, alias, - isInTable: false, + isInTable: forceIsInTable ?? false, }) return { @@ -423,6 +426,7 @@ const processStandardText = ( linkGenerator: LinkGenerator, settings: ReplaceLinksSettings = {}, resolvedAmbiguities?: Map, + forceIsInTable?: boolean, ): string => { let result = "" let i = 0 @@ -517,6 +521,7 @@ const processStandardText = ( linkGenerator, settings, resolvedAmbiguities, + forceIsInTable, ) if (koreanResult) { result += koreanResult.result @@ -581,7 +586,7 @@ const processStandardText = ( linkPath, sourcePath: filePath, alias, - isInTable, + isInTable: forceIsInTable ?? isInTable, }) result += finalLink @@ -600,6 +605,7 @@ const processStandardText = ( currentNamespace, linkGenerator, settings, + forceIsInTable, ) if (fallbackResult) { result += fallbackResult.result @@ -642,8 +648,43 @@ export const replaceLinks = ({ // Get the current namespace const currentNamespace = getCurrentNamespace(filePath, settings.baseDir) + const markdownOptions = { + protectHeadings: settings.ignoreHeadings, + protectCallouts: true, + protectTableRows: settings.ignoreMarkdownTables, + protectUrls: true, + } + + const replaceResolvedWikilink = ( + wikilink: string, + start: number, + ): string => { + if (!resolvedAmbiguities?.has(wikilink)) { + return wikilink + } + + const resolvedPath = resolvedAmbiguities.get(wikilink)! + const { linkPath, alias: resolvedAlias } = extractLinkParts(resolvedPath) + const existingLinkRegex = /\[\[([^|\]]+)(?:\|([^\]]+))?\]\]/ + const linkMatch = wikilink.match(existingLinkRegex) + const existingPath = linkMatch ? linkMatch[1] : "" + const existingAlias = linkMatch ? linkMatch[2] : undefined + const finalAlias = resolvedAlias || existingAlias || (wikilink.includes("|") ? undefined : existingPath) + + return linkGenerator({ + linkPath, + sourcePath: filePath, + alias: finalAlias, + isInTable: !settings.ignoreMarkdownTables + && isIndexInsideMarkdownTable(body, start), + }) + } + // Process segments of text - const processTextSegment = (text: string): string => { + const processTextSegment = ( + text: string, + forceIsInTable?: boolean, + ): string => { // Check if the text contains CJK characters const hasCjkText = isCjkText(text) @@ -657,6 +698,7 @@ export const replaceLinks = ({ linkGenerator, settings, resolvedAmbiguities, + forceIsInTable, ) } else { @@ -670,138 +712,61 @@ export const replaceLinks = ({ linkGenerator, settings, resolvedAmbiguities, + forceIsInTable, ) } } - const processTableAwareTextSegment = (text: string): string => { - if (!settings.ignoreMarkdownTables) { - return processTextSegment(text) + const processTableAwareTextSegment = ( + text: string, + segment: { start: number }, + ): string => { + if (!text.includes("\n")) { + const isInTable = !settings.ignoreMarkdownTables + && isIndexInsideMarkdownTable(body, segment.start) + return processTextSegment(text, isInTable) } - return text.replace(/[^\n]*(?:\n|$)/g, (line) => { + return text.replace(/[^\n]*(?:\n|$)/g, (line, offset) => { if (line === "") { return line } const lineContent = line.endsWith("\n") - ? line.slice(0, -1) + ? line.slice(0, -1).replace(/\r$/, "") : line - if (isMarkdownTableLine(lineContent)) { + if (lineContent === "") { return line } - return processTextSegment(line) + const absoluteIndex = segment.start + offset + const isInTable = !settings.ignoreMarkdownTables + && isIndexInsideMarkdownTable(body, absoluteIndex) + + return processTextSegment(line, isInTable) }) } - // Extract and protect fenced code blocks before any other block-level rules. - const { body: bodyAfterCodeBlocks, codeBlocks } = extractFencedCodeBlocks(body) + let bodyWithResolvedWikilinks = body + if (resolvedAmbiguities) { + bodyWithResolvedWikilinks = segmentMarkdown(body, markdownOptions) + .map((segment) => { + if ( + segment.kind === "protected" + && segment.protectedKind === "wikilink" + ) { + return replaceResolvedWikilink(segment.text, segment.start) + } - // Extract and protect headings first - const headingPattern = /^#{1,6}\s+.*$/gm - const headings: Array<{ placeholder: string, content: string }> = [] - let headingIndex = 0 - - let bodyAfterHeadings = bodyAfterCodeBlocks - if (settings.ignoreHeadings) { - bodyAfterHeadings = bodyAfterCodeBlocks.replace(headingPattern, (match) => { - const placeholder = `__HEADING_${headingIndex}__` - headings.push({ placeholder, content: match }) - headingIndex++ - return placeholder - }) - } - - // Extract and protect callout blocks first - // Match callout blocks: starts with > [!type] and continues with lines starting with > - const calloutPattern = /^>[ \t]*\[![\w-]+\].*?(\n>.*?)*(?=\n(?!>)|$)/gm - const callouts: Array<{ placeholder: string, content: string }> = [] - let calloutIndex = 0 - - // Replace callouts with placeholders - const bodyWithPlaceholders = bodyAfterHeadings.replace(calloutPattern, (match) => { - const placeholder = `__CALLOUT_${calloutIndex}__` - callouts.push({ placeholder, content: match }) - calloutIndex++ - return placeholder - }) - - // Process the entire body while preserving protected segments - let resultBody = "" - let lastIndex = 0 - let match: RegExpExecArray | null - - // Reset the regex to start from the beginning - REGEX_PATTERNS.PROTECTED.lastIndex = 0 - - while ( - (match = REGEX_PATTERNS.PROTECTED.exec(bodyWithPlaceholders)) !== null - ) { - const mIndex = match.index - const segment = bodyWithPlaceholders.slice(lastIndex, mIndex) - resultBody += processTableAwareTextSegment(segment) - - const fullMatch = match[0] - if ( - settings.ignoreMarkdownTables - && isIndexInsideMarkdownTable(bodyWithPlaceholders, mIndex) - ) { - resultBody += fullMatch - } - else if (resolvedAmbiguities?.has(fullMatch)) { - // Existing link replacement - const resolvedPath = resolvedAmbiguities.get(fullMatch)! - const { linkPath, alias: resolvedAlias } = extractLinkParts(resolvedPath) - - // Try to extract existing alias from the matched link - const existingLinkRegex = /\[\[([^|\]]+)(?:\|([^\]]+))?\]\]/ - const linkMatch = fullMatch.match(existingLinkRegex) - const existingPath = linkMatch ? linkMatch[1] : "" - const existingAlias = linkMatch ? linkMatch[2] : undefined - - // Use resolved alias if present, otherwise use existing alias, - // otherwise use existing path (as alias if it was a simple link) - const finalAlias = resolvedAlias || existingAlias || (fullMatch.includes("|") ? undefined : existingPath) - - const isInTable = isIndexInsideMarkdownTable(bodyWithPlaceholders, mIndex) - resultBody += linkGenerator({ - linkPath, - sourcePath: filePath, - alias: finalAlias, - isInTable, + return segment.text }) - } - else { - // Append the protected segment unchanged - resultBody += fullMatch - } - lastIndex = mIndex + fullMatch.length - - // Prevent infinite loop on zero-length matches - if (fullMatch.length === 0) { - REGEX_PATTERNS.PROTECTED.lastIndex++ - } + .join("") } - // Process the remaining text - resultBody += processTableAwareTextSegment(bodyWithPlaceholders.slice(lastIndex)) - - // Restore callouts - for (const { placeholder, content } of callouts) { - resultBody = resultBody.replace(placeholder, content) - } - - // Restore headings - for (const { placeholder, content } of headings) { - resultBody = resultBody.replace(placeholder, content) - } - - // Restore fenced code blocks - for (const { placeholder, content } of codeBlocks) { - resultBody = resultBody.replace(placeholder, content) - } - - return resultBody + return mapMarkdownProse( + bodyWithResolvedWikilinks, + processTableAwareTextSegment, + markdownOptions, + ) } diff --git a/src/replace-url-with-title/index.ts b/src/replace-url-with-title/index.ts index 831f650..7eca4cc 100644 --- a/src/replace-url-with-title/index.ts +++ b/src/replace-url-with-title/index.ts @@ -1,3 +1,5 @@ +import { mapMarkdownProse } from "../markdown-segments" + type Url = string type Title = string interface ReplaceUrlWithTitleOptions { @@ -13,103 +15,73 @@ export const replaceUrlWithTitle = ({ return body } - let resultBody = body - // Sort URLs by length descending to replace longer URLs first // This helps prevent partial replacements (e.g., replacing 'example.com' before 'sub.example.com') const sortedUrls = Array.from(urlTitleMap.keys()).sort( (a, b) => b.length - a.length, ) - for (const url of sortedUrls) { - const title = urlTitleMap.get(url) - // Should not happen with Map iteration, but good practice - if (!title) continue + const replaceUrlsInProse = (prose: string): string => { + let resultBody = prose - // Escape backslashes and special characters in title for link text safety if needed - // For now, assume title is safe. - const markdownLink = `[${title}](${url})` - let currentIndex = 0 - const newBodyParts: string[] = [] + for (const url of sortedUrls) { + const title = urlTitleMap.get(url) + if (!title) continue - // Find all occurrences of the current URL in the resultBody - // resultBody is updated in each iteration of the outer loop - while (currentIndex < resultBody.length) { - // Find the next occurrence of the URL, case-sensitive. - const nextOccurrence = resultBody.indexOf(url, currentIndex) + const markdownLink = `[${title}](${url})` + let currentIndex = 0 + const newBodyParts: string[] = [] - if (nextOccurrence === -1) { - // No more occurrences found, add the rest of the string - newBodyParts.push(resultBody.substring(currentIndex)) - break - } + while (currentIndex < resultBody.length) { + const nextOccurrence = resultBody.indexOf(url, currentIndex) - // Add the text segment before the match - newBodyParts.push( - resultBody.substring(currentIndex, nextOccurrence), - ) + if (nextOccurrence === -1) { + newBodyParts.push(resultBody.substring(currentIndex)) + break + } - // --- Context Check --- - let shouldReplace = true + newBodyParts.push( + resultBody.substring(currentIndex, nextOccurrence), + ) - // 1. Check if already part of a Markdown link: [...](url) - // Look for `](` immediately before the URL and `)` immediately after. - const precedingChars = resultBody.substring( - nextOccurrence - 2, - nextOccurrence, - ) - const followingChar = resultBody[nextOccurrence + url.length] - if (precedingChars === "](" && followingChar === ")") { - shouldReplace = false - } + let shouldReplace = true - // 2. Check if inside inline code: `... url ...` - // Count non-escaped backticks before the match. Odd count means inside code. - if (shouldReplace) { - const segmentBefore = resultBody.substring(0, nextOccurrence) - // Count non-escaped backticks `(?() let match - // --- Pre-calculate Fenced Code Block Ranges --- - const codeBlockRanges: { start: number, end: number }[] = [] - // Regex to find fenced code blocks (handles different fence lengths and optional language specifiers) - // Matches from ``` or ~~~ at the start of a line to the next ``` or ~~~ at the start of a line - const codeBlockRegex - = /^(?:```|~~~)[^\r\n]*?\r?\n([\s\S]*?)\r?\n^(?:```|~~~)$/gm - let blockMatch - while ((blockMatch = codeBlockRegex.exec(body)) !== null) { - codeBlockRanges.push({ - start: blockMatch.index, - end: blockMatch.index + blockMatch[0].length, - }) - } - // Reset regex state if needed, though new exec calls should handle this - codeBlockRegex.lastIndex = 0 + for (const segment of segmentMarkdown(body)) { + if (segment.kind === "protected") { + continue + } - while ((match = URL_REGEX.exec(body)) !== null) { - const url = match[0] - const matchIndex = match.index + URL_REGEX.lastIndex = 0 + while ((match = URL_REGEX.exec(segment.text)) !== null) { + const url = match[0] + const matchIndex = segment.start + match.index - // --- Fenced Code Block Check --- - // Check if the match index falls within any calculated code block range - let isInCodeBlock = false - for (const range of codeBlockRanges) { - if (matchIndex >= range.start && matchIndex < range.end) { - isInCodeBlock = true - break + let isBareUrl = true + + if (matchIndex >= 2) { + const followingCharIndex = matchIndex + url.length + if (followingCharIndex < body.length && body[followingCharIndex] === ")") { + const precedingChars = body.substring(matchIndex - 2, matchIndex) + if (precedingChars === "](") { + isBareUrl = false + } + } } - } - if (isInCodeBlock) { - continue // Skip this URL if it's inside a fenced code block - } - // --- Context Check --- - let isBareUrl = true - - // 1. Check if already part of a Markdown link: [...](url) - if (matchIndex >= 2) { // Need space for "](" - // Check if the URL is potentially followed by ')' - const followingCharIndex = matchIndex + url.length - if (followingCharIndex < body.length && body[followingCharIndex] === ")") { - // If followed by ')', check if preceded by "](" - const precedingChars = body.substring(matchIndex - 2, matchIndex) - if (precedingChars === "](") { - // Only if both conditions are met, it's a Markdown link + if (isBareUrl && matchIndex >= 1) { + const precedingChar = body[matchIndex - 1] + const followingChar = body[matchIndex + url.length] + if (precedingChar === "<" && followingChar === ">") { isBareUrl = false } } - } - // 2. Check if enclosed in angle brackets: - if (isBareUrl && matchIndex >= 1) { - const precedingChar = body[matchIndex - 1] - const followingChar = body[matchIndex + url.length] - if (precedingChar === "<" && followingChar === ">") { - isBareUrl = false - } - } + if (isBareUrl) { + let finalUrl = url + let shouldAdd = true - // 3. Check if inside inline code: `... url ...` - // Count non-escaped backticks before the match. Odd count means inside code. - if (isBareUrl) { - const segmentBefore = body.substring(0, matchIndex) - // Count non-escaped backticks `(? 0) { - try { - const parsedUrl = new URL(url) - const hostname = parsedUrl.hostname - if ( - ignoredDomains.some( - domain => - hostname === domain - || hostname.endsWith(`.${domain}`), + if (ignoredDomains && ignoredDomains.length > 0) { + try { + const parsedUrl = new URL(url) + const hostname = parsedUrl.hostname + if ( + ignoredDomains.some( + domain => + hostname === domain + || hostname.endsWith(`.${domain}`), + ) + ) { + shouldAdd = false + } + } + catch (e) { + console.warn( + `Failed to parse URL for domain check: ${url}`, + e, ) - ) { shouldAdd = false } } - catch (e) { - // If URL parsing fails, it's likely not a valid URL to add anyway - console.warn( - `Failed to parse URL for domain check: ${url}`, - e, - ) - shouldAdd = false - } - } - // 6. Clean Trailing Punctuation (only if not ignored) - // We only clean punctuation if the URL wasn't already excluded by context checks. - if (shouldAdd) { - const cleanedUrl = url.replace(TRAILING_PUNCTUATION_REGEX, "") - // Ensure cleaning didn't make it invalid (e.g., just "http://") - // or remove essential parts if the regex was too broad. - if (cleanedUrl.includes("://")) { - finalUrl = cleanedUrl // Use the cleaned URL only if it's still valid-looking + if (shouldAdd) { + const cleanedUrl = url.replace(TRAILING_PUNCTUATION_REGEX, "") + if (cleanedUrl.includes("://")) { + finalUrl = cleanedUrl + } + else { + finalUrl = url + console.warn(`URL cleaning potentially broke the URL: ${url} -> ${cleanedUrl}`) + } } - else { - // If cleaning resulted in an invalid URL, maybe don't add it, - // or reconsider the TRAILING_PUNCTUATION_REGEX. - // For now, let's stick with the original URL if cleaning fails badly. - // This case shouldn't happen often with the current regex. - finalUrl = url // Revert to original if cleaning broke it - console.warn(`URL cleaning potentially broke the URL: ${url} -> ${cleanedUrl}`) - // Decide if we should still add the original potentially dirty url - // shouldAdd = false; // Option: Don't add if cleaning failed - } - } - // --- Add URL if context checks pass and not ignored --- - if (shouldAdd) { - urls.add(finalUrl) + if (shouldAdd) { + urls.add(finalUrl) + } } } - - // Reset regex lastIndex to avoid issues with overlapping matches or zero-length matches - // Although our URL regex shouldn't produce zero-length matches. - // If the regex finds a match at index `i`, the next search starts at `i + 1`. - // If the match was length `l`, `exec` updates `lastIndex` to `i + l`. - // We need to ensure progress even if `l` is 0, but `URL_REGEX` won't match empty. - // If `isBareUrl` logic modified the string or indices, care would be needed. } return urls