diff --git a/src/scanner/scanners/external-links.ts b/src/scanner/scanners/external-links.ts index e620bbf..55da568 100644 --- a/src/scanner/scanners/external-links.ts +++ b/src/scanner/scanners/external-links.ts @@ -112,11 +112,15 @@ function isExternalUrl(text: string): boolean { export function extractBareUrls(content: string): string[] { const urls: string[] = []; const seen = new Set(); - const body = stripIgnoredMarkdownRegions(stripFrontmatter(content)); + // Markdown link syntax is masked so anything glued to it — emphasis + // markers, curly quotes, possessives — never reaches the bare extractor; + // those URLs are already collected from link metadata. + const body = stripIgnoredMarkdownRegions(stripFrontmatter(content)) + .replace(/!?\[[^\]\n]*\]\([ \t]*(?:<[^>\n]*>|[^)\s]*)[ \t]*\)/g, ""); const urlPattern = /https?:\/\/[^\s<>"']+/gi; for (const match of body.matchAll(urlPattern)) { - const url = trimUrlBoundary(match[0]); + const url = trimUrlBoundary(cutAtNonUrlCharacter(match[0])); if (!url || seen.has(url)) continue; seen.add(url); urls.push(url); @@ -125,6 +129,18 @@ export function extractBareUrls(content: string): string[] { return urls; } +// RFC 3986 URL characters plus unicode letters, numbers, and combining marks +// (unencoded non-ASCII punctuation in a URL is always prose contamination — +// conforming URLs must percent-encode it). +const urlCharacter = /[A-Za-z0-9\-._~!$&'()*+,;=:/?#%[\]@\p{L}\p{N}\p{M}]/u; + +function cutAtNonUrlCharacter(url: string): string { + for (let index = 0; index < url.length; index += 1) { + if (!urlCharacter.test(url[index])) return url.slice(0, index); + } + return url; +} + function stripFrontmatter(content: string): string { const match = /^---\r?\n[\s\S]*?\r?\n---(?:\r?\n|$)/.exec(content); return match ? content.slice(match[0].length) : content; diff --git a/src/tests/external-links.test.ts b/src/tests/external-links.test.ts index 51b6346..93bd95b 100644 --- a/src/tests/external-links.test.ts +++ b/src/tests/external-links.test.ts @@ -67,16 +67,35 @@ describe("externalLinksScanner", () => { }); it.each([ - ["*[Empis tessellata](https://en.wikipedia.org/wiki/Empis_tessellata)*", "https://en.wikipedia.org/wiki/Empis_tessellata"], - ["**[note](https://example.com/a)**", "https://example.com/a"], ["_https://example.com/a_", "https://example.com/a"], ["~~https://example.com/a~~", "https://example.com/a"], - ["[t](https://example.com/a_(b))*", "https://example.com/a_(b)"], ["https://example.com/a_", "https://example.com/a"], - ])("trims markdown emphasis markers after a URL: %s", (body, expected) => { + ])("trims markdown emphasis markers after a bare URL: %s", (body, expected) => { expect(extractBareUrls(body)).toEqual([expected]); }); + it.each([ + // Markdown link syntax is masked before bare extraction: those URLs are + // collected from link metadata, so anything glued to the syntax — + // emphasis, curly quotes, possessives — never reaches the extractor. + ["*[Empis tessellata](https://en.wikipedia.org/wiki/Empis_tessellata)*", []], + ["**[note](https://example.com/a)**", []], + ["[t](https://example.com/a_(b))*", []], + ["“[note](https://example.com/a)”", []], + ["*[t](https://x.com/a)*’s habitat", []], + // Nested brackets break the mask; the URL falls through to bare + // extraction cleanly and dedupes against the metadata link. + ["[a [b] c](https://example.com/a) leftover", ["https://example.com/a"]], + // Bare URLs cut at the first character that cannot appear in a URL; + // unicode letters/numbers (IDN paths) survive. + ["https://x.com/a’s page", ["https://x.com/a"]], + ["https://x.com/a”", ["https://x.com/a"]], + ["https://x.com/a」。", ["https://x.com/a"]], + ["https://ja.wikipedia.org/wiki/日本語", ["https://ja.wikipedia.org/wiki/日本語"]], + ])("handles unicode and markdown boundaries around URLs: %s", (body, expected) => { + expect(extractBareUrls(body)).toEqual(expected); + }); + it("checks an italic-wrapped markdown link as a single clean URL", async () => { const url = "https://en.wikipedia.org/wiki/Empis_tessellata"; const request = vi.fn(async (_url: string, method: "HEAD" | "GET") => ({ status: 200, method }));