Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 18 additions & 2 deletions src/scanner/scanners/external-links.ts
Original file line number Diff line number Diff line change
Expand Up @@ -112,11 +112,15 @@ function isExternalUrl(text: string): boolean {
export function extractBareUrls(content: string): string[] {
const urls: string[] = [];
const seen = new Set<string>();
const body = stripIgnoredMarkdownRegions(stripFrontmatter(content));
// Markdown link syntax is masked so anything glued to it — emphasis
// markers, curly quotes, possessives — never reaches the bare extractor;
// those URLs are already collected from link metadata.
const body = stripIgnoredMarkdownRegions(stripFrontmatter(content))
.replace(/!?\[[^\]\n]*\]\([ \t]*(?:<[^>\n]*>|[^)\s]*)[ \t]*\)/g, "");
const urlPattern = /https?:\/\/[^\s<>"']+/gi;

for (const match of body.matchAll(urlPattern)) {
const url = trimUrlBoundary(match[0]);
const url = trimUrlBoundary(cutAtNonUrlCharacter(match[0]));
if (!url || seen.has(url)) continue;
seen.add(url);
urls.push(url);
Expand All @@ -125,6 +129,18 @@ export function extractBareUrls(content: string): string[] {
return urls;
}

// RFC 3986 URL characters plus unicode letters, numbers, and combining marks
// (unencoded non-ASCII punctuation in a URL is always prose contamination —
// conforming URLs must percent-encode it).
const urlCharacter = /[A-Za-z0-9\-._~!$&'()*+,;=:/?#%[\]@\p{L}\p{N}\p{M}]/u;

function cutAtNonUrlCharacter(url: string): string {
for (let index = 0; index < url.length; index += 1) {
if (!urlCharacter.test(url[index])) return url.slice(0, index);
}
return url;
}

function stripFrontmatter(content: string): string {
const match = /^---\r?\n[\s\S]*?\r?\n---(?:\r?\n|$)/.exec(content);
return match ? content.slice(match[0].length) : content;
Expand Down
27 changes: 23 additions & 4 deletions src/tests/external-links.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -67,16 +67,35 @@ describe("externalLinksScanner", () => {
});

it.each([
["*[Empis tessellata](https://en.wikipedia.org/wiki/Empis_tessellata)*", "https://en.wikipedia.org/wiki/Empis_tessellata"],
["**[note](https://example.com/a)**", "https://example.com/a"],
["_https://example.com/a_", "https://example.com/a"],
["~~https://example.com/a~~", "https://example.com/a"],
["[t](https://example.com/a_(b))*", "https://example.com/a_(b)"],
["https://example.com/a_", "https://example.com/a"],
])("trims markdown emphasis markers after a URL: %s", (body, expected) => {
])("trims markdown emphasis markers after a bare URL: %s", (body, expected) => {
expect(extractBareUrls(body)).toEqual([expected]);
});

it.each([
// Markdown link syntax is masked before bare extraction: those URLs are
// collected from link metadata, so anything glued to the syntax —
// emphasis, curly quotes, possessives — never reaches the extractor.
["*[Empis tessellata](https://en.wikipedia.org/wiki/Empis_tessellata)*", []],
["**[note](https://example.com/a)**", []],
["[t](https://example.com/a_(b))*", []],
["“[note](https://example.com/a)”", []],
["*[t](https://x.com/a)*’s habitat", []],
// Nested brackets break the mask; the URL falls through to bare
// extraction cleanly and dedupes against the metadata link.
["[a [b] c](https://example.com/a) leftover", ["https://example.com/a"]],
// Bare URLs cut at the first character that cannot appear in a URL;
// unicode letters/numbers (IDN paths) survive.
["https://x.com/a’s page", ["https://x.com/a"]],
["https://x.com/a”", ["https://x.com/a"]],
["https://x.com/a」。", ["https://x.com/a"]],
["https://ja.wikipedia.org/wiki/日本語", ["https://ja.wikipedia.org/wiki/日本語"]],
])("handles unicode and markdown boundaries around URLs: %s", (body, expected) => {
expect(extractBareUrls(body)).toEqual(expected);
});

it("checks an italic-wrapped markdown link as a single clean URL", async () => {
const url = "https://en.wikipedia.org/wiki/Empis_tessellata";
const request = vi.fn(async (_url: string, method: "HEAD" | "GET") => ({ status: 200, method }));
Expand Down
Loading