diff --git a/netlify/edge-functions/markdown-negotiation.js b/netlify/edge-functions/markdown-negotiation.js index 203e21f7..1eb0f30f 100644 --- a/netlify/edge-functions/markdown-negotiation.js +++ b/netlify/edge-functions/markdown-negotiation.js @@ -83,6 +83,53 @@ function createBlockStore() { }; } +// Same idea as the block-store marker: hide `>` inside quoted attrs so `[^>]*` +// tag scans don't stop early. +const ATTR_GT = "\u0001"; + +function findTagClose(html, openIndex) { + let quote = null; + for (let i = openIndex + 1; i < html.length; i++) { + const c = html[i]; + if (quote) { + if (c === quote) { + quote = null; + } + continue; + } + if (c === '"' || c === "'") { + quote = c; + } else if (c === ">") { + return i; + } + } + return -1; +} + +function protectQuotedAngles(html) { + let result = ""; + let cursor = 0; + for (let i = 0; i < html.length; i++) { + if (html[i] !== "<") { + continue; + } + const close = findTagClose(html, i); + if (close === -1) { + break; + } + result += html.slice(cursor, i); + result += html.slice(i, close).replace(/>/g, ATTR_GT); + result += ">"; + cursor = close + 1; + i = close; + } + return result + html.slice(cursor); +} + +function restoreQuotedAngles(text) { + return text.includes(ATTR_GT) ? text.replace(/\u0001/g, ">") : text; +} + // Depth-tracked because the TOC containers nest divs, so the first closing tag // isn't the matching one. function dropElements(html, tagName, classPattern) { @@ -113,6 +160,105 @@ function dropElements(html, tagName, classPattern) { return result + html.slice(cursor); } +function findListClose(html, innerStart) { + const token = /<\/?(?:ul|ol)\b[^>]*>/gi; + token.lastIndex = innerStart; + let depth = 1; + let match; + while ((match = token.exec(html))) { + depth += match[0].startsWith("]*>/gi; + let listDepth = 0; + let liStart = null; + let match; + + while ((match = token.exec(inner))) { + const isClose = match[0].startsWith(" line.trim() !== ""); + if (lines.length === 0) { + return `\n${marker.trimEnd()}`; + } + const indent = " ".repeat(marker.length); + let out = `\n${marker}${lines[0].trimStart()}`; + for (let i = 1; i < lines.length; i++) { + out += `\n${indent}${lines[i].trimStart()}`; + } + return out; +} + +function renderList(inner, ordered) { + const items = extractDirectLis(inner); + if (items.length === 0) { + return ""; + } + return items + .map((item, index) => { + const marker = ordered ? `${index + 1}. ` : `- `; + return formatListItem(marker, convertLists(item)); + }) + .join(""); +} + +function convertLists(html) { + const open = /<(ul|ol)\b[^>]*>/gi; + let result = ""; + let cursor = 0; + let match; + + while ((match = open.exec(html))) { + result += html.slice(cursor, match.index); + const ordered = match[1].toLowerCase() === "ol"; + const closed = findListClose(html, match.index + match[0].length); + if (!closed) { + result += html.slice(match.index); + return result; + } + result += renderList(html.slice(match.index + match[0].length, closed.innerEnd), ordered); + cursor = closed.after; + open.lastIndex = closed.after; + } + + return result + html.slice(cursor); +} + // There are no newlines inside
: each line is a token-line ending in 
, // and the indentation sits inside the token spans. So
becomes the line // break, and the other tags go without a space in their place. @@ -168,9 +314,11 @@ function toMarkdownTable(tableHtml) { } function htmlToMarkdown(html, pageUrl) { - const rawTitle = html.match(/]*>([\s\S]*?)<\/title>/i)?.[1] ?? ""; - const rawDescription = - html.match(/]+name=["']description["'][^>]+content=["']([^"']*)["'][^>]*>/i)?.[1] ?? ""; + html = protectQuotedAngles(html); + const rawTitle = restoreQuotedAngles(html.match(/]*>([\s\S]*?)<\/title>/i)?.[1] ?? ""); + const rawDescription = restoreQuotedAngles( + html.match(/]+name=["']description["'][^>]+content=["']([^"']*)["'][^>]*>/i)?.[1] ?? "", + ); let content = html.match(/]*>([\s\S]*?)<\/main>/i)?.[1] ?? html; @@ -202,11 +350,6 @@ function htmlToMarkdown(html, pageUrl) { /]*>([\s\S]*?)<\/h\1>/gi, (match, level, text) => `\n\n${"#".repeat(Number(level))} ${text}\n\n`, ) - .replace(/]*>([\s\S]*?)<\/ol>/gi, (match, items) => { - let index = 0; - return items.replace(/]*>([\s\S]*?)<\/li>/gi, (_, item) => `\n${++index}. ${item}`); - }) - .replace(/]*>([\s\S]*?)<\/li>/gi, "\n- $1") .replace(/]*>([\s\S]*?)<\/p>/gi, "\n\n$1\n\n") .replace(//gi, "\n") .replace(/]*>/gi, (tag) => { @@ -214,8 +357,10 @@ function htmlToMarkdown(html, pageUrl) { const alt = tag.match(/\balt="([^"]*)"/i)?.[1] ?? ""; return src ? `![${alt}](${src})` : alt; }); + markdown = convertLists(markdown); - markdown = normalizeWhitespace(inlineToMarkdown(markdown).replace(/[ \t]{2,}/g, " ")); + // Keep leading indent so nested list markers aren't collapsed. + markdown = normalizeWhitespace(inlineToMarkdown(markdown).replace(/(\S)[ \t]{2,}/g, "$1 ")); const title = normalizeWhitespace(rawTitle) @@ -247,7 +392,7 @@ function htmlToMarkdown(html, pageUrl) { header.push("", `Source: ${pageUrl}`); // Restore last, so the stashed blocks aren't re-normalized. - return `${store.restore(`${header.join("\n")}\n\n${markdown}`)}\n`; + return restoreQuotedAngles(`${store.restore(`${header.join("\n")}\n\n${markdown}`)}\n`); } function estimateTokens(markdown) { diff --git a/test/markdown-negotiation.test.mjs b/test/markdown-negotiation.test.mjs index c82b8f0c..40db9e33 100644 --- a/test/markdown-negotiation.test.mjs +++ b/test/markdown-negotiation.test.mjs @@ -257,6 +257,41 @@ describe("block structure", () => { ); }); + // Regression for #688. + it("keeps nested lists nested", async () => { + assert.equal( + await body( + "
  1. Outer step one
    • sub A
    • sub B
  2. " + + "
  3. Outer step two
", + ), + "1. Outer step one\n - sub A\n - sub B\n2. Outer step two", + ); + }); + + it("keeps nested lists nested when items wrap content in

", async () => { + assert.equal( + await body( + "

  1. Install deps

    • Node 20
    • npm ci
  2. " + + "
  3. Build the site

", + ), + "1. Install deps\n - Node 20\n - npm ci\n2. Build the site", + ); + }); + + it("keeps nested ordered lists nested", async () => { + assert.equal( + await body("
  1. A
    1. A1
    2. A2
  2. B
"), + "1. A\n 1. A1\n 2. A2\n2. B", + ); + }); + + it("indents nested unordered lists under unordered parents", async () => { + assert.equal( + await body("
  • A
    • A1
  • B
"), + "- A\n - A1\n- B", + ); + }); + it("converts line breaks", async () => { assert.equal(await body("

one
two

"), "one\ntwo"); }); @@ -302,6 +337,23 @@ describe("inline formatting", () => { "[Read `values.yaml`](/a)", ); }); + + // Regression for #688. + it("does not leak when a quoted attribute value contains `>`", async () => { + assert.equal(await body('

Hello

'), "Hello"); + assert.equal( + await body('

link text

'), + "[link text](/x)", + ); + }); + + it("still finds
when a quoted attribute on it contains `>`", async () => { + assert.equal(await body('

inside

'), "inside"); + }); + + it("still strips tags when the `>` in an attribute is entity-encoded", async () => { + assert.equal(await body('

Hello

'), "Hello"); + }); }); describe("code blocks", () => {