diff --git a/.changeset/keep-heading-formatting.md b/.changeset/keep-heading-formatting.md new file mode 100644 index 0000000..b9cf108 --- /dev/null +++ b/.changeset/keep-heading-formatting.md @@ -0,0 +1,5 @@ +--- +"@neuledge/context": patch +--- + +Keep bold, italic, link and inline-code text in section titles. Only a heading's top-level text was read, so words inside formatting or links were dropped from the title, which has the highest search weight. In the Python docs, 300 of 6,475 sections lost words (`Numeric Types — int, float, complex` became `Numeric Types — , , `), including 152 FAQ and guide sections whose heading is a link and which were titled "Introduction". Markdown was hit too: `## Using [superjson](...)` became "Using ". Heading permalinks such as Sphinx's "¶" stay out of the title. diff --git a/packages/context/src/build.test.ts b/packages/context/src/build.test.ts index 3054820..8e7860b 100644 --- a/packages/context/src/build.test.ts +++ b/packages/context/src/build.test.ts @@ -76,6 +76,29 @@ Use brackets for dynamic segments. expect(result.sections[2].sectionTitle).toBe("Dynamic Routes"); }); + it("keeps formatted and linked text in section titles", () => { + const source = `## Using [superjson](https://github.com/blitz-js/superjson) + +Serializes dates and maps. + +## **Should I use \`generate\` or \`push\`?** + +They are two different commands. + +## The *strict* option + +Turns on strict checks. +`; + + const result = parseMarkdown(source, "docs/faq.md"); + + expect(result.sections.map((s) => s.sectionTitle)).toEqual([ + "Using superjson", + "Should I use generate or push?", + "The strict option", + ]); + }); + it("uses docTitle from frontmatter", () => { const source = `--- title: My Guide diff --git a/packages/context/src/build.ts b/packages/context/src/build.ts index 8d18dc0..eccc15e 100644 --- a/packages/context/src/build.ts +++ b/packages/context/src/build.ts @@ -3,7 +3,7 @@ * Parses markdown/MDX, AsciiDoc, and reStructuredText files and chunks them by section. */ -import type { Content, Heading, Root, Yaml } from "mdast"; +import type { Content, Heading, PhrasingContent, Root, Yaml } from "mdast"; import remarkFrontmatter from "remark-frontmatter"; import remarkParse from "remark-parse"; import { unified } from "unified"; @@ -157,15 +157,14 @@ function extractFrontmatter(tree: Root): DocFrontmatter { /** Get heading text from AST node. */ function getHeadingText(node: Heading): string { - let text = ""; - for (const child of node.children) { - if (child.type === "text") { - text += child.value; - } else if (child.type === "inlineCode") { - text += child.value; - } - } - return text; + return node.children.map(getInlineText).join(""); +} + +/** Text of an inline node, including text nested in emphasis, strong and links. */ +function getInlineText(node: PhrasingContent): string { + if (node.type === "text" || node.type === "inlineCode") return node.value; + if ("children" in node) return node.children.map(getInlineText).join(""); + return ""; } /** Convert AST nodes back to markdown text (simplified). */ diff --git a/packages/context/src/html.test.ts b/packages/context/src/html.test.ts index 961ce73..3455c39 100644 --- a/packages/context/src/html.test.ts +++ b/packages/context/src/html.test.ts @@ -35,3 +35,31 @@ describe("DocBook HTML examples", () => { expect(parsed.sections[0]?.content).toContain("```sh\necho hello\n```"); }); }); + +describe("HTML headings", () => { + it("keeps linked heading text and drops permalink anchors", () => { + const parsed = parseHtml( + `

Design FAQ

+

Why are Python strings immutable?

+

There are several advantages.

+

Performance

+

Strings of fixed size can be stored efficiently.

+

Constants added by the site module

+

The site module adds several constants.

+

Traits§

+

Shared behaviour for types.

+

Setup​

+

Install the package first.

`, + "faq/design.html", + ); + expect(parsed.sections.map((s) => s.sectionTitle)).toEqual([ + "Why are Python strings immutable?", + "Constants added by the site module", + "Traits", + "Setup", + ]); + // Only section titles change: anchors in other headings stay in the content, which + // keeps its link ratio (and so the table-of-contents filter's verdict) as before. + expect(parsed.sections[0]?.content).toContain("[¶](#performance"); + }); +}); diff --git a/packages/context/src/html.ts b/packages/context/src/html.ts index cabfcf5..3e20ef5 100644 --- a/packages/context/src/html.ts +++ b/packages/context/src/html.ts @@ -38,6 +38,23 @@ for (const tag of REMOVED_TAGS) { turndown.remove(tag); } +// Permalink anchors in section headings: "¶" (Sphinx, systemd), "§" (rustdoc), "#" +// (VuePress) or a zero-width space (Docusaurus). Section titles keep link text, so +// without this they end in the symbol. Only

, which becomes the section title: +// anchors in other headings stay in the content as before. A rule, not remove(): +// the link rule would match first. +const PERMALINK_TEXT = /^[¶§#🔗]?$/u; +turndown.addRule("sectionPermalink", { + filter: (node) => + node.nodeName === "A" && + (node.getAttribute("href") ?? "").startsWith("#") && + PERMALINK_TEXT.test( + (node.textContent ?? "").replace(/\u200b/g, "").trim(), + ) && + node.closest("h2") !== null, + replacement: () => "", +}); + // DocBook emits bare
 elements; Turndown's code rule requires 
.
 // Preserve their whitespace and prevent Markdown escaping of unit-file examples.
 turndown.addRule("barePre", {