From 496088a2fd63fadba25a1bdbaf6d685a1c7ab4b1 Mon Sep 17 00:00:00 2001 From: Jeremy Levartovsky <140487034+JayOfTheKeyboard@users.noreply.github.com> Date: Wed, 30 Sep 2026 16:18:35 +1000 Subject: [PATCH] fix(context): keep formatted and linked text in section titles MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit getHeadingText read only a heading's direct text and inlineCode children, so text inside emphasis, strong and links was dropped. A heading that was entirely a link or bold got an empty title and fell through to "Introduction". Read inline text recursively, and drop permalink anchors (a # link whose text is only ¶, §, # or a zero-width space) inside

, which would otherwise now end up in the title. --- .changeset/keep-heading-formatting.md | 5 +++++ packages/context/src/build.test.ts | 23 ++++++++++++++++++++++ packages/context/src/build.ts | 19 +++++++++--------- packages/context/src/html.test.ts | 28 +++++++++++++++++++++++++++ packages/context/src/html.ts | 17 ++++++++++++++++ 5 files changed, 82 insertions(+), 10 deletions(-) create mode 100644 .changeset/keep-heading-formatting.md diff --git a/.changeset/keep-heading-formatting.md b/.changeset/keep-heading-formatting.md new file mode 100644 index 0000000..b9cf108 --- /dev/null +++ b/.changeset/keep-heading-formatting.md @@ -0,0 +1,5 @@ +--- +"@neuledge/context": patch +--- + +Keep bold, italic, link and inline-code text in section titles. Only a heading's top-level text was read, so words inside formatting or links were dropped from the title, which has the highest search weight. In the Python docs, 300 of 6,475 sections lost words (`Numeric Types — int, float, complex` became `Numeric Types — , , `), including 152 FAQ and guide sections whose heading is a link and which were titled "Introduction". Markdown was hit too: `## Using [superjson](...)` became "Using ". Heading permalinks such as Sphinx's "¶" stay out of the title. diff --git a/packages/context/src/build.test.ts b/packages/context/src/build.test.ts index 3054820..8e7860b 100644 --- a/packages/context/src/build.test.ts +++ b/packages/context/src/build.test.ts @@ -76,6 +76,29 @@ Use brackets for dynamic segments. expect(result.sections[2].sectionTitle).toBe("Dynamic Routes"); }); + it("keeps formatted and linked text in section titles", () => { + const source = `## Using [superjson](https://github.com/blitz-js/superjson) + +Serializes dates and maps. + +## **Should I use \`generate\` or \`push\`?** + +They are two different commands. + +## The *strict* option + +Turns on strict checks. +`; + + const result = parseMarkdown(source, "docs/faq.md"); + + expect(result.sections.map((s) => s.sectionTitle)).toEqual([ + "Using superjson", + "Should I use generate or push?", + "The strict option", + ]); + }); + it("uses docTitle from frontmatter", () => { const source = `--- title: My Guide diff --git a/packages/context/src/build.ts b/packages/context/src/build.ts index 8d18dc0..eccc15e 100644 --- a/packages/context/src/build.ts +++ b/packages/context/src/build.ts @@ -3,7 +3,7 @@ * Parses markdown/MDX, AsciiDoc, and reStructuredText files and chunks them by section. */ -import type { Content, Heading, Root, Yaml } from "mdast"; +import type { Content, Heading, PhrasingContent, Root, Yaml } from "mdast"; import remarkFrontmatter from "remark-frontmatter"; import remarkParse from "remark-parse"; import { unified } from "unified"; @@ -157,15 +157,14 @@ function extractFrontmatter(tree: Root): DocFrontmatter { /** Get heading text from AST node. */ function getHeadingText(node: Heading): string { - let text = ""; - for (const child of node.children) { - if (child.type === "text") { - text += child.value; - } else if (child.type === "inlineCode") { - text += child.value; - } - } - return text; + return node.children.map(getInlineText).join(""); +} + +/** Text of an inline node, including text nested in emphasis, strong and links. */ +function getInlineText(node: PhrasingContent): string { + if (node.type === "text" || node.type === "inlineCode") return node.value; + if ("children" in node) return node.children.map(getInlineText).join(""); + return ""; } /** Convert AST nodes back to markdown text (simplified). */ diff --git a/packages/context/src/html.test.ts b/packages/context/src/html.test.ts index 961ce73..3455c39 100644 --- a/packages/context/src/html.test.ts +++ b/packages/context/src/html.test.ts @@ -35,3 +35,31 @@ describe("DocBook HTML examples", () => { expect(parsed.sections[0]?.content).toContain("```sh\necho hello\n```"); }); }); + +describe("HTML headings", () => { + it("keeps linked heading text and drops permalink anchors", () => { + const parsed = parseHtml( + `

Design FAQ

+

Why are Python strings immutable?

+

There are several advantages.

+

Performance

+

Strings of fixed size can be stored efficiently.

+

Constants added by the site module

+

The site module adds several constants.

+

Traits§

+

Shared behaviour for types.

+

Setup​

+

Install the package first.

`, + "faq/design.html", + ); + expect(parsed.sections.map((s) => s.sectionTitle)).toEqual([ + "Why are Python strings immutable?", + "Constants added by the site module", + "Traits", + "Setup", + ]); + // Only section titles change: anchors in other headings stay in the content, which + // keeps its link ratio (and so the table-of-contents filter's verdict) as before. + expect(parsed.sections[0]?.content).toContain("[¶](#performance"); + }); +}); diff --git a/packages/context/src/html.ts b/packages/context/src/html.ts index cabfcf5..3e20ef5 100644 --- a/packages/context/src/html.ts +++ b/packages/context/src/html.ts @@ -38,6 +38,23 @@ for (const tag of REMOVED_TAGS) { turndown.remove(tag); } +// Permalink anchors in section headings: "¶" (Sphinx, systemd), "§" (rustdoc), "#" +// (VuePress) or a zero-width space (Docusaurus). Section titles keep link text, so +// without this they end in the symbol. Only

, which becomes the section title: +// anchors in other headings stay in the content as before. A rule, not remove(): +// the link rule would match first. +const PERMALINK_TEXT = /^[¶§#🔗]?$/u; +turndown.addRule("sectionPermalink", { + filter: (node) => + node.nodeName === "A" && + (node.getAttribute("href") ?? "").startsWith("#") && + PERMALINK_TEXT.test( + (node.textContent ?? "").replace(/\u200b/g, "").trim(), + ) && + node.closest("h2") !== null, + replacement: () => "", +}); + // DocBook emits bare
 elements; Turndown's code rule requires 
.
 // Preserve their whitespace and prevent Markdown escaping of unit-file examples.
 turndown.addRule("barePre", {