From ce7834ecc4f6e5ad3d487c2b3c7bb8876324321f Mon Sep 17 00:00:00 2001 From: Brenda Bannan Date: Mon, 20 Jul 2026 13:00:27 -0400 Subject: [PATCH 1/2] fix(make-pdf): stop smartypants from corrupting autolinked reference URLs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A bare autolinked URL (X, the shape produced whenever markdown links a plain https:// URL directly) sits with zero whitespace before its own closing tag. smartypants' URL_RE used a greedy \S+, which swallowed the adjacent already-carved SMARTPANTS_PRESERVED_N placeholder (left by the earlier TAG_RE carve pass) into the URL match itself. That placeholder never got restored — it leaked through as literal "SMARTPANTS_PRESERVED_N" text glued onto the end of the link, and the link's real closing tag was lost, leaving the anchor open and bleeding link styling onto all following text until the next closed it. Found in DOI reference lists across the AI4LD ebook PDFs. Fix: exclude the NUL placeholder delimiter from the URL character class so the match can never cross into an adjacent carved zone. Also adds APA-style hanging indent for reference lists: a new wrapReferencesSection() detects an

References

heading and wraps its content in
, scoped print CSS applies the 0.5in hanging indent — this never existed before, reference entries rendered as plain paragraphs. 4 new regression tests (2 smartypants placeholder-leak cases, 2 references-wrapping cases); narrowed one pre-existing test's assertion that had banned text-indent globally instead of scoping it to ordinary body paragraphs. Co-Authored-By: Claude Sonnet 5 --- make-pdf/src/print-css.ts | 10 +++++++ make-pdf/src/render.ts | 39 ++++++++++++++++++++++++- make-pdf/src/smartypants.ts | 17 ++++++++++- make-pdf/test/render.test.ts | 56 ++++++++++++++++++++++++++++++++++-- 4 files changed, 117 insertions(+), 5 deletions(-) diff --git a/make-pdf/src/print-css.ts b/make-pdf/src/print-css.ts index bf6f862bdc..3cc525870f 100644 --- a/make-pdf/src/print-css.ts +++ b/make-pdf/src/print-css.ts @@ -329,6 +329,16 @@ function inlineRules(): string { `}`, `strong { font-weight: 700; }`, `em { font-style: italic; }`, + // APA-style hanging indent (0.5in) for reference-list entries. Scoped to + // .references so ordinary body paragraphs are untouched. wrapped by + // render.ts's wrapReferencesSection() around the content following an + //

References

heading. + `.references p {`, + ` margin: 0 0 11pt;`, + ` padding-left: 0.5in;`, + ` text-indent: -0.5in;`, + `}`, + `.references p:first-child { margin-top: 0; }`, ].join("\n"); } diff --git a/make-pdf/src/render.ts b/make-pdf/src/render.ts index 514fbbc89e..db436df403 100644 --- a/make-pdf/src/render.ts +++ b/make-pdf/src/render.ts @@ -128,7 +128,7 @@ export function render(opts: RenderOptions): RenderResult { // that already carry an id keep it — the ids array records the ACTUAL id // per heading so TOC entries always link to something real. const anchored = opts.toc ? addHeadingIds(typographicHtml) : { html: typographicHtml, ids: [] }; - const anchoredHtml = anchored.html; + const anchoredHtml = wrapReferencesSection(anchored.html); const tocBlock = opts.toc ? buildTocBlock(anchoredHtml, anchored.ids) @@ -368,6 +368,43 @@ function wrapChaptersByH1(html: string): string { return chunks.join("\n"); } +/** + * Wrap an APA-style "References" section in a `.references` container so + * print CSS can apply a hanging indent to each entry without touching + * ordinary body paragraphs. Detects an

whose text is exactly + * "References" (case-insensitive) and wraps everything from there to the + * next

(or end of document) in
...
. + * The heading itself stays outside the wrapper so its own styling is + * unaffected. No-op if no such heading exists. + */ +function wrapReferencesSection(html: string): string { + const h1Re = /]*>([\s\S]*?)<\/h1>/gi; + let m: RegExpExecArray | null; + let headingStart = -1; + let headingEnd = -1; + while ((m = h1Re.exec(html)) !== null) { + const text = decodeTextEntities(stripTags(m[1])).trim().toLowerCase(); + if (text === "references") { + headingStart = m.index; + headingEnd = m.index + m[0].length; + break; + } + } + if (headingStart === -1) return html; + + // Find the next H1 after the References heading (end of section), or EOF. + const nextH1Re = /]*>/gi; + nextH1Re.lastIndex = headingEnd; + const next = nextH1Re.exec(html); + const sectionEnd = next ? next.index : html.length; + + const before = html.slice(0, headingEnd); + const body = html.slice(headingEnd, sectionEnd); + const after = html.slice(sectionEnd); + + return `${before}
${body}
${after}`; +} + function extractFirstHeading(html: string): string | null { const m = html.match(/]*>([\s\S]*?)<\/h1>/i); return m ? decodeTextEntities(stripTags(m[1]).trim()) : null; diff --git a/make-pdf/src/smartypants.ts b/make-pdf/src/smartypants.ts index 2dfe097e09..103d937837 100644 --- a/make-pdf/src/smartypants.ts +++ b/make-pdf/src/smartypants.ts @@ -20,7 +20,22 @@ const CODE_ZONE_RE = /<(pre|code|script|style)\b[^>]*>[\s\S]*?<\/\1>/gi; const TAG_RE = /<[^>]+>/g; -const URL_RE = /\bhttps?:\/\/\S+/g; +// BUG FIX (2026-07-20): a bare autolinked URL like
X has +// the URL text sitting with zero whitespace before its own closing +// tag. TAG_RE carves that into a "\u0000SMARTPANTS_PRESERVED_N\u0000" +// placeholder BEFORE this pattern runs. The old pattern (\S+, non- +// whitespace-greedy) swallowed that adjacent placeholder into the URL +// match, since \u0000 and the placeholder text are all non-whitespace. +// The outer carve() then stored "https://...\u0000SMARTPANTS_PRESERVED_N\u0000" +// as ONE preserved zone. String.replace() with a global regex does not +// rescan replacement text for further matches, so that inner placeholder +// was never restored -- it leaked through as literal "SMARTPANTS_PRESERVED_N" +// text glued onto the end of the URL, corrupting the link (found in DOI +// reference lists across the ebook PDFs). Excluding \u0000 from the URL +// char class stops the match from ever crossing into an adjacent, +// already-carved zone. +const URL_RE = /\bhttps?:\/\/[^\s\u0000]+/g; + /** * Apply smartypants to an HTML string. Zones that should not be touched: diff --git a/make-pdf/test/render.test.ts b/make-pdf/test/render.test.ts index b54085ecb2..6d465d8fc2 100644 --- a/make-pdf/test/render.test.ts +++ b/make-pdf/test/render.test.ts @@ -61,6 +61,30 @@ describe("smartypants", () => { expect(out).toContain(`href="it's-a-test.html"`); }); + test("does NOT leak SMARTPANTS_PRESERVED placeholders into autolinked URLs", () => { + // Regression test: found 2026-07-20 in AI4LD ebook PDF reference + // lists. A bare autolinked URL (anchor text == href, zero whitespace + // before the closing tag) previously had its placeholder + // swallowed by the URL regex's greedy \S+, leaking raw + // "SMARTPANTS_PRESERVED_N" text into the rendered link. + const input = `

See https://doi.org/10.1016/j.caeai.2026.100637 for details.

`; + const out = smartypants(input); + expect(out).not.toContain("SMARTPANTS_PRESERVED"); + expect(out).toContain( + `https://doi.org/10.1016/j.caeai.2026.100637` + ); + }); + + test("does NOT leak placeholders when a linked URL is immediately followed by another tag", () => { + // Same bug, different adjacency: URL directly abutting a second tag + // (e.g. two consecutive auto-linked references with no separating + // whitespace) must not bleed a placeholder into either link. + const input = `

https://a.example/xhttps://b.example/y

`; + const out = smartypants(input); + expect(out).not.toContain("SMARTPANTS_PRESERVED"); + expect(out).toBe(input); + }); + test("does NOT convert -- in CLI flags", () => { // Prose like "try --verbose mode" should not turn -- into em dash const out = smartypants(`

Try --verbose mode.

`); @@ -157,6 +181,25 @@ describe("render (end-to-end)", () => { expect(result.html).toContain("\u2014"); }); + test("wraps a References section for APA hanging-indent styling", () => { + const result = render({ + markdown: `# My Ebook\n\nBody text.\n\n# References\n\nSmith, J. (2020). A paper.\n\nJones, K. (2021). Another paper.\n`, + }); + expect(result.html).toMatch( + /
[\s\S]*Smith, J\. \(2020\)[\s\S]*Jones, K\. \(2021\)[\s\S]*<\/div>/ + ); + // The heading itself stays outside the wrapper. + expect(result.html).not.toMatch(/
\s*

{ + const lower = render({ markdown: `# Doc\n\nBody.\n\n# references\n\nRef one.\n` }); + expect(lower.html).toContain('
'); + + const none = render({ markdown: `# Doc\n\nJust body text, no references section.\n` }); + expect(none.html).not.toContain('
'); + }); + test("derives title from first H1 when --title is not passed", () => { const result = render({ markdown: `# My Title\n\nBody.` }); expect(result.meta.title).toBe("My Title"); @@ -236,12 +279,19 @@ describe("render (end-to-end)", () => { expect(result.html).toContain("Safe"); }); - test("respects text-align: left — no justify in print CSS", () => { + test("respects text-align: left — no justify or first-line indent on body paragraphs", () => { const result = render({ markdown: `para1\n\npara2\n` }); - // The rule from the design-review fix: no p + p indent, text-align: left. + // The rule from the design-review fix: no p + p first-line indent, text-align: left. expect(result.printCss).toContain("text-align: left"); expect(result.printCss).not.toContain("text-align: justify"); - expect(result.printCss).not.toContain("text-indent"); + // Ordinary paragraphs (the bare `p { ... }` rule) must not carry a + // first-line text-indent. This does NOT ban text-indent everywhere — + // `.references p` intentionally uses a negative text-indent for the + // APA hanging-indent pattern (see the "wraps a References section" + // test above); scope the check to the base rule specifically. + const baseParagraphRule = result.printCss.match(/(? { From 9309984020e7e3d210e5a83da317ef9ee6bcf28b Mon Sep 17 00:00:00 2001 From: Brenda Bannan Date: Fri, 24 Jul 2026 11:54:05 -0400 Subject: [PATCH 2/2] =?UTF-8?q?feat(make-pdf):=20extractCustomCover=20?= =?UTF-8?q?=E2=80=94=20branded=20cover=20sections=20replace=20built-in=20c?= =?UTF-8?q?over?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A document starting with
...
gets that section extracted before title derivation and placed in the document-assembly slot the --cover template would occupy (mutually exclusive with --cover, and the TOC lands after it). Enables publisher-branded covers (background color, custom typography) that the built-in cover template can't produce. Written 2026-07-21 alongside ce7834e; committed 2026-07-24 with tests (74 pass). Co-Authored-By: Claude Fable 5 --- make-pdf/src/render.ts | 59 ++++++++++++++++++++++++++++++------ make-pdf/test/render.test.ts | 47 ++++++++++++++++++++++++++++ 2 files changed, 96 insertions(+), 10 deletions(-) diff --git a/make-pdf/src/render.ts b/make-pdf/src/render.ts index db436df403..9c35a972eb 100644 --- a/make-pdf/src/render.ts +++ b/make-pdf/src/render.ts @@ -87,8 +87,13 @@ export function render(opts: RenderOptions): RenderResult { // 4. Smartypants (code-safe) const typographicHtml = smartypants(decoded); + // 4.5. Custom cover extraction (see extractCustomCover doc comment). Must + // run before title derivation so extractFirstHeading scans the REST of + // the document, not any decorative markup inside the custom cover. + const { cover: customCoverHtml, rest: bodyAfterCover } = extractCustomCover(typographicHtml); + // 4. Derive metadata (title from first H1 if not provided) - const derivedTitle = opts.title ?? extractFirstHeading(typographicHtml) ?? "Document"; + const derivedTitle = opts.title ?? extractFirstHeading(bodyAfterCover) ?? "Document"; const derivedAuthor = opts.author ?? ""; const derivedDate = opts.date ?? formatToday(); @@ -113,21 +118,25 @@ export function render(opts: RenderOptions): RenderResult { const css = printCss(cssOptions); // 6. Assemble document - const coverBlock = opts.cover - ? buildCoverBlock({ - title: derivedTitle, - subtitle: opts.subtitle, - author: derivedAuthor, - date: derivedDate, - }) - : ""; + // A custom cover (extracted above) takes priority over the built-in + // template — they're mutually exclusive, never combined. + const coverBlock = customCoverHtml + ? customCoverHtml + : opts.cover + ? buildCoverBlock({ + title: derivedTitle, + subtitle: opts.subtitle, + author: derivedAuthor, + date: derivedDate, + }) + : ""; // TOC anchors must resolve: assign id="toc-N" to each H1-H3 in the same // order buildTocBlock scans them, or every TOC link is a dead href (masked // in PDFs by Chromium outline bookmarks, glaring in --to html). Headings // that already carry an id keep it — the ids array records the ACTUAL id // per heading so TOC entries always link to something real. - const anchored = opts.toc ? addHeadingIds(typographicHtml) : { html: typographicHtml, ids: [] }; + const anchored = opts.toc ? addHeadingIds(bodyAfterCover) : { html: bodyAfterCover, ids: [] }; const anchoredHtml = wrapReferencesSection(anchored.html); const tocBlock = opts.toc @@ -405,6 +414,36 @@ function wrapReferencesSection(html: string): string { return `${before}
${body}
${after}`; } +/** + * Extract a custom, fully-styled cover section from the very start of the + * document, so it can be slotted into the same position the built-in + * --cover template occupies (before the TOC), instead of being left as + * ordinary body content (which the assembly order in render() places + * AFTER the TOC — see the `coverBlock, tocBlock, chapterHtml` order below). + * + * Detects a top-level `
...
` + * as the first non-whitespace content in the document. This is a distinct + * class from the tool's own generated `.cover` (buildCoverBlock) so the two + * mechanisms never collide — a document uses one or the other, never both. + * Callers should NOT also pass --cover when using a custom cover section. + * + * No nested-tag balancing: the custom cover section is expected to be a + * single flat block (divs/spans/hr, no inner
), so the first + *
after the opening tag is assumed to close it. + */ +function extractCustomCover(html: string): { cover: string; rest: string } { + const trimmed = html.replace(/^\s+/, ""); + const openMatch = trimmed.match(/^]*>/i); + if (!openMatch) return { cover: "", rest: html }; + const closeIdx = trimmed.indexOf("", openMatch[0].length); + if (closeIdx === -1) return { cover: "", rest: html }; + const closeEnd = closeIdx + "".length; + return { + cover: trimmed.slice(0, closeEnd), + rest: trimmed.slice(closeEnd), + }; +} + function extractFirstHeading(html: string): string | null { const m = html.match(/]*>([\s\S]*?)<\/h1>/i); return m ? decodeTextEntities(stripTags(m[1]).trim()) : null; diff --git a/make-pdf/test/render.test.ts b/make-pdf/test/render.test.ts index 6d465d8fc2..8758b251f1 100644 --- a/make-pdf/test/render.test.ts +++ b/make-pdf/test/render.test.ts @@ -200,6 +200,53 @@ describe("render (end-to-end)", () => { expect(none.html).not.toContain('
'); }); + test("extracts a custom .ai4ld-cover section and places it before the TOC", () => { + const result = render({ + markdown: `
\n
My Custom Cover
\n
\n\n# Real Title\n\nBody.\n`, + toc: true, + }); + const coverIdx = result.html.indexOf("My Custom Cover"); + const tocIdx = result.html.indexOf('
'); + // "Body." (not "Real Title") — the TOC itself legitimately contains a + // link with the text "Real Title", which would make that string a false + // marker for body content; "Body." appears only in the real chapter. + const bodyIdx = result.html.indexOf("Body."); + expect(coverIdx).toBeGreaterThan(-1); + expect(tocIdx).toBeGreaterThan(-1); + expect(bodyIdx).toBeGreaterThan(-1); + // Custom cover must render BEFORE the TOC, which must render before body + // content — the bug this fixes: body content (including a cover + // written as ordinary markdown) always landed AFTER the TOC because + // the assembly order is coverBlock, tocBlock, chapterHtml, and only + // the built-in --cover template populated coverBlock. + expect(coverIdx).toBeLessThan(tocIdx); + expect(tocIdx).toBeLessThan(bodyIdx); + }); + + test("custom cover title text does not create a duplicate TOC entry", () => { + // The cover title must NOT be an

(that's exactly what caused the + // duplicate-heading bug) — verify a same-titled
inside the + // custom cover isn't picked up as a second heading by the TOC. + const result = render({ + markdown: `
\n
Real Title
\n
\n\n# Real Title\n\nBody.\n`, + toc: true, + }); + const matches = result.html.match(/]*>Real Title<\/a>/g) ?? []; + expect(matches.length).toBe(1); + }); + + test("derives title from the real first H1, not text inside a custom cover", () => { + const result = render({ + markdown: `
\n
Not The Title
\n
\n\n# Actual Title\n\nBody.\n`, + }); + expect(result.meta.title).toBe("Actual Title"); + }); + + test("no custom cover present is a no-op (ordinary documents unaffected)", () => { + const result = render({ markdown: `# Just a doc\n\nNo cover here.\n` }); + expect(result.html).not.toContain("ai4ld-cover"); + }); + test("derives title from first H1 when --title is not passed", () => { const result = render({ markdown: `# My Title\n\nBody.` }); expect(result.meta.title).toBe("My Title");