From 989bdcb986668afa5c805ed70bb9937d0cb3e7ea Mon Sep 17 00:00:00 2001 From: TsunamiNoAi Date: Wed, 22 Jul 2026 21:54:31 -0400 Subject: [PATCH 1/5] feat: relax footnote-definition label charset to control-ID set MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Footnote-definition labels previously accepted only `[a-zA-Z0-9]`, so a line like `[^SCF:GOV-01]: text` fell through to a plain paragraph. Labels now accept any byte except ASCII whitespace and `[` / `]` — the intersection of pulldown-cmark (Zola) and cmark-gfm (GitHub) — so control-ID-shaped labels (`IAC-21.5`, `SCF:GOV-01`) parse to the same definition everywhere. Behaviour change: a whitespace-free `[^word]: …` line now parses as a footnote definition rather than a paragraph; labels containing a space (e.g. `[^see note]: x`) still stay paragraphs, which keeps paragraph interruption conservative (tryFootnoteDef also drives isParaBreak). Updates the "enhanced parse and render" expected HTML, extends the footnote-label XSS test to cover the now-parsing definition path, and adds parser tests for the charset boundaries and the ref/def id round-trip. CommonMark/GFM conformance unchanged (652/652 + 24/24). --- src/markdown/combinators.zig | 33 ++++++++++++++++-- src/markdown/security_test.zig | 8 +++++ src/markdown/test.zig | 61 ++++++++++++++++++++++++++++++++++ src/root.zig | 8 ++++- 4 files changed, 106 insertions(+), 4 deletions(-) diff --git a/src/markdown/combinators.zig b/src/markdown/combinators.zig index 4428a61..9558ede 100644 --- a/src/markdown/combinators.zig +++ b/src/markdown/combinators.zig @@ -46,6 +46,33 @@ pub const letter = mecha.oneOf(.{ mecha.ascii.range('a', 'z'), mecha.ascii.range pub const alphanumeric = mecha.oneOf(.{ letter, digit }); pub const whitespace = mecha.oneOf(.{ space, tab }).many(.{ .collect = false, .min = 1 }); +/// A single byte permitted inside a footnote-definition label. +/// +/// Accepts any byte **except** ASCII whitespace (space, tab, CR, LF) and the +/// square brackets `[` / `]`. This charset is deliberately the *intersection* +/// of the two footnote dialects zigmark must interoperate with: +/// +/// * pulldown-cmark (the parser Zola uses) treats a footnote label like a +/// link label — effectively any run of non-bracket characters; while +/// * cmark-gfm (GitHub) additionally forbids internal whitespace. +/// +/// Taking the intersection means control-ID-shaped labels such as `IAC-21.5` +/// or `SCF:GOV-01` parse to the *same* definition in zigmark, on GitHub, and +/// in Zola. The bracket exclusion keeps the label unambiguous (the closing +/// `]` terminates it). The whitespace exclusion is what keeps ordinary prose +/// such as `[^see note]: x` a paragraph rather than a footnote definition — +/// this matters because `tryFootnoteDef` also drives paragraph interruption +/// (see `isParaBreak` in `parser.zig`), so a looser charset would silently +/// reclassify authored prose. +/// +/// The footnote *reference* scanner (`inline.zig`) is a permissive superset of +/// this charset; tightening it to match is noted as future work. +pub const footnote_label_char = mecha.ascii.not(mecha.oneOf(.{ + space, tab, + mecha.ascii.char('\n'), mecha.ascii.char('\r'), + lbracket, rbracket, +})); + pub const url_char = mecha.oneOf(.{ alphanumeric, mecha.ascii.char('.'), mecha.ascii.char('/'), mecha.ascii.char(':'), mecha.ascii.char('?'), mecha.ascii.char('='), @@ -126,9 +153,9 @@ pub const blockquote_line = mecha.combine(.{ }.f); pub const footnote_definition = mecha.combine(.{ - lbracket, caret, - mecha.many(mecha.oneOf(.{ letter, digit }), .{ .collect = false, .min = 1 }).asStr(), rbracket, - colon, space, + lbracket, caret, + footnote_label_char.many(.{ .collect = false, .min = 1 }).asStr(), rbracket, + colon, space, mecha.rest.asStr(), }).map(struct { fn f(r: anytype) FootnoteDefResult { diff --git a/src/markdown/security_test.zig b/src/markdown/security_test.zig index 5d12a42..ef80f4f 100644 --- a/src/markdown/security_test.zig +++ b/src/markdown/security_test.zig @@ -81,12 +81,20 @@ test "security: ordinary and relative URLs are untouched" { // ── XSS: footnote label injection ───────────────────────────────────────────── test "security: footnote label is escaped in attribute and text" { + // The label carries no internal whitespace, so under the relaxed label + // charset (0.11.0) the `[^…]: note` line now parses as a real footnote + // *definition* rather than a paragraph. Both the reference site and the + // definition div embed the label; assert neither leaks a raw tag. const src = "x[^a\">]\n\n[^a\">]: note\n"; const out = try renderHtml(tst.allocator, src, .{}); defer tst.allocator.free(out); // No unescaped attribute-breaking quote or raw tag from the label. try tst.expect(mem.indexOf(u8, out, "= 1); + for (doc.children.items) |b| try tst.expect(b != .footnote_definition); +} + +test "footnote def label: control-ID charset parses" { + // Hyphen, dot, and colon+underscore labels all parse to a definition, + // matching pulldown-cmark (Zola) and cmark-gfm (GitHub). + try expectDefLabel("[^IAC-01]: control body\n", "IAC-01"); + try expectDefLabel("[^IAC-21.5]: dotted body\n", "IAC-21.5"); + try expectDefLabel("[^SCF:GOV_01]: colon and underscore\n", "SCF:GOV_01"); +} + +test "footnote def label: whitespace or bracket keeps it prose" { + // A space in the label is the boundary that keeps the line a paragraph + // (and keeps paragraph interruption conservative). + try expectStaysParagraph("[^see note]: not a definition\n"); + // A `[` cannot appear in a label, so the closing `]` never matches. + try expectStaysParagraph("[^fo[o]: not a definition\n"); +} + +test "footnote def: hyphenated definition interrupts a paragraph" { + const alloc = tst.allocator; + var p = Parser.init(); + var doc = try p.parseMarkdown(alloc, "Some prose\n[^IAC-01]: control body\n"); + defer doc.deinit(alloc); + try tst.expectEqual(@as(usize, 2), doc.children.items.len); + try tst.expect(doc.children.items[0] == .paragraph); + try tst.expect(doc.children.items[1] == .footnote_definition); + try tst.expectEqualStrings("IAC-01", doc.children.items[1].footnote_definition.label); +} + +test "footnote def: HTML ref and def id round-trip on hyphenated label" { + const alloc = tst.allocator; + var p = Parser.init(); + var doc = try p.parseMarkdown(alloc, "See [^IAC-01].\n\n[^IAC-01]: control body\n"); + defer doc.deinit(alloc); + const out = try html.render(alloc, doc); + defer alloc.free(out); + // The reference anchor and the definition div share the same id. + try tst.expect(std.mem.indexOf(u8, out, "href=\"#fn:IAC-01\"") != null); + try tst.expect(std.mem.indexOf(u8, out, "id=\"fn:IAC-01\"") != null); +} + test "backslash escape" { const alloc = tst.allocator; var p = Parser.init(); diff --git a/src/root.zig b/src/root.zig index 5522d2d..0fdd503 100644 --- a/src/root.zig +++ b/src/root.zig @@ -29,6 +29,10 @@ pub const AST = @import("markdown/ast.zig"); pub const Frontmatter = @import("markdown/frontmatter.zig"); /// A queryable collection of parsed Markdown documents with frontmatter. pub const Library = @import("markdown/library.zig").Library; +/// Programmatic footnote synthesis: resolve undefined `[^label]` references to +/// definitions supplied by a callback (`resolve`), and enumerate the ones that +/// remain undefined (`dangling`). +pub const footnotes = @import("markdown/footnotes.zig"); /// Markdown parser that transforms raw text into an `AST.Document`. pub const Parser = @import("markdown/parser.zig"); const ai = @import("markdown/renderers/ai.zig"); @@ -520,7 +524,9 @@ test "enhanced parse and render" { "

a footnoteSCF:GOV-01\n" ++ "a footnote2

\n" ++ "

Heading 2

\n" ++ - "

SCF:GOV-01: Footnote 1

\n" ++ + // The `[^SCF:GOV-01]:` line now parses as a real footnote definition + // (the relaxed label charset accepts `:` and `-`), matching Zola/GitHub. + "
\n

SCF:GOV-01: Footnote 1

\n
\n" ++ "
\n

2: Footnote 2

\n
\n", h); } From ecac6cfc7347ea44d11bb83395670132e02d802c Mon Sep 17 00:00:00 2001 From: TsunamiNoAi Date: Wed, 22 Jul 2026 21:54:43 -0400 Subject: [PATCH 2/5] feat: add footnotes module for programmatic definition synthesis MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit New `zigmark.footnotes` module resolves undefined `[^label]` references to definitions supplied by a `Resolver` callback, synthesised into the AST as real `footnote_definition` blocks: - `resolve(alloc, &doc, resolver, .{})` walks references recursively (paragraphs, headings, blockquotes, list items, table cells, and footnote-definition bodies; emphasis/strong/strikethrough/link inlines), parses each resolver result, and appends definitions in first-reference order. Single-pass by design. - `dangling(alloc, &doc)` returns the deduplicated labels still undefined, in first-reference order, so consumers can hard-fail. Synthesis happens at the AST level, so every renderer benefits with zero renderer changes — the Typst back-end expands the now-defined references to native `#footnote[…]`. Parsed blocks are moved out of a temporary document (label duped; only the list shell freed) to avoid a double-free. Tests run under std.testing.allocator and cover synthesis, dedup, first-ref order, null/unresolved counts, multi-block content, refs in tables/blockquotes, the resolver-error path, hostile-label escaping, single-pass dangling detection, and HTML/Typst/Markdown end-to-end. --- src/markdown/ast.zig | 1 + src/markdown/footnotes.zig | 231 +++++++++++++++++++++++++++ src/markdown/footnotes_test.zig | 274 ++++++++++++++++++++++++++++++++ 3 files changed, 506 insertions(+) create mode 100644 src/markdown/footnotes.zig create mode 100644 src/markdown/footnotes_test.zig diff --git a/src/markdown/ast.zig b/src/markdown/ast.zig index 86d5893..2be285a 100644 --- a/src/markdown/ast.zig +++ b/src/markdown/ast.zig @@ -968,4 +968,5 @@ test { _ = @import("query_test.zig"); _ = @import("library_test.zig"); _ = @import("mutation_test.zig"); + _ = @import("footnotes_test.zig"); } diff --git a/src/markdown/footnotes.zig b/src/markdown/footnotes.zig new file mode 100644 index 0000000..2098218 --- /dev/null +++ b/src/markdown/footnotes.zig @@ -0,0 +1,231 @@ +//! Programmatic footnote synthesis. +//! +//! A document may *reference* footnotes (`[^label]`) that have no matching +//! `[^label]: …` definition — for example when the definition text lives in an +//! external data source (a control catalog, a glossary, a database). This +//! module lets a caller supply those definitions on demand through a +//! `Resolver` callback and have them synthesised into the AST as real +//! `footnote_definition` blocks. +//! +//! Because synthesis happens at the **AST level** (not in a renderer), every +//! back-end benefits with zero renderer changes: +//! +//! * the Typst renderer's existing footnote pre-pass turns the now-defined +//! references into native `#footnote[…]` (PDF/UA-1-friendly); +//! * the HTML renderer links references to the appended definition divs; and +//! * the Markdown renderer round-trips the synthesised `[^label]: …` lines. +//! +//! ## Usage +//! +//! ```zig +//! const report = try footnotes.resolve(allocator, &doc, resolver, .{}); +//! // report.synthesized — definitions added; report.unresolved — refs the +//! // resolver returned null for. Use footnotes.dangling() to list the latter. +//! ``` +//! +//! `resolve` is **single-pass**: footnote references that appear *inside* +//! resolver-returned content are not themselves resolved. Call `dangling` +//! afterwards to discover any such (or otherwise unknown) labels — consumers +//! typically hard-fail a build on a non-empty result. +const std = @import("std"); +const Allocator = std.mem.Allocator; + +const AST = @import("ast.zig"); +const Parser = @import("parser.zig"); + +/// Supplies Markdown source for footnote definitions on demand. +pub const Resolver = struct { + /// Opaque context threaded to `resolveFn` (e.g. a `*const Library`). + ctx: ?*anyopaque = null, + /// Return Markdown source for `label`'s definition body, or `null` when the + /// label is unknown. A non-null result must be allocated with the passed + /// `allocator`; zigmark frees it (the `MermaidRendererFn` ownership + /// contract). The returned Markdown may contain multiple blocks. + resolveFn: *const fn (ctx: ?*anyopaque, allocator: Allocator, label: []const u8) anyerror!?[]const u8, + + fn call(self: Resolver, allocator: Allocator, label: []const u8) anyerror!?[]const u8 { + return self.resolveFn(self.ctx, allocator, label); + } +}; + +/// Options controlling `resolve`. +pub const ResolveOptions = struct { + /// Parser used to turn resolver-returned Markdown into AST blocks. Defaults + /// to a plain parser; pass one with matching flags (e.g. `.math = true`) so + /// synthesised definitions parse the same way as the host document. + parser: Parser = .{}, +}; + +/// Outcome of a `resolve` call. +pub const ResolveReport = struct { + /// Number of footnote definitions synthesised and appended to the document. + synthesized: usize = 0, + /// Number of undefined references the resolver returned `null` for. + unresolved: usize = 0, +}; + +/// Synthesise definitions for every undefined footnote reference in `doc`. +/// +/// For each *distinct* referenced label (in first-reference order) that lacks a +/// top-level `footnote_definition`, the `resolver` is invoked. A non-null +/// result is parsed and appended to `doc` as a new `footnote_definition` block; +/// a `null` result increments `unresolved`. +/// +/// Definitions are only ever top-level in `doc.children`, so only that level is +/// scanned for existing definitions and that is where synthesised blocks land. +pub fn resolve( + allocator: Allocator, + doc: *AST.Document, + resolver: Resolver, + opts: ResolveOptions, +) !ResolveReport { + var report = ResolveReport{}; + + // Existing top-level definitions — never re-synthesise these. + var defined = std.StringHashMap(void).init(allocator); + defer defined.deinit(); + for (doc.children.items) |*block| { + if (block.* == .footnote_definition) + try defined.put(block.footnote_definition.label, {}); + } + + // Referenced labels in first-reference order, deduplicated. The label + // slices borrow from the documents' inline-source buffers, which are not + // moved by appending blocks below, so they stay valid across the mutation. + var seen = std.StringHashMap(void).init(allocator); + defer seen.deinit(); + var ordered = std.ArrayList([]const u8).empty; + defer ordered.deinit(allocator); + for (doc.children.items) |*block| try walkBlockRefs(block, &seen, &ordered, allocator); + + for (ordered.items) |label| { + if (defined.contains(label)) continue; + const md = try resolver.call(allocator, label) orelse { + report.unresolved += 1; + continue; + }; + defer allocator.free(md); + + var fn_def = try synthesizeDefinition(allocator, opts.parser, label, md); + doc.edit().appendBlock(allocator, .{ .footnote_definition = fn_def }) catch |err| { + fn_def.deinit(allocator); + return err; + }; + report.synthesized += 1; + } + + return report; +} + +/// Return the deduplicated labels of every footnote reference in `doc` that has +/// no matching top-level definition, in first-reference order. +/// +/// The returned slice and each label are owned by the caller: free every entry +/// and then the slice with the same allocator. +pub fn dangling(allocator: Allocator, doc: *const AST.Document) ![][]const u8 { + var defined = std.StringHashMap(void).init(allocator); + defer defined.deinit(); + for (doc.children.items) |*block| { + if (block.* == .footnote_definition) + try defined.put(block.footnote_definition.label, {}); + } + + var seen = std.StringHashMap(void).init(allocator); + defer seen.deinit(); + var ordered = std.ArrayList([]const u8).empty; + defer ordered.deinit(allocator); + for (doc.children.items) |*block| try walkBlockRefs(block, &seen, &ordered, allocator); + + var out = std.ArrayList([]const u8).empty; + errdefer { + for (out.items) |s| allocator.free(s); + out.deinit(allocator); + } + for (ordered.items) |label| { + if (defined.contains(label)) continue; + const owned = try allocator.dupe(u8, label); + errdefer allocator.free(owned); + try out.append(allocator, owned); + } + return out.toOwnedSlice(allocator); +} + +// ── Internal ────────────────────────────────────────────────────────────────── + +/// Parse `markdown` (a resolver's definition body) and build a +/// `FootnoteDefinition` for `label` that OWNS the parsed blocks. +/// +/// The parsed blocks are **moved** out of the temporary document; only the +/// document's list shell is freed here. Calling `tmp.deinit` would double-free +/// the moved blocks, so it is never called. The label is duped — definitions +/// own their labels (see `AST.FootnoteDefinition.deinit`). +fn synthesizeDefinition( + allocator: Allocator, + parser: Parser, + label: []const u8, + markdown: []const u8, +) !AST.FootnoteDefinition { + var tmp = try parser.parseMarkdown(allocator, markdown); + // While `tmp` still owns the blocks, any failure must free it in full. + var moved = false; + errdefer if (!moved) tmp.deinit(allocator); + + const owned_label = try allocator.dupe(u8, label); + var fn_def = AST.FootnoteDefinition.init(allocator, owned_label); + // Past here `fn_def` owns the label (and, once moved, the blocks); a failure + // frees it and only `tmp`'s list shell. + errdefer fn_def.deinit(allocator); + + // Reserve up-front so the moves below cannot fail (all-or-nothing transfer). + try fn_def.children.ensureTotalCapacity(allocator, tmp.children.items.len); + for (tmp.children.items) |block| fn_def.children.appendAssumeCapacity(block); + + // Ownership transferred; free only the now-logically-empty list shell. + moved = true; + tmp.children.deinit(allocator); + + return fn_def; +} + +fn walkBlockRefs( + block: *const AST.Block, + seen: *std.StringHashMap(void), + ordered: *std.ArrayList([]const u8), + allocator: Allocator, +) Allocator.Error!void { + switch (block.*) { + .paragraph => |*p| try walkInlineRefs(p.children.items, seen, ordered, allocator), + .heading => |*h| try walkInlineRefs(h.children.items, seen, ordered, allocator), + .blockquote => |*bq| for (bq.children.items) |*b| try walkBlockRefs(b, seen, ordered, allocator), + .list => |*l| for (l.items.items) |*item| { + for (item.children.items) |*b| try walkBlockRefs(b, seen, ordered, allocator); + }, + .footnote_definition => |*fd| for (fd.children.items) |*b| try walkBlockRefs(b, seen, ordered, allocator), + .table => |*t| { + for (t.header.cells.items) |*c| try walkInlineRefs(c.children.items, seen, ordered, allocator); + for (t.body.items) |*row| { + for (row.cells.items) |*c| try walkInlineRefs(c.children.items, seen, ordered, allocator); + } + }, + else => {}, + } +} + +fn walkInlineRefs( + inlines: []const AST.Inline, + seen: *std.StringHashMap(void), + ordered: *std.ArrayList([]const u8), + allocator: Allocator, +) Allocator.Error!void { + for (inlines) |*inl| switch (inl.*) { + .footnote_reference => |fr| { + const gop = try seen.getOrPut(fr.label); + if (!gop.found_existing) try ordered.append(allocator, fr.label); + }, + .emphasis => |*e| try walkInlineRefs(e.children.items, seen, ordered, allocator), + .strong => |*s| try walkInlineRefs(s.children.items, seen, ordered, allocator), + .strikethrough => |*s| try walkInlineRefs(s.children.items, seen, ordered, allocator), + .link => |*l| try walkInlineRefs(l.children.items, seen, ordered, allocator), + else => {}, + }; +} diff --git a/src/markdown/footnotes_test.zig b/src/markdown/footnotes_test.zig new file mode 100644 index 0000000..3f9a086 --- /dev/null +++ b/src/markdown/footnotes_test.zig @@ -0,0 +1,274 @@ +//! Tests for programmatic footnote synthesis (`footnotes.zig`). +//! +//! All tests run under `std.testing.allocator`, so any leak — including on the +//! resolver-error path — fails the test. +const std = @import("std"); +const tst = std.testing; +const mem = std.mem; +const Allocator = std.mem.Allocator; + +const AST = @import("ast.zig"); +const Parser = @import("parser.zig"); +const footnotes = @import("footnotes.zig"); +const html = @import("renderers/html.zig"); +const typst = @import("renderers/typst.zig"); +const md_renderer = @import("renderers/markdown.zig"); + +// ── Test resolvers ──────────────────────────────────────────────────────────── + +/// Records how many times a resolver was invoked. +const Recorder = struct { + calls: usize = 0, +}; + +/// Resolves a small fixed catalog; unknown labels return `null`. +fn resolveCatalog(ctx: ?*anyopaque, allocator: Allocator, label: []const u8) anyerror!?[]const u8 { + if (ctx) |p| { + const rec: *Recorder = @ptrCast(@alignCast(p)); + rec.calls += 1; + } + if (mem.eql(u8, label, "IAC-01")) + return try allocator.dupe(u8, "IAC-01 — Identification. Covered by Access Control Policy."); + if (mem.eql(u8, label, "GOV-01")) + return try allocator.dupe(u8, "Governance definition body."); + if (mem.eql(u8, label, "MULTI")) + return try allocator.dupe(u8, "First paragraph.\n\nSecond paragraph."); + return null; +} + +/// Resolves every label to the same benign body. +fn resolveAny(ctx: ?*anyopaque, allocator: Allocator, label: []const u8) anyerror!?[]const u8 { + _ = ctx; + _ = label; + return try allocator.dupe(u8, "definition body"); +} + +/// Always fails — exercises the error path. +fn resolveErr(ctx: ?*anyopaque, allocator: Allocator, label: []const u8) anyerror!?[]const u8 { + _ = ctx; + _ = allocator; + _ = label; + return error.ResolverFailed; +} + +fn parse(alloc: Allocator, src: []const u8) !AST.Document { + var p = Parser.init(); + return p.parseMarkdown(alloc, src); +} + +// ── resolve() ───────────────────────────────────────────────────────────────── + +test "resolve: synthesizes a definition for a missing reference" { + const alloc = tst.allocator; + var doc = try parse(alloc, "See [^IAC-01].\n"); + defer doc.deinit(alloc); + + const report = try footnotes.resolve(alloc, &doc, .{ .resolveFn = resolveCatalog }, .{}); + try tst.expectEqual(@as(usize, 1), report.synthesized); + try tst.expectEqual(@as(usize, 0), report.unresolved); + + // A footnote_definition for IAC-01 now exists at the top level. + var found = false; + for (doc.children.items) |b| { + if (b == .footnote_definition and mem.eql(u8, b.footnote_definition.label, "IAC-01")) found = true; + } + try tst.expect(found); +} + +test "resolve: existing definition is untouched and resolver is not called" { + const alloc = tst.allocator; + var doc = try parse(alloc, "See [^IAC-01].\n\n[^IAC-01]: already defined\n"); + defer doc.deinit(alloc); + + var rec = Recorder{}; + const report = try footnotes.resolve(alloc, &doc, .{ .ctx = &rec, .resolveFn = resolveCatalog }, .{}); + try tst.expectEqual(@as(usize, 0), report.synthesized); + try tst.expectEqual(@as(usize, 0), report.unresolved); + try tst.expectEqual(@as(usize, 0), rec.calls); +} + +test "resolve: repeated references are deduplicated (one synthesis)" { + const alloc = tst.allocator; + var doc = try parse(alloc, "[^IAC-01] and again [^IAC-01].\n"); + defer doc.deinit(alloc); + + var rec = Recorder{}; + const report = try footnotes.resolve(alloc, &doc, .{ .ctx = &rec, .resolveFn = resolveCatalog }, .{}); + try tst.expectEqual(@as(usize, 1), report.synthesized); + try tst.expectEqual(@as(usize, 1), rec.calls); +} + +test "resolve: definitions are appended in first-reference order" { + const alloc = tst.allocator; + var doc = try parse(alloc, "[^GOV-01] then [^IAC-01].\n"); + defer doc.deinit(alloc); + + const report = try footnotes.resolve(alloc, &doc, .{ .resolveFn = resolveCatalog }, .{}); + try tst.expectEqual(@as(usize, 2), report.synthesized); + + const n = doc.children.items.len; + try tst.expect(n >= 3); + // The paragraph is first; the two synthesised defs follow in ref order. + try tst.expect(doc.children.items[n - 2] == .footnote_definition); + try tst.expect(doc.children.items[n - 1] == .footnote_definition); + try tst.expectEqualStrings("GOV-01", doc.children.items[n - 2].footnote_definition.label); + try tst.expectEqualStrings("IAC-01", doc.children.items[n - 1].footnote_definition.label); +} + +test "resolve: null result counts as unresolved" { + const alloc = tst.allocator; + var doc = try parse(alloc, "[^UNKNOWN] and [^IAC-01].\n"); + defer doc.deinit(alloc); + + const report = try footnotes.resolve(alloc, &doc, .{ .resolveFn = resolveCatalog }, .{}); + try tst.expectEqual(@as(usize, 1), report.synthesized); + try tst.expectEqual(@as(usize, 1), report.unresolved); +} + +test "resolve: multi-block resolver content becomes multiple child blocks" { + const alloc = tst.allocator; + var doc = try parse(alloc, "[^MULTI]\n"); + defer doc.deinit(alloc); + + const report = try footnotes.resolve(alloc, &doc, .{ .resolveFn = resolveCatalog }, .{}); + try tst.expectEqual(@as(usize, 1), report.synthesized); + + const n = doc.children.items.len; + const def = doc.children.items[n - 1].footnote_definition; + try tst.expectEqual(@as(usize, 2), def.children.items.len); + try tst.expect(def.children.items[0] == .paragraph); + try tst.expect(def.children.items[1] == .paragraph); +} + +test "resolve: references inside blockquotes and table cells are found" { + const alloc = tst.allocator; + const src = + \\> A quote referencing [^GOV-01]. + \\ + \\| Column [^IAC-01] | + \\|---| + \\| cell | + ; + var doc = try parse(alloc, src); + defer doc.deinit(alloc); + + const report = try footnotes.resolve(alloc, &doc, .{ .resolveFn = resolveCatalog }, .{}); + try tst.expectEqual(@as(usize, 2), report.synthesized); +} + +test "resolve: no leak when the resolver errors" { + const alloc = tst.allocator; + var doc = try parse(alloc, "[^IAC-01]\n"); + defer doc.deinit(alloc); + + try tst.expectError( + error.ResolverFailed, + footnotes.resolve(alloc, &doc, .{ .resolveFn = resolveErr }, .{}), + ); +} + +// ── End-to-end rendering ──────────────────────────────────────────────────────── + +test "resolve then HTML: reference links to synthesised definition div" { + const alloc = tst.allocator; + var doc = try parse(alloc, "See [^IAC-01].\n"); + defer doc.deinit(alloc); + + _ = try footnotes.resolve(alloc, &doc, .{ .resolveFn = resolveCatalog }, .{}); + + const out = try html.render(alloc, doc); + defer alloc.free(out); + try tst.expect(mem.indexOf(u8, out, "href=\"#fn:IAC-01\"") != null); + try tst.expect(mem.indexOf(u8, out, "
") != null); + try tst.expect(mem.indexOf(u8, out, "Identification") != null); +} + +test "resolve then Typst: reference expands to a native footnote (no placeholder)" { + const alloc = tst.allocator; + var doc = try parse(alloc, "See [^IAC-01].\n"); + defer doc.deinit(alloc); + + _ = try footnotes.resolve(alloc, &doc, .{ .resolveFn = resolveCatalog }, .{}); + + const out = try typst.render(alloc, doc); + defer alloc.free(out); + // Native Typst footnote, expanded from the synthesised definition body + // (not the bare-label placeholder used for undefined references). + try tst.expect(mem.indexOf(u8, out, "#footnote[") != null); + try tst.expect(mem.indexOf(u8, out, "Identification") != null); +} + +test "resolve then Markdown: synthesised definition round-trips" { + const alloc = tst.allocator; + var doc = try parse(alloc, "See [^IAC-01].\n"); + defer doc.deinit(alloc); + + _ = try footnotes.resolve(alloc, &doc, .{ .resolveFn = resolveCatalog }, .{}); + + const out = try md_renderer.render(alloc, doc); + defer alloc.free(out); + try tst.expect(mem.indexOf(u8, out, "[^IAC-01]") != null); + try tst.expect(mem.indexOf(u8, out, "[^IAC-01]: ") != null); +} + +test "resolve then HTML: hostile label from synthesis path is escaped" { + const alloc = tst.allocator; + var doc = try parse(alloc, "Danger [^].\n"); + defer doc.deinit(alloc); + + // resolveAny returns a definition for the hostile label, so it is + // synthesised — exercising label escaping through the definition path. + const report = try footnotes.resolve(alloc, &doc, .{ .resolveFn = resolveAny }, .{}); + try tst.expectEqual(@as(usize, 1), report.synthesized); + + const out = try html.render(alloc, doc); + defer alloc.free(out); + try tst.expect(mem.indexOf(u8, out, " Date: Wed, 22 Jul 2026 21:55:04 -0400 Subject: [PATCH 3/5] feat: add Library.footnoteResolver MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `Library.footnoteResolver()` returns a `footnotes.Resolver` that sources definition bodies from the footnote definitions found across the library's documents (first match wins, entries then document order). The matching definition's child blocks are rendered back to Markdown via the Markdown renderer's `renderBlock`, now made `pub` for this purpose. Nothing domain-specific lives here: the intended pattern is to build a glossary document of `[^ID]: …` lines from external data, `add()` it to the library, and pass `lib.footnoteResolver()` to `footnotes.resolve`. Tests cover resolving from a glossary, unknown labels surfacing as unresolved + dangling, and first-entry-wins. --- src/markdown/library.zig | 50 ++++++++++++++++++++ src/markdown/library_test.zig | 72 +++++++++++++++++++++++++++++ src/markdown/renderers/markdown.zig | 5 +- 3 files changed, 126 insertions(+), 1 deletion(-) diff --git a/src/markdown/library.zig b/src/markdown/library.zig index c542d64..22793cf 100644 --- a/src/markdown/library.zig +++ b/src/markdown/library.zig @@ -73,6 +73,8 @@ const mem = std.mem; const AST = @import("ast.zig"); const Frontmatter = @import("frontmatter.zig"); const Parser = @import("parser.zig"); +const footnotes = @import("footnotes.zig"); +const markdown_renderer = @import("renderers/markdown.zig"); /// A queryable collection of parsed Markdown documents with frontmatter. pub const Library = struct { @@ -244,6 +246,54 @@ pub const Library = struct { mem.sort(Result, results, Ctx{ .field = field, .ascending = ascending }, Ctx.lt); } + // ── Footnote resolution ───────────────────────────────────────────────────── + + /// Return a `footnotes.Resolver` that sources definition bodies from the + /// footnote definitions found across this library's documents. + /// + /// The resolver scans every entry's top-level `footnote_definition` blocks + /// for one whose label matches; the **first match wins** (entries in + /// insertion order, definitions in document order). The matching + /// definition's child blocks are rendered back to Markdown and returned. + /// + /// Nothing domain-specific lives here: the intended pattern is to build a + /// glossary document of `[^ID]: …` lines from external data, `add()` it to + /// the library, and pass `lib.footnoteResolver()` to `footnotes.resolve`. + /// + /// The returned resolver borrows `self`; it must not outlive the `Library`. + pub fn footnoteResolver(self: *const Library) footnotes.Resolver { + return .{ + .ctx = @constCast(self), + .resolveFn = resolveFromLibrary, + }; + } + + fn resolveFromLibrary(ctx: ?*anyopaque, allocator: Allocator, label: []const u8) anyerror!?[]const u8 { + const self: *const Library = @ptrCast(@alignCast(ctx.?)); + for (self.entries.items) |*entry| { + for (entry.document.children.items) |*block| { + if (block.* == .footnote_definition and + mem.eql(u8, block.footnote_definition.label, label)) + { + return try renderDefinitionBody(allocator, &block.footnote_definition); + } + } + } + return null; + } + + /// Serialise a footnote definition's child blocks back to Markdown (the + /// definition *body*, without the `[^label]:` prefix), blank-line separated. + fn renderDefinitionBody(allocator: Allocator, def: *const AST.FootnoteDefinition) ![]const u8 { + var aw: std.Io.Writer.Allocating = .init(allocator); + defer aw.deinit(); + for (def.children.items, 0..) |child, idx| { + if (idx > 0) try aw.writer.writeByte('\n'); + try markdown_renderer.renderBlock(allocator, &aw.writer, child); + } + return aw.toOwnedSlice(); + } + // ── Internal ────────────────────────────────────────────────────────────── /// A single frontmatter filter derived from one `path` or `path=value` diff --git a/src/markdown/library_test.zig b/src/markdown/library_test.zig index 3b7cce7..c46c194 100644 --- a/src/markdown/library_test.zig +++ b/src/markdown/library_test.zig @@ -2,6 +2,9 @@ const std = @import("std"); const tst = std.testing; const Library = @import("library.zig").Library; +const Parser = @import("parser.zig"); +const footnotes = @import("footnotes.zig"); +const md_renderer = @import("renderers/markdown.zig"); // ── helpers ────────────────────────────────────────────────────────────────── @@ -630,3 +633,72 @@ test "library: addFromDir recurses into subdirectories" { try tst.expectEqual(@as(usize, 2), lib.entries.items.len); } + +// ── footnoteResolver ──────────────────────────────────────────────────────────── + +const glossary = + \\[^IAC-01]: Identification and Authentication control. + \\[^GOV-01]: Governance and oversight control. +; + +test "library: footnoteResolver resolves definitions from a glossary" { + const alloc = tst.allocator; + var lib = Library.init(alloc); + defer lib.deinit(); + try lib.add(glossary, "glossary.md"); + + var p = Parser.init(); + var doc = try p.parseMarkdown(alloc, "See [^IAC-01] and [^GOV-01].\n"); + defer doc.deinit(alloc); + + const report = try footnotes.resolve(alloc, &doc, lib.footnoteResolver(), .{}); + try tst.expectEqual(@as(usize, 2), report.synthesized); + try tst.expectEqual(@as(usize, 0), report.unresolved); + + const out = try md_renderer.render(alloc, doc); + defer alloc.free(out); + try tst.expect(std.mem.indexOf(u8, out, "Identification and Authentication control.") != null); + try tst.expect(std.mem.indexOf(u8, out, "Governance and oversight control.") != null); +} + +test "library: footnoteResolver reports unknown labels as unresolved and dangling" { + const alloc = tst.allocator; + var lib = Library.init(alloc); + defer lib.deinit(); + try lib.add(glossary, "glossary.md"); + + var p = Parser.init(); + var doc = try p.parseMarkdown(alloc, "Known [^IAC-01], unknown [^ZZZ-99].\n"); + defer doc.deinit(alloc); + + const report = try footnotes.resolve(alloc, &doc, lib.footnoteResolver(), .{}); + try tst.expectEqual(@as(usize, 1), report.synthesized); + try tst.expectEqual(@as(usize, 1), report.unresolved); + + const missing = try footnotes.dangling(alloc, &doc); + defer { + for (missing) |m| alloc.free(m); + alloc.free(missing); + } + try tst.expectEqual(@as(usize, 1), missing.len); + try tst.expectEqualStrings("ZZZ-99", missing[0]); +} + +test "library: footnoteResolver first matching entry wins" { + const alloc = tst.allocator; + var lib = Library.init(alloc); + defer lib.deinit(); + try lib.add("[^IAC-01]: First definition wins.\n", "first.md"); + try lib.add("[^IAC-01]: Second definition loses.\n", "second.md"); + + var p = Parser.init(); + var doc = try p.parseMarkdown(alloc, "See [^IAC-01].\n"); + defer doc.deinit(alloc); + + _ = try footnotes.resolve(alloc, &doc, lib.footnoteResolver(), .{}); + + const out = try md_renderer.render(alloc, doc); + defer alloc.free(out); + try tst.expect(std.mem.indexOf(u8, out, "First definition wins.") != null); + try tst.expect(std.mem.indexOf(u8, out, "Second definition loses.") == null); +} diff --git a/src/markdown/renderers/markdown.zig b/src/markdown/renderers/markdown.zig index 7439708..c7ab3f3 100644 --- a/src/markdown/renderers/markdown.zig +++ b/src/markdown/renderers/markdown.zig @@ -182,7 +182,10 @@ fn renderInlines(writer: anytype, inlines: []const AST.Inline) !void { /// Render a single block to `writer`. Each block ends with exactly one `\n`; /// the caller inserts the blank-line separator `\n` between blocks. -fn renderBlock(alloc: Allocator, writer: anytype, block: AST.Block) !void { +/// +/// Exposed so callers (e.g. `Library.footnoteResolver`) can serialise the +/// child blocks of a node back to Markdown without wrapping the whole document. +pub fn renderBlock(alloc: Allocator, writer: anytype, block: AST.Block) !void { switch (block) { .paragraph => |p| { try renderInlines(writer, p.children.items); From e636cb391212a8f74c90265441e27f4eae8c4615 Mon Sep 17 00:00:00 2001 From: TsunamiNoAi Date: Wed, 22 Jul 2026 21:55:15 -0400 Subject: [PATCH 4/5] chore: release 0.11.0 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bump version to 0.11.0, document the footnotes module and `Library.footnoteResolver` in the changelog and README (label rules + a short resolve example), and call out the label-charset behaviour change (whitespace-free `[^…]:`-shaped prose lines now parse as definitions). --- CHANGELOG.md | 47 ++++++++++++++++++++++++++++++++++++++++ README.md | 60 ++++++++++++++++++++++++++++++++++++++++++++++++++- build.zig.zon | 2 +- 3 files changed, 107 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index cde7e0e..4827c40 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,53 @@ same stability guarantee. ## [Unreleased] +## [0.11.0] — 2026-07-22 + +### Added + +- **Programmatic footnote synthesis** (`zigmark.footnotes`). A new module lets a + caller supply footnote definitions on demand through a `Resolver` callback: + `resolve(alloc, &doc, resolver, .{})` finds every `[^label]` reference that + has no matching definition (walking paragraphs, headings, blockquotes, list + items, table cells, and footnote-definition bodies, plus emphasis/strong/ + strikethrough/link inlines), parses the resolver's Markdown, and appends real + `footnote_definition` blocks in first-reference order. Because synthesis + happens at the AST level, every renderer benefits with **zero renderer + changes** — in particular the Typst back-end expands the now-defined + references to native `#footnote[…]`. `resolve` is single-pass; + `dangling(alloc, &doc)` returns the deduplicated labels that are still + undefined (in first-reference order) so consumers can hard-fail a build. The + resolver-returned slice is owned and freed by zigmark (the `MermaidRendererFn` + ownership contract). See #82. +- **`Library.footnoteResolver()`** — a `footnotes.Resolver` that sources + definition bodies from the footnote definitions found across a library's + documents (first match wins). The intended pattern is to build a glossary + document of `[^ID]: …` lines from external data, `add()` it to the library, + and pass `lib.footnoteResolver()` to `footnotes.resolve`. Nothing + domain-specific lands in zigmark. +- `renderers/markdown.zig`'s `renderBlock` is now `pub`, so callers can + serialise a node's child blocks back to Markdown without wrapping a whole + document (used by `footnoteResolver`). + +### Changed + +- **Footnote-definition labels now accept a much wider charset.** A label may + contain any byte except ASCII whitespace (space, tab, CR, LF) and the square + brackets `[` / `]`, rather than only `[a-zA-Z0-9]`. This is the intersection + of pulldown-cmark (Zola) and cmark-gfm (GitHub), so control-ID-shaped labels + such as `IAC-21.5` or `SCF:GOV-01` now parse to the same definition in + zigmark, on GitHub, and in Zola. Both parser call sites (footnote definition + parsing and paragraph interruption) route through the one combinator, and the + reference parser is unchanged (it already accepts a permissive superset). + - **Behaviour change:** a whitespace-free line shaped like `[^word]: …` — with + a label the old charset rejected (e.g. `[^SCF:GOV-01]: text`) — now parses + as a footnote **definition** where previous releases treated it as an + ordinary paragraph containing a footnote reference. Lines whose label + contains a space (e.g. `[^see note]: x`) still stay paragraphs. The 0.8.0 + HTML-escaping guarantee for footnote labels extends to (and is tested on) + the definition and synthesis paths. CommonMark/GFM spec conformance is + unchanged (652/652 + 24/24). + ## [0.10.0] — 2026-07-18 ### Changed diff --git a/README.md b/README.md index 10e855e..eef6f27 100644 --- a/README.md +++ b/README.md @@ -720,13 +720,71 @@ Run the GFM suite with `zig build gfm`. ### Extensions - **Frontmatter** — YAML (`---`), TOML (`+++`), JSON (`{`), and ZON (`.{`) extraction, all normalised to `std.json.Value` -- **Footnotes** — `[^label]` references and definitions +- **Footnotes** — `[^label]` references and definitions, plus programmatic synthesis (see [Footnotes](#footnotes)) - **GFM Tables** — pipe-delimited tables with optional column alignment - **GFM Task lists** — `- [x]` / `- [ ]` items rendered as disabled checkboxes - **GFM Strikethrough** — `~~text~~` rendered as `text` - **GFM Extended autolinks** — bare `www.`, `http(s)://`, `ftp://`, and email autolinks - **GFM Disallowed raw HTML** — dangerous tags escaped at render time +### Footnotes + +`[^label]` marks a reference; `[^label]: …` on its own line defines it. A label +may contain any byte **except** ASCII whitespace and the brackets `[` / `]` — +the intersection of pulldown-cmark (used by Zola) and cmark-gfm (GitHub) — so +control-ID-shaped labels such as `IAC-21.5` or `SCF:GOV-01` parse to the same +definition in zigmark, on GitHub, and in Zola: + +```markdown +Access is authenticated per policy.[^IAC-01] + +[^IAC-01]: Identification & Authentication — see the access-control policy. +``` + +> **Behaviour note:** because the label charset now permits `:`, `.`, `-`, and +> other punctuation (rather than only `[a-zA-Z0-9]`), a whitespace-free line +> shaped like `[^word]: …` now parses as a footnote *definition* where a +> previous release treated it as a paragraph. + +**Programmatic synthesis.** References whose definition text lives in external +data (a control catalog, a glossary, a database) can be filled in at the AST +level via the `zigmark.footnotes` module. Supply a `Resolver` callback that +returns Markdown for a given label; `resolve` parses it and appends real +`footnote_definition` blocks, so every renderer works unchanged — including the +Typst back-end, which expands the now-defined references to native +`#footnote[…]`: + +```zig +const zigmark = @import("zigmark"); + +var doc = try parser.parseMarkdown(alloc, source); +defer doc.deinit(alloc); + +const resolver = zigmark.footnotes.Resolver{ + .resolveFn = struct { + fn f(_: ?*anyopaque, a: std.mem.Allocator, label: []const u8) anyerror!?[]const u8 { + if (std.mem.eql(u8, label, "IAC-01")) + return try a.dupe(u8, "Identification & Authentication control."); + return null; // unknown label → left dangling + } + }.f, +}; + +const report = try zigmark.footnotes.resolve(alloc, &doc, resolver, .{}); +// report.synthesized — definitions added; report.unresolved — nulls returned + +// Labels still without a definition (e.g. to hard-fail a build): +const missing = try zigmark.footnotes.dangling(alloc, &doc); +defer { + for (missing) |m| alloc.free(m); + alloc.free(missing); +} +``` + +`Library.footnoteResolver()` builds such a resolver from footnote definitions +found across a library's documents (first match wins) — for example a generated +glossary document of `[^ID]: …` lines that you `add()` to the library. + ## Building \& Testing ```bash diff --git a/build.zig.zon b/build.zig.zon index 93e45cf..998e088 100644 --- a/build.zig.zon +++ b/build.zig.zon @@ -9,7 +9,7 @@ .name = .zigmark, // This is a [Semantic Version](https://semver.org/). // In a future version of Zig it will be used for package deduplication. - .version = "0.10.0", + .version = "0.11.0", // Together with name, this represents a globally unique package // identifier. This field is generated by the Zig toolchain when the // package is first created, and then *never changes*. This allows From de10a33e6105a0821d20deb2dd535ded9e549675 Mon Sep 17 00:00:00 2001 From: TsunamiNoAi Date: Thu, 23 Jul 2026 07:44:14 -0400 Subject: [PATCH 5/5] style: zig fmt combinators.zig --- src/markdown/combinators.zig | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/src/markdown/combinators.zig b/src/markdown/combinators.zig index 9558ede..5f18449 100644 --- a/src/markdown/combinators.zig +++ b/src/markdown/combinators.zig @@ -68,9 +68,9 @@ pub const whitespace = mecha.oneOf(.{ space, tab }).many(.{ .collect = false, .m /// The footnote *reference* scanner (`inline.zig`) is a permissive superset of /// this charset; tightening it to match is noted as future work. pub const footnote_label_char = mecha.ascii.not(mecha.oneOf(.{ - space, tab, + space, tab, mecha.ascii.char('\n'), mecha.ascii.char('\r'), - lbracket, rbracket, + lbracket, rbracket, })); pub const url_char = mecha.oneOf(.{ @@ -153,9 +153,9 @@ pub const blockquote_line = mecha.combine(.{ }.f); pub const footnote_definition = mecha.combine(.{ - lbracket, caret, + lbracket, caret, footnote_label_char.many(.{ .collect = false, .min = 1 }).asStr(), rbracket, - colon, space, + colon, space, mecha.rest.asStr(), }).map(struct { fn f(r: anytype) FootnoteDefResult {