commit 35bc42623b1eb0bcdfa0881ae9ec4c233e6fdab3 Author: Petr Balvín Date: Sun Sep 27 19:15:50 2026 +0200 feat: render Markdown, mathematics and Mermaid diagrams server-side diff --git a/.gitea/workflows/race.yml b/.gitea/workflows/race.yml new file mode 100644 index 0000000..be68f0c --- /dev/null +++ b/.gitea/workflows/race.yml @@ -0,0 +1,37 @@ +# Race, Go. Dispatched by hand, and never a gate on a push or a tag: the release tag +# is cut only after `just gates` has already raced the tree, so this workflow is the +# explicit second opinion, not a step of the release. +# +# The race detector roughly doubles both time and memory, which the shared runner box +# cannot afford on every push. Locally it belongs to `just gates`, which runs it once +# per task; here it is a decision rather than a routine. +# +# Every step is one command, so the step that fails is the gate that failed. +name: Race + +on: + workflow_dispatch: + +env: + # One core: parallelism buys no speed here and costs memory the box does not have. + GOFLAGS: -p=1 + GOMAXPROCS: "2" + +jobs: + race: + runs-on: fedora + timeout-minutes: 20 + steps: + - uses: actions/checkout@v7 + + - uses: actions/setup-go@v6 + with: + go-version-file: go.mod + cache: true + + - name: Install gcc + # The race detector needs cgo and the runner image carries no C compiler. + run: dnf install -y gcc + + - name: Race + run: go test -race -count=1 -timeout 10m ./... diff --git a/.gitea/workflows/test.yml b/.gitea/workflows/test.yml new file mode 100644 index 0000000..e21130f --- /dev/null +++ b/.gitea/workflows/test.yml @@ -0,0 +1,91 @@ +# Test, Go. Push and pull request to development. Never on main. +# +# The gates are the ones the justfile's `gates` recipe runs, minus race: the shared +# runner box cannot afford the race detector on every push, so it lives in race.yml. +# The box is one core and 2 GB beside Gitea, so parallelism is bounded on purpose and +# everything runs in one job. Extra jobs would duplicate the checkout, the Go setup and +# the dependency download three times without buying any parallelism. +# +# Every step is one command, so the step that fails is the gate that failed, and no +# shell option has to be trusted for the run to stop. The scripted steps are Perl, not +# shell and not Python: Perl behaves the same on both runner images, there is no +# bashism to trip over on ash, and it is one language instead of two. The Perl uses +# builtins only, because Fedora packages the Perl modules separately and nothing +# beyond `perl` itself may be assumed present. +name: Test + +on: + push: + branches: [development] + pull_request: + branches: [development] + +env: + # One core: parallelism buys no speed here and costs memory the box does not have. + GOFLAGS: -p=1 + GOMAXPROCS: "2" + +# A superseded run of the same ref is cancelled instead of queueing behind one that +# no longer matters. +concurrency: + group: ${{ gitea.workflow }}-${{ gitea.ref }} + cancel-in-progress: true + +jobs: + test: + runs-on: fedora + timeout-minutes: 10 + steps: + - uses: actions/checkout@v7 + + - uses: actions/setup-go@v6 + with: + # The module is the source of truth for the version, so it cannot drift. + go-version-file: go.mod + cache: true + + - name: Install Perl + # The runner images are minimal and Perl is not guaranteed. The install is a + # no-op where it is already present; drop this step once verified on the box. + run: dnf install -y perl + + # The steps follow the `gates` order of the justfile contract: build, format, + # vet, test. The vet gate is go vet and go fix -diff, two steps here. + - name: Build + run: go build ./... + + - name: Format + run: | + perl -e ' + open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!}; + my @bad = <$g>; + close($g); + print @bad; + exit(@bad ? 1 : 0); + ' + + - name: Vet + run: go vet ./... + + - name: Modernise + # Exits non-zero when it has something to rewrite, so it needs no output capture. + run: go fix -diff ./... + + - name: Tests + # The suite must be fast: a push pipeline that cannot finish in a few minutes + # moves its heavy part behind a dispatch. The inner timeout matches the job's, + # so a hanging test reports its own goroutine dump rather than a silent job + # kill. The pattern stays equal to `packages` in the project's justfile. + run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./... + + - name: Coverage floor + run: | + perl -e ' + open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!}; + my $total; + while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} } + close($c); + die qq{no total line in coverage.out\n} unless defined $total; + printf qq{Total coverage: %s%%\n}, $total; + exit($total < 80 ? 1 : 0); + ' diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..2c78186 --- /dev/null +++ b/.gitignore @@ -0,0 +1,6 @@ +.idea/ +.zcode/ + +# Go build and test output +/coverage.out +*.test diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..ec0f08e --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,28 @@ +# Changelog + +All notable changes to **Scriptorium** are documented in this file. + +The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and +this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). + +## [development] + +### Added + +- `Render` renders Markdown to HTML: the full CommonMark 0.31.2 grammar, the GitHub Flavored Markdown extensions (tables, strikethrough, task lists, extended autolinks), footnotes and definition lists. Output is deterministic for the same input. +- `RenderMath` and `RenderMathDisplay` render TeX mathematics to MathML Core: the symbol tables, fractions, radicals, scripts with movable limits, stretchy delimiters, the amsmath environments, accents, styles, colours, extensible arrows and bounded macros. Constructs outside the mappable surface stay visible as their verbatim source in an merror element. +- `RenderDiagram` renders Mermaid diagrams to SVG: the flowchart grammar including the historical `graph` spelling (node shapes, edge kinds with labels, subgraphs with their own direction, classes and styles) and the sequenceDiagram grammar (participants and actors, all arrow kinds, notes, activations, the block constructs, coloured rects, autonumber and dividers). Every other diagram type is refused with an error naming it. + + diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..00b6cdf --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,121 @@ +# Contributing + +Contributions to **scriptorium** are governed by the Contributor terms +below; submitting one means you accept them. + +## Contributor terms + +1. This project belongs to its owner alone. The owner decides what is + accepted, in what form and when; the decision is final and needs no + justification. +2. By submitting a contribution you assign to Petr Balvín + all present and future copyright and + related rights in it, worldwide, for the full term of the rights, + with the right to relicense and sublicense without restriction, + including under proprietary terms. +3. Where that assignment is not effective, it counts as a perpetual, + irrevocable, royalty-free licence with the same scope. +4. To the fullest extent permitted by law, you waive any right of + attribution and integrity in the contribution. The project names no + contributors and keeps no credits list. +5. By submitting you represent that the work is yours and that you + hold the rights to assign it as above. + +## Development setup + +Requirements: Go 1.27.1, and [just](https://github.com/casey/just) +for the recipes. The race detector in `just gates` needs a C compiler, so gcc must be +installed. + +```sh +git clone https://sourcedock.dev/petrbalvin/scriptorium.git +cd scriptorium +just build +just test +``` + +## Workflow + +1. Branch from `development`. Never commit directly to `main`, which is release-only. +2. Commit in [Conventional Commits](https://www.conventionalcommits.org/) form: + `type(scope): description`, subject line only, imperative mood, lowercase after the + colon, no trailing full stop. Allowed types: `feat`, `fix`, `docs`, `style`, + `refactor`, `perf`, `test`, `chore`, `ci`, `build`, `revert`. +3. One logical change per commit. A refactor, a behaviour change and a formatting pass + are three commits, never one. +4. Record every user-visible change in `CHANGELOG.md` under `## [development]`. +5. Add or update tests. Coverage stays at 80 percent or more; it is a hard gate. +6. Update the documentation when the public API, the configuration or the behaviour + changes. +7. Never commit while `just gates` is red; run it locally first. +8. Open a pull request against `development`. + +Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The release +workflow builds the artefacts and publishes the release and its notes. + +## Code style + +`gofmt` and `go vet` run through `just fmt` and `just vet`, with zero diff and zero +warnings tolerated. `just vet` also runs `go fix -diff`, so modernisations are part of +the gate and not a follow-up. `just gates` is the definition of done in one command, +and the recipe file names what it contains. Errors are checked explicitly, wrapped as +`fmt.Errorf("context: %w", err)`, and nothing panics outside `main`. + +The library depends on the Go standard library alone. Adding a dependency of any +kind, `golang.org/x/` included, is a decision for the owner, not a convenience for a +contributor. + +New source files open with the project's two-line licence header, whose SPDX +identifier matches `LICENSE`. Configuration files, workflows and dotfiles do not carry +it. + +## AI contribution policy + +AI tools are welcome as productivity aids and are a normal part of modern software +development. What matters is that the contribution stays understandable, reviewable and +genuinely useful. + +- **Disclose the assistance.** If AI helped draft any part of a commit, issue, pull + request or review, say so. +- **Commit messages carry exactly one trailer**, on the line after the subject: + + ``` + Assisted-by: MODEL + ``` + + Name the model that did the work, spelled the way its maker spells it, for example + `GLM 5.3`, `DeepSeek V4.1 Flash` or `Qwen 3.8 Flash`. No `Co-Authored-By`, no `Signed-off-by`, + no other trailers, and no prose: the trailer is the disclosure. +- **Issues and pull requests** attribute the assistance in a comment, for example + `_Assisted-by: GLM 5.3_`. It does not belong in the pull request description. +- **Take responsibility.** You are accountable for the accuracy, completeness, and + intent of everything you submit, whether or not AI produced it. +- **Review before marking ready.** Read the diff carefully, run it locally, and add the + tests it needs. Do not mark a pull request ready until you can defend every change in + it. +- **Quality over quantity.** Contributions that look like un-reviewed output, or whose + author cannot engage substantively during review, may be closed. +- **Preferred models.** Prefer open-weight models with transparent training data and + minimal output filtering. + +AI assists. It does not replace judgement. + +## Continuous integration + +Workflows live in `.gitea/workflows/` and run on the project's own runners: + +| Workflow | Trigger | What it does | +|---|---|---| +| Test | push or pull request to `development` | compile, format check, vet, modernisation, the test suite with the coverage floor | +| Race | dispatched by hand | the suite under the race detector, as a second opinion after the local gate | + +The local equivalent is `just gates`, which is the same set plus the race detector. + +## Reporting bugs + +Open an issue at `https://sourcedock.dev/petrbalvin/scriptorium/issues` with the +version, the operating system and architecture, the exact command, the full output, +and the expected against the actual behaviour. + +**Security issues do not go in the issue tracker.** Report them as +[SECURITY.md](SECURITY.md) describes. diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..f83dd2a --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md new file mode 100644 index 0000000..119bf18 --- /dev/null +++ b/README.md @@ -0,0 +1,174 @@ +# Scriptorium + +Scriptorium renders scientific documents server-side in Go: Markdown into +HTML, TeX mathematics into MathML Core, and Mermaid flowchart and sequence +diagrams into SVG. Three engines, each written from scratch in this +repository on the Go standard library alone, producing markup the reader's +browser already knows how to draw: no JavaScript runs on the page that reads +the result. + +## The strengths, each with its proof + +**Complete Markdown.** The engine renders the full CommonMark 0.31.2 +grammar plus the GitHub Flavored Markdown extensions. Measured against the +official suites: 649 of 652 CommonMark examples pass, and the three +exceptions are bare URLs and email addresses that the GFM extended +autolinks turn into links, the same trade GitHub's own reference +implementation makes; all 23 extension examples of the GFM specification +pass byte for byte. Footnotes and definition lists come on top, rendered +in the shapes GitHub and Pandoc established. + +**Complete mathematics.** Every one of the 391 mathematical symbols in the +KaTeX coverage table maps to MathML Core, verified by a systematic diff +against KaTeX's own source tables, together with fractions, radicals, +scripts with movable limits, stretchy delimiters, the amsmath environments, +accents, styles, colours, extensible arrows and bounded macros. Where a +KaTeX-compatible renderer needs JavaScript in the reader's browser, +scriptorium emits MathML Core that Chromium, Firefox and Safari draw +natively. + +**Deterministic.** The same input produces byte-identical output on every +call, in every process, on every machine. Diagram layouts break ties by the +order of appearance and never by map iteration; tests assert byte equality +across repeated renders in all three engines. Output is safe to cache, to +diff and to sign. + +**Zero dependencies.** The Go standard library alone, direct and +transitive. There is no `go.sum`, because there is nothing to sum: no +parser to supply-chain, no renderer to version-pin, no JavaScript bundle to +ship. Compiles wherever Go compiles. + +**Honest about its edges.** A construct outside a mappable surface is never +dropped and never guessed at: mathematics degrades in place, the unknown +command standing as its verbatim source inside a marked element while the +rest of the expression still renders, and a diagram type outside the two +grammars is refused with an error naming the type and the source line. + +**Ready for servers.** The engines hold no state between calls, so every +function is safe for concurrent use. Rendering never fails on malformed +input: Markdown and mathematics always produce output, and a diagram error +tells the author the line to fix. + +**No sanitisation, by contract.** The library renders trusted input and +writes the output verbatim; whether the result may reach a given audience +is the consumer's policy. A consumer of untrusted input keeps its own +sanitiser and applies it to the rendered result. + +## A taste + +```go +package main + +import ( + "os" + + "sourcedock.dev/petrbalvin/scriptorium" +) + +func main() { + os.Stdout.Write(scriptorium.Render([]byte( + "# Measured, not guessed\n\nA *claim* with [a source](/uri) and some `code`.\n"))) + os.Stdout.Write(scriptorium.RenderMathDisplay([]byte(`\sum_{i=1}^{n} \frac{1}{i^2}`))) + svg, err := scriptorium.RenderDiagram([]byte("flowchart LR\n A[Input] --> B{Model} --> C[Result]\n")) + if err != nil { + panic(err) + } + os.Stdout.Write(svg) +} +``` + +The first call prints exactly: + +```html +

Measured, not guessed

+

A claim with a source and some code.

+``` + +The second prints a complete MathML Core element, the summation limits +above and below the operator the way display mathematics sets them: + +```html +∑i=1n1i2 +``` + +The third returns a self-contained SVG document of the flowchart, every +element positioned by the library's own layout in whole pixels: ranks by +longest path, ordering by barycentre sweeps. + +## The engines + +### Markdown + +CommonMark 0.31.2 in full, with the GFM extensions: tables with per column +alignment, strikethrough, task lists and extended autolinks (`www.`, +`http(s)://`, `ftp://` and bare email addresses, with the punctuation, +parenthesis and entity trimming rules of the specification), plus footnotes +and definition lists. Tables render in the exact shape of GitHub's +reference implementation. + +### Mathematics + +The standard TeX command surface that maps to MathML Core, measured against +the KaTeX coverage list: the Greek letters including the variants, the +relation and operator tables including negations and the colon relations, +the arrows including the extensible `\x` family, the big operators, the +delimiters, and the letter-like symbols; the amsmath environments (`matrix` +and every delimited variant, `cases`, `aligned`, `align`, `gather`, +`equation`, `array`, `substack`); accents and over and under lines; the +font commands; spacing; colours; and bounded `\newcommand`, `\def` and +`\DeclareMathOperator` macros with expansion limits that turn a runaway +definition into an honest error rather than a hang. + +### Diagrams + +The two Mermaid grammars the forge's documents actually use, a coverage +measured across every repository on the forge. Flowchart (including the +historical `graph` alias): every node shape, every edge kind with labels, +branching, subgraphs with their own direction, classes and styles. +SequenceDiagram: participants and actors, all arrow kinds, notes, +activations, the `alt`/`opt`/`loop`/`par`/`critical`/`break` blocks, +coloured regions, `autonumber` and dividers. Everything else is refused +with a clear error naming the type. + +## The API + +| Function | Renders | Fails | +|---|---|---| +| `Render(source []byte) []byte` | Markdown to HTML | never | +| `RenderMath(source []byte) []byte` | TeX to inline MathML Core | never | +| `RenderMathDisplay(source []byte) []byte` | TeX to display MathML Core | never | +| `RenderDiagram(source []byte) ([]byte, error)` | Mermaid to SVG | on an unsupported type or a malformed statement | +| `SupportedDiagram(kind string) bool` | reports the Mermaid scope | | + +## Install + +```sh +go get sourcedock.dev/petrbalvin/scriptorium +``` + +Requires Go 1.27.1 or newer. + +## Documentation + +- [docs/API.md](docs/API.md): the surface in full, with the exact + boundary of the mathematics surface and every command that degrades +- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): how the engines are built +- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): the workflow and the recipes +- [CHANGELOG.md](CHANGELOG.md): the user-visible history +- [CONTRIBUTING.md](CONTRIBUTING.md): how to contribute +- [SECURITY.md](SECURITY.md): how to report a vulnerability + +## Development + +```sh +just build # compile every package +just test # the test suite with the coverage floor, under a memory fence +just fmt # format +just gates # the definition of done: build, format, vet, tests, race +``` + +## Licence + +MIT. See [LICENSE](LICENSE). + +Copyright © 2026 [Petr Balvín](https://petrbalvin.org) diff --git a/SECURITY.md b/SECURITY.md new file mode 100644 index 0000000..379652a --- /dev/null +++ b/SECURITY.md @@ -0,0 +1,35 @@ +# Security policy + +## Supported versions + +Security fixes go to the newest release and to the `development` branch. +Older releases do not receive them. + +## Reporting a vulnerability + +**Do not open a public issue for a security problem.** A public report tells everyone +about the flaw before there is a fix. Report it privately to +**opensource@petrbalvin.org**. + +Include: + +- the version or commit you tested, and the platform +- what the problem is, and what an attacker gains from it +- the smallest reproducer you have, ideally a test or a single command +- a suggested fix, if you have one + +## What to expect + +- A human reads the report, and you get an acknowledgement. +- You are kept informed while the fix is being made, and told when it ships. +- The fix is released before the details are published, and the timing is agreed with + you. + +## Out of scope + +- Findings that require the attacker to already run code as the user, or to have local + access. +- Missing hardening with no demonstrated impact. +- Flaws in a consumer of this library: report them to that project. Scriptorium + renders trusted input and sanitises nothing, so what a consumer does with the + output is the consumer's own security surface. diff --git a/docs/API.md b/docs/API.md new file mode 100644 index 0000000..3166b9a --- /dev/null +++ b/docs/API.md @@ -0,0 +1,124 @@ +# API + +Scriptorium is a library with one package and five exported functions. The +signatures below are the whole public surface; everything else lives under +`internal/` and reaches no caller. `go doc sourcedock.dev/petrbalvin/scriptorium` +is the authority on the signatures, and this file explains what each function +does, when it fails, and where the edges of the rendered surfaces are. + +## Library + +### `Render(source []byte) []byte` + +Renders Markdown to HTML. The grammar is CommonMark 0.31.2 in full, with the +GitHub Flavored Markdown extensions on top: tables, strikethrough, task lists +and extended autolinks, plus footnotes and definition lists. Rendering never +fails and never returns an error: any input parses, and a construct the +grammar does not know stays in the output as literal text. + +The extended autolinks supersede three CommonMark examples: a bare +`www.`/`http(s)://`/`ftp://` address or a bare email address becomes a link +where CommonMark without the extension would leave it as text. That is the +specification's own trade, and GitHub's reference implementation makes it +too. + +```go +html := scriptorium.Render([]byte("# Title\n\nText with [a link](/uri).\n")) +``` + +The Markdown engine embeds no mathematics and no diagrams: a `$…$` run stays +what the CommonMark grammar says it is. A consumer that wants mathematics +inside its documents calls `RenderMath` on the math spans it recognises and +splices the results into the HTML itself. + +### `RenderMath(source []byte) []byte` + +Renders TeX mathematics to a complete MathML Core `` element in the +inline form. Returns one line of markup with the namespace declared, ready to +embed in HTML. Never fails: see the degradation contract below. + +### `RenderMathDisplay(source []byte) []byte` + +The display form: the same engine, with `display="block"` on the element and +the movable limits of big operators set above and below instead of beside. + +### `RenderDiagram(source []byte) ([]byte, error)` + +Renders Mermaid source to a self-contained SVG document. The first line names +the diagram type; only two families are carried: + +- `flowchart` and its historical `graph` alias, with any direction + (`TB`/`TD`, `BT`, `LR`, `RL`), every node shape, every edge kind, subgraphs + with their own direction, and the `classDef`, `class`, `style` and + `linkStyle` statements; +- `sequenceDiagram`, with participants and actors, all arrow kinds, notes, + activations in both spellings, the `alt`, `opt`, `loop`, `par`, `critical` + and `break` blocks, coloured `rect` regions, `autonumber` and dividers. + +Returns an error when the type is outside the two families, when a direction +is unknown, or when a statement does not parse; every error names the diagram +type or the source line, so the author can find the sentence at fault. The +`click` statement is refused on purpose: the SVG is static and carries no +interactivity. + +### `SupportedDiagram(kind string) bool` + +Reports whether `RenderDiagram` carries the named Mermaid type. It answers +the question the error path would answer, before the call. + +## What every function guarantees + +- **Determinism.** The same input produces byte-identical output on every + call, in every process, on every machine. No layout decision reads a map + iteration order; the tests assert byte equality across repeated renders. +- **No sanitisation.** The input is trusted, and the output is written to + HTML, MathML or SVG verbatim. Whether the output may reach a given audience + is the consumer's policy: a consumer of untrusted input keeps its own + sanitiser and applies it to the rendered result. +- **Concurrency safety.** The engines hold no state between calls and share + nothing, so every function is safe for concurrent use. +- **No dependencies.** The standard library alone; there is no `go.sum` + because there is nothing to sum. + +## The mathematics surface + +The mathematics engine renders the standard TeX command surface that maps +into MathML Core, measured against the coverage list of KaTeX as the external +standard: the Greek letters including the variants, the relation and operator +tables including the negations and the colon relations, the arrows including +the extensible `\x` family, the big operators, the delimiters, the amsmath +environments, accents, the font and spacing commands, colours, and bounded +macros (`\newcommand`, `\renewcommand`, `\providecommand`, `\def`, +`\DeclareMathOperator`) with expansion limits. + +Multi-character relations have no single Unicode character, so they render as +the linear two-character operators KaTeX's own MathML branch produces +(`\coloneqq` as `≔`, `\coloneq` as `:−`, `\Coloneqq` as `∷=`). + +### Degradation, not failure + +A construct outside the mappable surface degrades in place: the offending +command stays in the output as its verbatim source inside an `` +element, and the rest of the expression still renders. The author sees what +was not understood, in the document, at the place it happened. + +The commands that degrade are the ones no MathML Core construct can carry: +positioning (`\raisebox`, `\vcenter`, `\smash`, the `\llap`/`\rlap`/ +`\mathclap` family, `\kern`, `\mkern`, `\hskip`), boxes and overlays +(`\boxed`, `\colorbox`, `\fcolorbox`, `\sideset`), text annotations of the +page rather than the formula (`\tag`, `\label`, `\notag`, `\verb`, +`\char`, `\mathstrut`), and the environments outside the table family +(`\begin{CD}`, `\begin{tikzpicture}` and every other unknown name). +A macro whose expansion would not terminate is cut by the expansion limits +and degrades at the call site. + +A degraded render is still a valid render: the output remains well-formed +MathML Core, and the tests assert well-formedness for every degraded corpus +case. + +## Notes + +The exported surface is documented in godoc form in the source, and `go doc ./...` is +the authority on signatures and types. This file explains what the surface is for and +how the parts fit together. It does not repeat the signatures, because a copy of a +signature is a future lie. diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md new file mode 100644 index 0000000..73804c7 --- /dev/null +++ b/docs/ARCHITECTURE.md @@ -0,0 +1,91 @@ +# Architecture + +How scriptorium is put together. Every node, package and arrow below exists in the +source tree; nothing is aspirational. + +## Overview + +```mermaid +flowchart TD + Caller[consumer] --> Root[scriptorium, public surface] + Root --> Markdown[internal/markdown] + Root --> Math[internal/mathml] + Root --> Diagram[internal/diagram] + Markdown --> Block[parse.go, block parser] + Markdown --> Inline[inline.go, inline parser] + Markdown --> Writer[render.go, HTML writer] + Math --> Tok[token.go, tokeniser] + Math --> MParse[parse.go and command.go, parser] + Math --> Tables[symbols.go, symbol tables] + Math --> Env[environments.go, tables] + Math --> Macros[macros.go, expansion] + Diagram --> DParse[flowparse.go and sequence.go, parsers] + Diagram --> DLayout[flowlayout.go and sequence.go, layout] + Diagram --> DWrite[svg.go, writer] +``` + +The root package is the whole public surface: `Render` hands Markdown source to +the Markdown engine, `RenderMath` and `RenderMathDisplay` hand TeX source to the +mathematics engine, `RenderDiagram` hands Mermaid source to the diagram engine, +and `SupportedDiagram` states the Mermaid scope. Everything else lives in +internal packages and reaches no caller. + +The Markdown engine parses line by line into the block tree CommonMark defines; +`RenderHTML` walks the tree, parses the inline content of every leaf against the +document's link reference definitions and footnotes, and writes HTML. + +The mathematics engine tokenises the TeX source, expands the bounded macros, +parses atoms with their scripts into a MathML tree and writes one line of XML. A +construct the tables do not carry degrades in place to its verbatim source inside +an merror element. + +The diagram engine parses statement lines into flowchart or sequence models, +lays the flowchart out by rank and barycentre and the sequence along its +lifelines, and writes SVG. A diagram type outside the two grammars is refused +with an error naming it. + +## Packages + +| Package | Responsibility | +|---|---| +| `scriptorium` (root) | the public surface: `Render`, `RenderMath`, `RenderMathDisplay`, `RenderDiagram`, `SupportedDiagram`. Owns nothing else. | +| `internal/markdown` | the Markdown engine: block parsing, inline parsing, HTML writing, and the scanners both phases share. Carries no state between calls. | +| `internal/mathml` | the mathematics engine: tokenising, macro expansion, symbol tables, environments, accents, styles and the XML writer. Carries no state between calls. | +| `internal/diagram` | the diagram engine: the flowchart and sequence parsers, both layouts and the SVG writer. Carries no state between calls. | + +## Data flow + +```mermaid +sequenceDiagram + participant Caller + participant S as scriptorium.RenderMath + participant T as mathml tokeniser + participant P as mathml parser + participant W as mathml writer + Caller->>S: TeX source + S->>T: source + T-->>P: tokens + P->>P: expand macros, parse atoms and scripts + P-->>W: MathML tree + W-->>Caller: one math element +``` + +Rendering cannot fail. A construct outside the grammar stays in the output as its +literal source rather than becoming an error, and nothing is sanitised on the way +out: that is the consumer's policy. + +Parsing and rendering are pure: no globals, no caches, no goroutines. The same source +produces byte-identical output on every call, which the tests assert. + +## State and lifetime + +- Everything is per-call: the block tree, the delimiter stack, the reference map, + the token list, the macro table, the diagram models and the layouts die with + the render. +- All functions are safe for concurrent use, because nothing is shared. + +## Dependencies + +None beyond the standard library. That is a contract of the project, not an +accident: every engine writes its own scanners and tables where the standard +library has nothing to offer. diff --git a/docs/DEVELOPMENT.md b/docs/DEVELOPMENT.md new file mode 100644 index 0000000..66b935b --- /dev/null +++ b/docs/DEVELOPMENT.md @@ -0,0 +1,76 @@ +# Development + +How to work on scriptorium. + +## Prerequisites + +- Go 1.27.1, the newest stable release. +- [just](https://github.com/casey/just) for the recipes. +- gcc, because the race detector in `just gates` needs cgo. + +## Setup + +```sh +git clone https://sourcedock.dev/petrbalvin/scriptorium.git +cd scriptorium +just build +``` + +## Recipes + +Every recipe in the project's file, and what it does. Taken from the file itself, so +the names and the list match it exactly. A library produces no binary, so the +recipes that carry one do not exist here. + +| Recipe | What it does | +|---|---| +| `just gates` | the definition of done: compile, format check, vet, modernisation, the test suite with the coverage floor, and the race detector | +| `just build` | compiles every package, zero warnings | +| `just test` | the full suite as CI runs it, with the coverage floor | +| `just unit . 'TestName'` | a fast scoped run for iterating | +| `just fuzz ` | a time-boxed fuzz of one target, never a gate | +| `just bench` | benchmarks, on an idle machine only | +| `just fmt` | gofmt, in place | +| `just fmt-check` | gofmt, zero diff | +| `just vet` | `go vet` and `go fix -diff` | +| `just clean` | removes the coverage profile | + +## Running a single test + +```sh +go test -run TestName ./... +``` + +Add `-v` for the sub-test names, and `-race` when the change touches concurrency. +`-count=1` defeats the test cache when a result looks stale. + +## Coverage + +```sh +just test +go tool cover -func=coverage.out +``` + +The `total:` line is the number that matters, and it stays at 80 percent or more. + +## Benchmarks + +```sh +go test -run='^$' -bench=. -benchmem -count=5 ./... +``` + +Benchmark on an idle machine, and compare only runs made in one process against each +other: runs in separate processes, or on a loaded machine, differ by more than the +effects being measured. + +## Continuous integration + +Workflows live in `.gitea/workflows/` and run on the project's own runners. They are +written by hand rather than through `just`, but they enforce the same set of gates, so +a green `just gates` locally is the fastest way to a green pipeline. + +## Releases + +Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The tag +drives the release workflow, which publishes the notes it extracted from +`CHANGELOG.md`. diff --git a/go.mod b/go.mod new file mode 100644 index 0000000..c608aba --- /dev/null +++ b/go.mod @@ -0,0 +1,3 @@ +module sourcedock.dev/petrbalvin/scriptorium + +go 1.27.1 diff --git a/internal/diagram/diagram.go b/internal/diagram/diagram.go new file mode 100644 index 0000000..7bedac8 --- /dev/null +++ b/internal/diagram/diagram.go @@ -0,0 +1,65 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +// Package diagram renders Mermaid diagrams to SVG with an engine of its +// own, built on the standard library alone. The flowchart grammar, +// including the historical "graph" spelling, and the sequenceDiagram +// grammar are covered; every other diagram type is refused with an error +// naming it. +// +// The output is deterministic: the same source always renders +// byte-identical SVG, because every layout decision falls back to the +// order of appearance and never to map iteration. +package diagram + +import ( + "fmt" + "strings" +) + +// Render renders the Mermaid diagram source to SVG. Statements are lines; +// blank lines and lines opening with %% are comments. A diagram type the +// engine does not carry is refused with an error naming the type, and a +// statement the grammar does not know is refused with an error naming the +// line. +func Render(source []byte) ([]byte, error) { + lines := statementLines(source) + if len(lines) == 0 { + return nil, fmt.Errorf("diagram: empty source") + } + header := strings.Fields(lines[0]) + switch strings.ToLower(header[0]) { + case "flowchart", "graph": + dir := "TB" + if len(header) > 1 { + dir = strings.ToUpper(header[1]) + if !flowDirections[dir] { + return nil, fmt.Errorf("diagram: unknown flow direction %q", header[1]) + } + } + return renderFlowchart(lines[1:], dir) + case "sequencediagram": + auto := len(header) > 1 && strings.ToLower(header[1]) == "autonumber" + return renderSequence(lines[1:], auto) + default: + return nil, fmt.Errorf("diagram: unsupported diagram type %q", header[0]) + } +} + +var flowDirections = map[string]bool{ + "TB": true, "TD": true, "BT": true, "LR": true, "RL": true, +} + +// statementLines splits the source into statements: one statement per +// line, comments and blank lines dropped. +func statementLines(source []byte) []string { + var lines []string + for raw := range strings.SplitSeq(string(source), "\n") { + line := strings.TrimSpace(raw) + if line == "" || strings.HasPrefix(line, "%%") { + continue + } + lines = append(lines, line) + } + return lines +} diff --git a/internal/diagram/diagram_test.go b/internal/diagram/diagram_test.go new file mode 100644 index 0000000..8007044 --- /dev/null +++ b/internal/diagram/diagram_test.go @@ -0,0 +1,163 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package diagram + +import ( + "bytes" + "encoding/xml" + "os" + "path/filepath" + "testing" +) + +// goldenCorpus is the project's own hand-written corpus. Every input has +// its rendered SVG in testdata, byte for byte. +var goldenCorpus = []struct { + name string + input string +}{ + {"fc-basic", "flowchart TD\n A[Start] --> B[End]\n"}, + {"fc-lr", "flowchart LR\n A --> B --> C\n"}, + {"fc-shapes", "flowchart TD\n A[Rect]\n B(Round)\n C{Diamond}\n D((Circle))\n E[[Subroutine]]\n F>Asymmetric]\n G([Stadium])\n H{{Hexagon}}\n"}, + {"fc-edge-kinds", "flowchart TD\n A --- B\n B --> C\n C -.- D\n D -.-> E\n E ==> F\n F ---> G\n"}, + {"fc-edge-labels", "flowchart TD\n A -- text --> B\n A -->|piped| C\n A -. dotted text .-> D\n A == thick text ==> E\n"}, + {"fc-branching", "flowchart TD\n A --> B & C\n D & E --> F\n B --> F\n"}, + {"fc-quoted-labels", `flowchart TD + A["Label with (brackets) and \"quotes\""] --> B['single'] +`}, + {"fc-subgraph", "flowchart TD\n A --> B\n subgraph Group One\n B --> C\n end\n C --> D\n"}, + {"fc-subgraph-direction", "flowchart TB\n A --> B\n subgraph Inner\n direction LR\n B --> C\n B --> D\n end\n"}, + {"fc-classes", "flowchart TD\n A[First]:::urgent --> B[Second]\n classDef urgent fill:#f96,stroke:#333,stroke-width:2px\n class A,B urgent\n style B fill:#9f9\n linkStyle 0 stroke:#f39\n"}, + {"fc-graph-alias", "graph LR\n A[Alpha] --> B[Beta]\n"}, + {"fc-cycle", "flowchart TD\n A --> B --> C --> A\n C --> D\n"}, + {"fc-single", "flowchart TD\n Lonely\n"}, + {"fc-multiline", "flowchart TD\n A[\"first line
second line\"] --> B\n"}, + {"fc-direction-line", "flowchart TD\n A --> B\n direction LR\n"}, + {"sq-basic", "sequenceDiagram\n Alice->>Bob: Hello\n Bob-->>Alice: Hi\n"}, + {"sq-arrows", "sequenceDiagram\n A->>B: solid arrow\n B-->>A: dashed arrow\n A-xB: solid cross\n B--xA: dashed cross\n A->B: solid open\n B-->A: dashed open\n"}, + {"sq-participants", "sequenceDiagram\n participant A as Alice\n actor B as Bob\n participant C\n A->>B: hi\n B->>C: forward\n"}, + {"sq-notes", "sequenceDiagram\n participant A\n participant B\n Note left of A: on the left\n Note right of B: on the right\n Note over A: alone\n Note over A,B: spanning\n"}, + {"sq-alt", "sequenceDiagram\n A->>B: check\n alt yes\n A->>B: proceed\n else no\n B-->>A: refuse\n end\n"}, + {"sq-loop-opt", "sequenceDiagram\n loop each round\n A->>B: tick\n end\n opt maybe\n B-->>A: tock\n end\n"}, + {"sq-par", "sequenceDiagram\n par left\n A->>B: one\n and right\n C->>D: two\n end\n"}, + {"sq-critical", "sequenceDiagram\n critical locked\n A->>B: work\n option unlocked\n B-->>A: skip\n end\n"}, + {"sq-break", "sequenceDiagram\n break failure\n A->>B: stop\n end\n"}, + {"sq-rect", "sequenceDiagram\n rect rgb(200, 255, 200)\n A->>B: inside\n end\n rect #e0e0ff\n B-->>A: also inside\n end\n"}, + {"sq-autonumber", "sequenceDiagram autonumber\n A->>B: first\n B-->>A: second\n A->>A: self\n"}, + {"sq-divider", "sequenceDiagram\n A->>B: before\n ... section break ...\n A->>B: after\n"}, + {"sq-activation", "sequenceDiagram\n activate A\n A->>+B: request\n B-->>-A: response\n deactivate A\n"}, + {"sq-self", "sequenceDiagram\n A->>A: think\n A-->A: rethink\n"}, + {"sq-nested", "sequenceDiagram\n alt outer\n loop inner\n A->>B: x\n end\n else other\n B->>A: y\n end\n"}, +} + +func TestGolden(t *testing.T) { + for _, tc := range goldenCorpus { + t.Run(tc.name, func(t *testing.T) { + got, err := Render([]byte(tc.input)) + if err != nil { + t.Fatalf("render: %v", err) + } + golden := filepath.Join("testdata", tc.name+".svg") + if os.Getenv("SCRIPTORIUM_GOLDEN_UPDATE") == "1" { + if err := os.WriteFile(golden, got, 0o644); err != nil { + t.Fatalf("write golden: %v", err) + } + return + } + want, err := os.ReadFile(golden) + if err != nil { + t.Fatalf("read golden: %v (run with SCRIPTORIUM_GOLDEN_UPDATE=1 once)", err) + } + if !bytes.Equal(got, want) { + t.Errorf("output changed; run with SCRIPTORIUM_GOLDEN_UPDATE=1 after review\ngot: %s\nwant: %s", got, want) + } + }) + } +} + +func TestWellFormed(t *testing.T) { + for _, tc := range goldenCorpus { + out, err := Render([]byte(tc.input)) + if err != nil { + t.Fatalf("%s: render: %v", tc.name, err) + } + dec := xml.NewDecoder(bytes.NewReader(out)) + for { + _, err := dec.Token() + if err != nil { + if err.Error() == "EOF" { + break + } + t.Errorf("%s: malformed XML: %v\n%s", tc.name, err, out) + break + } + } + } +} + +func TestDeterministic(t *testing.T) { + for _, tc := range goldenCorpus { + first, err := Render([]byte(tc.input)) + if err != nil { + t.Fatalf("%s: render: %v", tc.name, err) + } + for range 5 { + next, err := Render([]byte(tc.input)) + if err != nil || !bytes.Equal(first, next) { + t.Fatalf("%s: output differs between calls", tc.name) + } + } + } +} + +func TestRefusesOtherTypes(t *testing.T) { + for _, src := range []string{ + "pie\n \"a\": 50\n", + "gantt\n title x\n", + "classDiagram\n A <|-- B\n", + "stateDiagram-v2\n [*] --> A\n", + "erDiagram\n", + } { + _, err := Render([]byte(src)) + if err == nil { + t.Errorf("expected refusal for %q", src) + } + } +} + +func TestParseErrors(t *testing.T) { + cases := []struct { + src string + want string + }{ + {"flowchart TD\n A -->\n", "flowchart line 2"}, + {"flowchart TD\n subgraph S\n A --> B\n", "has no end"}, + {"flowchart TD\n end\n", "end without subgraph"}, + {"flowchart TD\n click A callback\n", "click is unsupported"}, + {"flowchart SIDEWAYS\n A --> B\n", "unknown flow direction"}, + {"flowchart TD\n A -- B\n", "unfinished edge token"}, + {"sequenceDiagram\n A->B no colon\n", "message needs a colon"}, + {"sequenceDiagram\n nonsense line\n", "cannot parse"}, + {"sequenceDiagram\n end\n", "end without a block"}, + {"sequenceDiagram\n alt x\n A->>B: y\n", "has no end"}, + {"sequenceDiagram\n rect blue\n A->>B: y\n end\n", "rect needs"}, + {"sequenceDiagram\n note behind A: x\n", "note needs left of"}, + } + for _, tc := range cases { + _, err := Render([]byte(tc.src)) + if err == nil { + t.Errorf("input %q: expected error", tc.src) + continue + } + if !bytes.Contains([]byte(err.Error()), []byte(tc.want)) { + t.Errorf("input %q: error %q does not mention %q", tc.src, err, tc.want) + } + } +} + +func TestEmptySource(t *testing.T) { + if _, err := Render([]byte("%% only a comment\n\n")); err == nil { + t.Error("expected an error for an empty diagram") + } +} diff --git a/internal/diagram/flowlayout.go b/internal/diagram/flowlayout.go new file mode 100644 index 0000000..7c5b3f6 --- /dev/null +++ b/internal/diagram/flowlayout.go @@ -0,0 +1,494 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package diagram + +import ( + "fmt" + "slices" + "sort" + "strings" +) + +// Layout constants, in SVG units. +const ( + margin = 32 + rankGap = 64 + laneGap = 48 + nodeMinWidth = 96 + textPad = 28 +) + +type laidNode struct { + x, y, w, h int // centre and size in the TB orientation +} + +type flowLayout struct { + laid []laidNode // parallel to d.nodes + rank []int // parallel to d.nodes + width int + height int +} + +func renderFlowchart(lines []string, dir string) ([]byte, error) { + d, err := parseFlowchart(lines, dir) + if err != nil { + return nil, err + } + lay := layoutFlowchart(d) + return writeFlowchart(d, lay), nil +} + +// layoutFlowchart assigns ranks by longest path, orders each rank by +// barycentre sweeps with the order of appearance as the tie breaker, and +// packs the rows. Every step walks nodes and edges by index, so the +// layout never depends on map iteration. +func layoutFlowchart(d *flowDiagram) *flowLayout { + n := len(d.nodes) + lay := &flowLayout{laid: make([]laidNode, n), rank: make([]int, n)} + + for i, node := range d.nodes { + lay.laid[i].w, lay.laid[i].h = nodeSize(node) + } + + // Ranks: relax along the edges, in index order, one pass per node at + // most. A cycle cannot push the ranks past that bound. + for pass := 0; pass <= len(d.nodes); pass++ { + changed := false + for _, e := range d.edges { + if lay.rank[e.to] < lay.rank[e.from]+1 { + lay.rank[e.to] = lay.rank[e.from] + 1 + changed = true + } + } + if !changed { + break + } + } + + maxRank := 0 + for _, r := range lay.rank { + if r > maxRank { + maxRank = r + } + } + rows := make([][]int, maxRank+1) + for i, r := range lay.rank { + rows[r] = append(rows[r], i) + } + + // Four alternating barycentre sweeps. + for pass := range 4 { + down := pass%2 == 0 + ranks := make([]int, 0, len(rows)) + if down { + for r := range rows { + ranks = append(ranks, r) + } + } else { + for r := range slices.Backward(rows) { + ranks = append(ranks, r) + } + } + for _, r := range ranks { + if down && r == 0 || !down && r == len(rows)-1 { + continue + } + neighbour := r - 1 + if !down { + neighbour = r + 1 + } + position := map[int]int{} + for p, i := range rows[neighbour] { + position[i] = p + } + bary := map[int]int{} + sum := map[int]int{} + for _, e := range d.edges { + from, to := e.from, e.to + if !down { + from, to = to, from + } + if lay.rank[from] == neighbour && lay.rank[to] == r { + if p, ok := position[from]; ok { + sum[to] += p + bary[to]++ + } + } + } + sort.SliceStable(rows[r], func(a, b int) bool { + ia, ib := rows[r][a], rows[r][b] + switch { + case bary[ia] == 0 && bary[ib] == 0: + return ia < ib + case bary[ia] == 0: + return false + case bary[ib] == 0: + return true + } + ma := sum[ia] / bary[ia] + mb := sum[ib] / bary[ib] + if ma != mb { + return ma < mb + } + return ia < ib + }) + } + } + + // Rank heights, row widths and positions. + rowHeight := make([]int, len(rows)) + rowWidth := make([]int, len(rows)) + for r, row := range rows { + for _, i := range row { + if lay.laid[i].h > rowHeight[r] { + rowHeight[r] = lay.laid[i].h + } + rowWidth[r] += lay.laid[i].w + laneGap + } + rowWidth[r] -= laneGap + } + maxWidth := 0 + y := margin + rowY := make([]int, len(rows)) + for r := range rows { + rowY[r] = y + rowHeight[r]/2 + y += rowHeight[r] + rankGap + if rowWidth[r] > maxWidth { + maxWidth = rowWidth[r] + } + } + for r, row := range rows { + x := margin + (maxWidth-rowWidth[r])/2 + laneGap/2 + for _, i := range row { + x += lay.laid[i].w / 2 + lay.laid[i].x = x + lay.laid[i].y = rowY[r] + x += lay.laid[i].w/2 + laneGap/2 + } + } + lay.width = maxWidth + 2*margin + lay.height = y - rankGap + margin + + // A subgraph with its own direction lays its members out inside the + // box they occupy, in that direction. + for _, sg := range d.subgraphs { + if sg.dir == "" || sg.dir == d.dir || len(sg.nodes) < 2 { + continue + } + reLayoutSubgraph(d, lay, sg) + } + return lay +} + +func reLayoutSubgraph(d *flowDiagram, lay *flowLayout, sg *flowSubgraph) { + minX, minY, maxX, maxY := boundingBox(lay, sg.nodes) + horizontal := sg.dir == "LR" || sg.dir == "RL" + members := map[int]bool{} + for _, i := range sg.nodes { + members[i] = true + } + rank := map[int]int{} + for pass := 0; pass <= len(sg.nodes); pass++ { + changed := false + for _, e := range d.edges { + if members[e.from] && members[e.to] && rank[e.to] < rank[e.from]+1 { + rank[e.to] = rank[e.from] + 1 + changed = true + } + } + if !changed { + break + } + } + maxRank := 0 + for _, i := range sg.nodes { + if rank[i] > maxRank { + maxRank = rank[i] + } + } + rows := make([][]int, maxRank+1) + for _, i := range sg.nodes { + rows[rank[i]] = append(rows[rank[i]], i) + } + centreS := (minX + maxX) / 2 + centreP := (minY + maxY) / 2 + if horizontal { + centreS, centreP = centreP, centreS + } + spanP := (maxY - minY) - laneGap + if horizontal { + spanP = (maxX - minX) - laneGap + } + for r, row := range rows { + p := centreP - spanP/2 + (spanP*(2*r+1))/(2*(maxRank+1)) + var sSize int + for _, i := range row { + if horizontal { + sSize += lay.laid[i].h + } else { + sSize += lay.laid[i].w + } + } + sSize += laneGap / 2 * (len(row) - 1) + s := centreS - sSize/2 + for _, i := range row { + if horizontal { + s += lay.laid[i].h / 2 + lay.laid[i].x = p + lay.laid[i].y = s + s += lay.laid[i].h/2 + laneGap/2 + } else { + s += lay.laid[i].w / 2 + lay.laid[i].y = p + lay.laid[i].x = s + s += lay.laid[i].w/2 + laneGap/2 + } + } + } + // Keep every member inside the box. + for _, i := range sg.nodes { + l := &lay.laid[i] + l.x = clamp(l.x, minX+l.w/2, maxX-l.w/2) + l.y = clamp(l.y, minY+l.h/2, maxY-l.h/2) + } +} + +func clamp(v, low, high int) int { + if v < low { + return low + } + if v > high { + return high + } + return v +} + +func boundingBox(lay *flowLayout, nodes []int) (int, int, int, int) { + minX, minY := 1<<30, 1<<30 + maxX, maxY := -1<<30, -1<<30 + for _, i := range nodes { + l := lay.laid[i] + minX = min(minX, l.x-l.w/2) + maxX = max(maxX, l.x+l.w/2) + minY = min(minY, l.y-l.h/2) + maxY = max(maxY, l.y+l.h/2) + } + return minX, minY, maxX, maxY +} + +func nodeSize(node *flowNode) (int, int) { + lines := labelLines(node.label) + width := 0 + for _, l := range lines { + width = max(width, textWidth(l, 14)) + } + h := 26 + 18*len(lines) + w := max(width+textPad, nodeMinWidth) + switch node.shape { + case "diamond": + w = max(width*2+textPad*2, 150) + h = max(30+26*len(lines), w/2) + case "circle": + d := max(max(width+40, h), 68) + w, h = d, d + } + return w, h +} + +// mapPoint maps a TB-space point into the final orientation. +func mapPoint(dir string, p point, width, height int) point { + switch dir { + case "BT": + return point{p.x, height - p.y} + case "LR": + return point{p.y, p.x} + case "RL": + return point{height - p.y, p.x} + } + return p +} + +// edgeStroke gives the path attributes for an edge kind, with the styles +// of the linkStyle declarations appended. +func edgeStroke(kind string, styles []stylePair) string { + attrs := ` fill="none" stroke="#555" stroke-width="2"` + switch { + case strings.HasPrefix(kind, "thick"): + attrs = ` fill="none" stroke="#555" stroke-width="3.5"` + case strings.HasPrefix(kind, "dotted"): + attrs = ` fill="none" stroke="#555" stroke-width="2" stroke-dasharray="6 5"` + } + return attrs + styleString(styles) +} + +// nodeStyles gathers the inline styles and the class declarations of a +// node, declaration order preserved. +func (d *flowDiagram) nodeStyles(n *flowNode) []stylePair { + pairs := append([]stylePair{}, n.styles...) + for _, class := range n.classes { + pairs = append(pairs, d.classes[class]...) + } + return pairs +} + +func writeFlowchart(d *flowDiagram, lay *flowLayout) []byte { + width, height := lay.width, lay.height + if d.dir == "LR" || d.dir == "RL" { + width, height = height, width + } + svg := newSVGBuilder(width, height) + + // Subgraph boxes, outer before inner, so the parents frame their + // children. + for _, sg := range slices.Backward(d.subgraphs) { + + if len(sg.nodes) == 0 { + continue + } + minX, minY, maxX, maxY := boundingBox(lay, sg.nodes) + a := mapPoint(d.dir, point{minX - 20, minY - 40}, width, height) + b := mapPoint(d.dir, point{maxX + 20, maxY + 18}, width, height) + x0, y0 := min(a.x, b.x), min(a.y, b.y) + x1, y1 := max(a.x, b.x), max(a.y, b.y) + svg.rect(x0, y0, x1-x0, y1-y0, 8, ` fill="#f5f5f5" fill-opacity="0.7" stroke="#999"`) + svg.text(point{x0 + 10, y0 + 18}, sg.title, "start", ` font-size="14" font-weight="bold"`) + } + + // Edges. + type drawn struct { + p0, c1, c2, p3 point + arrow bool + label string + attrs string + } + var drawnEdges []drawn + for i, e := range d.edges { + a, b := lay.laid[e.from], lay.laid[e.to] + var p0, p3, c1, c2 point + switch { + case e.from == e.to: + p0 = point{a.x + a.w/2, a.y - 8} + p3 = point{a.x + a.w/2, a.y + 8} + c1 = point{a.x + a.w/2 + 46, a.y - 28} + c2 = point{a.x + a.w/2 + 46, a.y + 28} + case lay.rank[e.to] > lay.rank[e.from]: + p0 = point{a.x, a.y + a.h/2} + p3 = point{b.x, b.y - b.h/2} + mid := max((p3.y-p0.y)/2, 24) + c1 = point{p0.x, p0.y + mid} + c2 = point{p3.x, p3.y - mid} + case lay.rank[e.to] < lay.rank[e.from]: + p0 = point{a.x, a.y - a.h/2} + p3 = point{b.x, b.y + b.h/2} + mid := max((p0.y-p3.y)/2, 24) + c1 = point{p0.x, p0.y - mid} + c2 = point{p3.x, p3.y + mid} + default: + if b.x >= a.x { + p0 = point{a.x + a.w/2, a.y} + p3 = point{b.x - b.w/2, b.y} + } else { + p0 = point{a.x - a.w/2, a.y} + p3 = point{b.x + b.w/2, b.y} + } + c1 = point{p0.x + 42, p0.y} + c2 = point{p3.x - 42, p3.y} + } + styles := append([]stylePair{}, d.linkDefault...) + styles = append(styles, d.linkByIndex[i]...) + drawnEdges = append(drawnEdges, drawn{ + p0: mapPoint(d.dir, p0, width, height), + c1: mapPoint(d.dir, c1, width, height), + c2: mapPoint(d.dir, c2, width, height), + p3: mapPoint(d.dir, p3, width, height), + arrow: strings.HasSuffix(e.kind, "-arrow"), + label: e.label, + attrs: edgeStroke(e.kind, styles), + }) + } + for _, e := range drawnEdges { + svg.path(fmt.Sprintf("M %d %d C %d %d, %d %d, %d %d", e.p0.x, e.p0.y, e.c1.x, e.c1.y, e.c2.x, e.c2.y, e.p3.x, e.p3.y), e.attrs) + } + for _, e := range drawnEdges { + if !e.arrow { + continue + } + svg.polygon(arrowHead(e.p3, e.c2), ` fill="#555"`) + } + + // Nodes. + for i, node := range d.nodes { + l := lay.laid[i] + c := mapPoint(d.dir, point{l.x, l.y}, width, height) + w, h := l.w, l.h + if d.dir == "LR" || d.dir == "RL" { + w, h = h, w + } + attrs := ` fill="#ffffff" stroke="#333" stroke-width="1.5"` + styleString(d.nodeStyles(node)) + switch node.shape { + case "round": + svg.rect(c.x-w/2, c.y-h/2, w, h, 10, attrs) + case "stadium": + svg.rect(c.x-w/2, c.y-h/2, w, h, min(w, h)/2, attrs) + case "circle": + svg.rect(c.x-w/2, c.y-h/2, w, h, w/2, attrs) + case "diamond": + svg.polygon([]point{{c.x, c.y - h/2}, {c.x + w/2, c.y}, {c.x, c.y + h/2}, {c.x - w/2, c.y}}, attrs) + case "hex": + cut := min(20, w/4) + svg.polygon([]point{{c.x - w/2 + cut, c.y - h/2}, {c.x + w/2 - cut, c.y - h/2}, {c.x + w/2, c.y}, {c.x + w/2 - cut, c.y + h/2}, {c.x - w/2 + cut, c.y + h/2}, {c.x - w/2, c.y}}, attrs) + case "asym": + svg.polygon([]point{{c.x - w/2, c.y - h/2}, {c.x + w/2 - 18, c.y - h/2}, {c.x + w/2, c.y}, {c.x + w/2 - 18, c.y + h/2}, {c.x - w/2, c.y + h/2}}, attrs) + case "sub": + svg.rect(c.x-w/2, c.y-h/2, w, h, 0, attrs) + if d.dir == "LR" || d.dir == "RL" { + svg.line(c.x, c.y-h/2+5, c.x, c.y+h/2-5, ` stroke="#333" stroke-width="1.5"`) + svg.line(c.x, c.y-h/2+10, c.x, c.y+h/2-10, ` stroke="#333" stroke-width="1.5"`) + } else { + svg.line(c.x-w/2+5, c.y, c.x+w/2-5, c.y, ` stroke="#333" stroke-width="1.5"`) + } + default: + svg.rect(c.x-w/2, c.y-h/2, w, h, 0, attrs) + } + lines := labelLines(node.label) + for k, ln := range lines { + y := c.y + 5 + (k-(len(lines)-1)/2)*18 + if len(lines)%2 == 0 { + y = c.y - 4 + k*18 + } + svg.text(point{c.x, y}, ln, "middle", ` font-size="14"`) + } + } + + // Edge labels on top. + for _, e := range drawnEdges { + if e.label == "" { + continue + } + mid := point{(e.p0.x + e.p3.x) / 2, (e.p0.y+e.p3.y)/2 - 7} + svg.text(mid, e.label, "middle", ` font-size="12"`) + } + return svg.finish() +} + +// arrowHead builds a filled triangle at tip pointing from the control +// point towards the tip. +func arrowHead(tip, ctrl point) []point { + dx, dy := tip.x-ctrl.x, tip.y-ctrl.y + n := max(abs(dx)+abs(dy), 1) + ux, uy := dx*1000/n, dy*1000/n + base := point{tip.x - ux*11/1000, tip.y - uy*11/1000} + return []point{ + tip, + {base.x - uy*5/1000, base.y + ux*5/1000}, + {base.x + uy*5/1000, base.y - ux*5/1000}, + } +} + +func abs(v int) int { + if v < 0 { + return -v + } + return v +} diff --git a/internal/diagram/flowparse.go b/internal/diagram/flowparse.go new file mode 100644 index 0000000..c5e1bb7 --- /dev/null +++ b/internal/diagram/flowparse.go @@ -0,0 +1,511 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package diagram + +import ( + "fmt" + "strings" +) + +// stylePair is one key:value declaration of a classDef, style or +// linkStyle statement. +type stylePair struct { + key string + value string +} + +type flowNode struct { + id string + label string + shape string + classes []string + styles []stylePair +} + +type flowEdge struct { + from, to int + kind string + label string + styles []stylePair +} + +type flowSubgraph struct { + id, title, dir string + nodes []int + depth int +} + +type flowDiagram struct { + dir string + nodes []*flowNode + index map[string]int + edges []*flowEdge + subgraphs []*flowSubgraph + classes map[string][]stylePair + linkDefault []stylePair + linkByIndex map[int][]stylePair +} + +func parseFlowchart(lines []string, dir string) (*flowDiagram, error) { + d := &flowDiagram{ + dir: dir, + index: map[string]int{}, + classes: map[string][]stylePair{}, + linkByIndex: map[int][]stylePair{}, + } + var stack []*flowSubgraph + for n, line := range lines { + if err := d.parseFlowStatement(line, &stack, n+2); err != nil { + return nil, err + } + } + if len(stack) > 0 { + return nil, fmt.Errorf("diagram: flowchart line %d: subgraph %q has no end", len(lines)+1, stack[len(stack)-1].id) + } + return d, nil +} + +func (d *flowDiagram) parseFlowStatement(line string, stack *[]*flowSubgraph, lineNo int) error { + sc := &scanner{src: []rune(line)} + word := sc.word() + switch strings.ToLower(word) { + case "subgraph": + return d.parseSubgraph(sc, stack, lineNo) + case "end": + if len(*stack) == 0 { + return fmt.Errorf("diagram: flowchart line %d: end without subgraph", lineNo) + } + *stack = (*stack)[:len(*stack)-1] + return nil + case "direction": + sc.skipSpaces() + dir := strings.ToUpper(sc.rest()) + if !flowDirections[dir] { + return fmt.Errorf("diagram: flowchart line %d: unknown flow direction %q", lineNo, dir) + } + if len(*stack) > 0 { + (*stack)[len(*stack)-1].dir = dir + } else { + d.dir = dir + } + return nil + case "classdef": + return d.parseClassDef(sc, lineNo) + case "class": + return d.parseClass(sc, lineNo) + case "style": + return d.parseStyle(sc, lineNo) + case "linkstyle": + return d.parseLinkStyle(sc, lineNo) + case "click": + return fmt.Errorf("diagram: flowchart line %d: click is unsupported, the SVG carries no interactivity", lineNo) + } + sc.pos = 0 + return d.parseChain(sc, *stack, lineNo) +} + +// node returns the index of a node, creating it when it first appears. +// The innermost open subgraph claims a node at its first appearance. +func (d *flowDiagram) node(id, label, shape string, stack []*flowSubgraph) int { + if i, ok := d.index[id]; ok { + if shape != "" { + d.nodes[i].shape = shape + d.nodes[i].label = label + } + return i + } + if label == "" { + label = id + } + if shape == "" { + shape = "rect" + } + n := &flowNode{id: id, label: label, shape: shape} + d.nodes = append(d.nodes, n) + i := len(d.nodes) - 1 + d.index[id] = i + if len(stack) > 0 { + sg := stack[len(stack)-1] + sg.nodes = append(sg.nodes, i) + } + return i +} + +func (d *flowDiagram) parseChain(sc *scanner, stack []*flowSubgraph, lineNo int) error { + left, err := d.parseNodeList(sc, stack, lineNo) + if err != nil { + return err + } + for { + sc.skipSpaces() + if !sc.atEdgeChar() { + return nil + } + kind, label, err := sc.parseEdge(lineNo) + if err != nil { + return err + } + right, err := d.parseNodeList(sc, stack, lineNo) + if err != nil { + return err + } + for _, from := range left { + for _, to := range right { + d.edges = append(d.edges, &flowEdge{from: from, to: to, kind: kind, label: label}) + } + } + left = right + } +} + +// parseNodeList reads one or more nodes separated by &. A missing node is +// an error: an edge must land somewhere. +func (d *flowDiagram) parseNodeList(sc *scanner, stack []*flowSubgraph, lineNo int) ([]int, error) { + var list []int + for { + i, err := d.parseNode(sc, stack, lineNo) + if err != nil { + return nil, err + } + list = append(list, i) + sc.skipSpaces() + if sc.peek() == '&' { + sc.pos++ + continue + } + return list, nil + } +} + +func (d *flowDiagram) parseNode(sc *scanner, stack []*flowSubgraph, lineNo int) (int, error) { + sc.skipSpaces() + var id strings.Builder + for !sc.eof() && isIDRune(sc.peek()) { + id.WriteRune(sc.peek()) + sc.pos++ + } + if id.Len() == 0 { + return 0, fmt.Errorf("diagram: flowchart line %d: expected a node", lineNo) + } + label, shape := sc.parseShape() + return d.node(id.String(), label, shape, stack), nil +} + +// parseShape reads an optional shape with its label. +func (sc *scanner) parseShape() (string, string) { + switch { + case sc.hasPrefix("[["): + return sc.bracketed("[[", "]]"), "sub" + case sc.hasPrefix("(["): + return sc.bracketed("([", "])"), "stadium" + case sc.hasPrefix("(("): + return sc.bracketed("((", "))"), "circle" + case sc.hasPrefix("["): + return sc.bracketed("[", "]"), "rect" + case sc.hasPrefix("{{"): + return sc.bracketed("{{", "}}"), "hex" + case sc.hasPrefix("{"): + return sc.bracketed("{", "}"), "diamond" + case sc.hasPrefix("("): + return sc.bracketed("(", ")"), "round" + case sc.hasPrefix(">"): + return sc.bracketed(">", "]"), "asym" + } + return "", "" +} + +// bracketed consumes the opener and reads until the closer, honouring +// double quotes. A label that is exactly a quoted string loses its +// quotes. +func (sc *scanner) bracketed(open, close string) string { + sc.pos += len(open) + start := sc.pos + for !sc.eof() { + if sc.peek() == '"' { + sc.pos++ + for !sc.eof() && sc.peek() != '"' { + sc.pos++ + } + sc.pos++ + continue + } + if sc.hasPrefix(close) { + label := string(sc.src[start:sc.pos]) + sc.pos += len(close) + return unquote(strings.TrimSpace(label)) + } + sc.pos++ + } + return unquote(strings.TrimSpace(string(sc.src[start:]))) +} + +func unquote(s string) string { + if len(s) >= 2 && strings.HasPrefix(s, `"`) && strings.HasSuffix(s, `"`) { + return s[1 : len(s)-1] + } + return s +} + +func (d *flowDiagram) parseSubgraph(sc *scanner, stack *[]*flowSubgraph, lineNo int) error { + sc.skipSpaces() + id := sc.word() + title := id + sc.skipSpaces() + switch { + case id == "": + id = fmt.Sprintf("subgraph-%d", len(d.subgraphs)) + title = id + case sc.hasPrefix("["): + title = sc.bracketed("[", "]") + default: + rest := strings.TrimSpace(sc.rest()) + if rest != "" { + id = id + " " + rest + title = id + } + } + sg := &flowSubgraph{id: id, title: title, depth: len(*stack)} + d.subgraphs = append(d.subgraphs, sg) + *stack = append(*stack, sg) + return nil +} + +func (d *flowDiagram) parseClassDef(sc *scanner, lineNo int) error { + sc.skipSpaces() + name := sc.word() + if name == "" { + return fmt.Errorf("diagram: flowchart line %d: classDef needs a name", lineNo) + } + pairs, err := parseStylePairs(sc.rest()) + if err != nil { + return fmt.Errorf("diagram: flowchart line %d: %v", lineNo, err) + } + d.classes[name] = append(d.classes[name], pairs...) + return nil +} + +func (d *flowDiagram) parseClass(sc *scanner, lineNo int) error { + sc.skipSpaces() + ids := strings.Split(sc.word(), ",") + sc.skipSpaces() + names := strings.Split(strings.TrimSpace(sc.rest()), ",") + if len(ids) == 0 || ids[0] == "" || len(names) == 0 || names[0] == "" { + return fmt.Errorf("diagram: flowchart line %d: class needs nodes and a class name", lineNo) + } + for _, id := range ids { + id = strings.TrimSpace(id) + if id == "" { + continue + } + i, ok := d.index[id] + if !ok { + i = d.node(id, "", "", nil) + } + for _, name := range names { + name = strings.TrimSpace(name) + if name != "" { + d.nodes[i].classes = append(d.nodes[i].classes, name) + } + } + } + return nil +} + +func (d *flowDiagram) parseStyle(sc *scanner, lineNo int) error { + sc.skipSpaces() + id := sc.word() + if id == "" { + return fmt.Errorf("diagram: flowchart line %d: style needs a node", lineNo) + } + pairs, err := parseStylePairs(sc.rest()) + if err != nil { + return fmt.Errorf("diagram: flowchart line %d: %v", lineNo, err) + } + i, ok := d.index[id] + if !ok { + i = d.node(id, "", "", nil) + } + d.nodes[i].styles = append(d.nodes[i].styles, pairs...) + return nil +} + +func (d *flowDiagram) parseLinkStyle(sc *scanner, lineNo int) error { + sc.skipSpaces() + target := sc.word() + rest := sc.rest() + pairs, err := parseStylePairs(rest) + if err != nil { + return fmt.Errorf("diagram: flowchart line %d: %v", lineNo, err) + } + if strings.ToLower(target) == "default" { + d.linkDefault = append(d.linkDefault, pairs...) + return nil + } + for part := range strings.SplitSeq(target, ",") { + var n int + if _, err := fmt.Sscanf(strings.TrimSpace(part), "%d", &n); err != nil { + return fmt.Errorf("diagram: flowchart line %d: linkStyle needs an index or default", lineNo) + } + d.linkByIndex[n] = append(d.linkByIndex[n], pairs...) + } + return nil +} + +// parseStylePairs reads comma separated key:value declarations. +func parseStylePairs(s string) ([]stylePair, error) { + var pairs []stylePair + for part := range strings.SplitSeq(s, ",") { + part = strings.TrimSpace(part) + if part == "" { + continue + } + key, value, found := strings.Cut(part, ":") + if !found { + return nil, fmt.Errorf("expected key:value, got %q", part) + } + pairs = append(pairs, stylePair{key: strings.TrimSpace(key), value: unquote(strings.TrimSpace(value))}) + } + if len(pairs) == 0 { + return nil, fmt.Errorf("expected style declarations") + } + return pairs, nil +} + +func isIDRune(r rune) bool { + return r >= 'a' && r <= 'z' || r >= 'A' && r <= 'Z' || r >= '0' && r <= '9' || r == '_' +} + +// scanner walks one statement. +type scanner struct { + src []rune + pos int +} + +func (sc *scanner) eof() bool { return sc.pos >= len(sc.src) } + +func (sc *scanner) peek() rune { + if sc.eof() { + return 0 + } + return sc.src[sc.pos] +} + +func (sc *scanner) skipSpaces() { + for !sc.eof() && (sc.peek() == ' ' || sc.peek() == '\t') { + sc.pos++ + } +} + +func (sc *scanner) rest() string { return string(sc.src[sc.pos:]) } + +func (sc *scanner) word() string { + start := sc.pos + for !sc.eof() && sc.peek() != ' ' && sc.peek() != '\t' { + sc.pos++ + } + return string(sc.src[start:sc.pos]) +} + +func (sc *scanner) hasPrefix(s string) bool { + runes := []rune(s) + if sc.pos+len(runes) > len(sc.src) { + return false + } + for i, r := range runes { + if sc.src[sc.pos+i] != r { + return false + } + } + return true +} + +func (sc *scanner) atEdgeChar() bool { + c := sc.peek() + return c == '-' || c == '.' || c == '=' +} + +func isEdgeRune(r rune) bool { return r == '-' || r == '.' || r == '=' } + +// edgeRun consumes a run of edge characters and returns it. +func (sc *scanner) edgeRun() string { + start := sc.pos + for !sc.eof() && isEdgeRune(sc.peek()) { + sc.pos++ + } + return string(sc.src[start:sc.pos]) +} + +// parseEdge reads one edge token: a run of -, . and = characters, an +// optional arrow head, and a label either in pipes or as text between two +// runs. +func (sc *scanner) parseEdge(lineNo int) (kind, label string, err error) { + runs := sc.edgeRun() + arrow := false + switch { + case sc.peek() == '|': + sc.pos++ + start := sc.pos + for !sc.eof() && sc.peek() != '|' { + sc.pos++ + } + label = unquote(strings.TrimSpace(string(sc.src[start:sc.pos]))) + if !sc.eof() { + sc.pos++ + } + case sc.peek() == '>': + sc.pos++ + arrow = true + if sc.peek() == '|' { + // A pipe label behind the arrow head: -->|text| + sc.pos++ + start := sc.pos + for !sc.eof() && sc.peek() != '|' { + sc.pos++ + } + label = unquote(strings.TrimSpace(string(sc.src[start:sc.pos]))) + if !sc.eof() { + sc.pos++ + } + } + default: + save := sc.pos + sc.skipSpaces() + if !sc.eof() && !isEdgeRune(sc.peek()) { + start := sc.pos + for !sc.eof() && !isEdgeRune(sc.peek()) { + sc.pos++ + } + second := sc.edgeRun() + if second == "" { + // No closing run: the words ahead are the next node, not a + // label. + sc.pos = save + } else { + label = unquote(strings.TrimSpace(string(sc.src[start : sc.pos-len(second)]))) + runs += second + if sc.peek() == '>' { + sc.pos++ + arrow = true + } + } + } else { + sc.pos = save + } + } + if len(runs) < 2 || !arrow && len(runs) < 3 { + return "", "", fmt.Errorf("diagram: flowchart line %d: unfinished edge token", lineNo) + } + switch { + case strings.Contains(runs, "="): + kind = "thick" + case strings.Contains(runs, "."): + kind = "dotted" + default: + kind = "solid" + } + if arrow { + kind += "-arrow" + } + return kind, label, nil +} diff --git a/internal/diagram/sequence.go b/internal/diagram/sequence.go new file mode 100644 index 0000000..f80af41 --- /dev/null +++ b/internal/diagram/sequence.go @@ -0,0 +1,569 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package diagram + +import ( + "fmt" + "strings" +) + +// Sequence layout constants, in SVG units. +const ( + seqMargin = 24 + seqHeaderTop = 24 + seqHeaderH = 44 + seqRowGap = 30 +) + +type seqParticipant struct { + id, name string + actor bool + x int +} + +// seqItem is one row: a message, a note, a divider, a frame boundary or +// an activation change. +type seqItem struct { + kind string + from string + to string + arrow string + text string + num int + side string + over string + frame *seqFrame + act string + actOn bool + y, h int +} + +// seqFrame is one frame of a block construct: from its label row to the +// row of the boundary that closes it. +type seqFrame struct { + kind, label string + colour string + first, last int + depth int +} + +type seqDiagram struct { + participants []*seqParticipant + index map[string]int + items []*seqItem + frames []*seqFrame + autonumber bool +} + +type seqConstruct struct { + frame *seqFrame +} + +func renderSequence(lines []string, auto bool) ([]byte, error) { + d, err := parseSequence(lines, auto) + if err != nil { + return nil, err + } + return writeSequence(d), nil +} + +func (d *seqDiagram) participant(id string) int { + if i, ok := d.index[id]; ok { + return i + } + p := &seqParticipant{id: id, name: id} + d.participants = append(d.participants, p) + i := len(d.participants) - 1 + d.index[id] = i + return i +} + +func parseSequence(lines []string, auto bool) (*seqDiagram, error) { + d := &seqDiagram{index: map[string]int{}, autonumber: auto} + var stack []*seqConstruct + for n, line := range lines { + lineNo := n + 2 + if err := d.parseSeqStatement(line, &stack, lineNo); err != nil { + return nil, err + } + } + if len(stack) > 0 { + return nil, fmt.Errorf("diagram: sequence line %d: block %q has no end", len(lines)+1, stack[len(stack)-1].frame.kind) + } + return d, nil +} + +var seqBlockOpeners = map[string]bool{ + "alt": true, "opt": true, "loop": true, "par": true, + "critical": true, "break": true, +} + +var seqBlockElse = map[string]bool{ + "else": true, "and": true, "option": true, +} + +func (d *seqDiagram) parseSeqStatement(line string, stack *[]*seqConstruct, lineNo int) error { + lower := strings.ToLower(line) + keyword := firstWord(lower) + switch { + case strings.HasPrefix(lower, "participant "), strings.HasPrefix(lower, "actor "): + return d.parseSeqParticipant(line, lineNo) + case keyword == "autonumber": + d.autonumber = !strings.HasSuffix(lower, "off") + return nil + case strings.HasPrefix(lower, "activate "), strings.HasPrefix(lower, "deactivate "): + on := strings.HasPrefix(lower, "activate") + id := strings.TrimSpace(line[len("activate "):]) + if strings.HasPrefix(lower, "deactivate ") { + id = strings.TrimSpace(line[len("deactivate "):]) + } + if id == "" { + return fmt.Errorf("diagram: sequence line %d: activation needs a participant", lineNo) + } + d.participant(id) + d.items = append(d.items, &seqItem{kind: "act", act: id, actOn: on, h: 8}) + return nil + case strings.HasPrefix(lower, "note "): + return d.parseSeqNote(line, lineNo) + case strings.HasPrefix(line, "..."): + text := strings.Trim(line, ". ") + d.items = append(d.items, &seqItem{kind: "divider", text: text, h: 34}) + return nil + case seqBlockOpeners[keyword] || (keyword == "rect" && strings.HasPrefix(lower, "rect ")): + text := strings.TrimSpace(line[len(keyword):]) + frame := &seqFrame{kind: keyword, depth: len(*stack)} + if keyword == "rect" { + colour, err := parseRectColour(text, lineNo) + if err != nil { + return err + } + frame.colour = colour + } else { + frame.label = keyword + " " + text + } + frame.first = len(d.items) + d.frames = append(d.frames, frame) + d.items = append(d.items, &seqItem{kind: "open", frame: frame, h: 30}) + *stack = append(*stack, &seqConstruct{frame: frame}) + return nil + case seqBlockElse[keyword]: + if len(*stack) == 0 { + return fmt.Errorf("diagram: sequence line %d: %s outside a block", lineNo, keyword) + } + text := strings.TrimSpace(line[len(keyword):]) + construct := (*stack)[len(*stack)-1] + construct.frame.last = len(d.items) - 1 + frame := &seqFrame{kind: keyword, label: keyword + " " + text, depth: construct.frame.depth} + frame.first = len(d.items) + d.frames = append(d.frames, frame) + d.items = append(d.items, &seqItem{kind: "open", frame: frame, h: 30}) + construct.frame = frame + return nil + case keyword == "end": + if len(*stack) == 0 { + return fmt.Errorf("diagram: sequence line %d: end without a block", lineNo) + } + construct := (*stack)[len(*stack)-1] + *stack = (*stack)[:len(*stack)-1] + construct.frame.last = len(d.items) + d.items = append(d.items, &seqItem{kind: "close", frame: construct.frame, h: 14}) + return nil + } + return d.parseSeqMessage(line, lineNo) +} + +func firstWord(s string) string { + w, _, _ := strings.Cut(s, " ") + return w +} + +func (d *seqDiagram) parseSeqParticipant(line string, lineNo int) error { + sc := &scanner{src: []rune(line)} + sc.word() // participant or actor + actor := strings.HasPrefix(strings.ToLower(line), "actor") + sc.skipSpaces() + id := sc.word() + if id == "" { + return fmt.Errorf("diagram: sequence line %d: participant needs an id", lineNo) + } + sc.skipSpaces() + name := id + if strings.HasPrefix(sc.rest(), "as ") { + name = unquote(strings.TrimSpace(sc.rest()[3:])) + } else if strings.HasPrefix(sc.rest(), "[") { + name = sc.bracketed("[", "]") + } else if rest := strings.TrimSpace(sc.rest()); rest != "" { + name = unquote(rest) + } + i := d.participant(id) + d.participants[i].name = name + d.participants[i].actor = actor + return nil +} + +func (d *seqDiagram) parseSeqNote(line string, lineNo int) error { + rest := strings.TrimSpace(line[len("note "):]) + lower := strings.ToLower(rest) + var side, spec string + switch { + case strings.HasPrefix(lower, "left of "): + side, spec = "left", strings.TrimSpace(rest[len("left of "):]) + case strings.HasPrefix(lower, "right of "): + side, spec = "right", strings.TrimSpace(rest[len("right of "):]) + case strings.HasPrefix(lower, "over "): + side, spec = "over", strings.TrimSpace(rest[len("over "):]) + default: + return fmt.Errorf("diagram: sequence line %d: note needs left of, right of or over", lineNo) + } + colon := strings.Index(spec, ":") + if colon < 0 { + return fmt.Errorf("diagram: sequence line %d: note needs a colon before its text", lineNo) + } + text := strings.TrimSpace(spec[colon+1:]) + spec = strings.TrimSpace(spec[:colon]) + id, over := spec, "" + if side == "over" { + if a, b, found := strings.Cut(spec, ","); found { + id, over = strings.TrimSpace(a), strings.TrimSpace(b) + } + } + if id == "" { + return fmt.Errorf("diagram: sequence line %d: note needs a participant", lineNo) + } + d.participant(id) + if over != "" { + d.participant(over) + } + d.items = append(d.items, &seqItem{kind: "note", side: side, from: id, over: over, text: text}) + return nil +} + +// seqArrows lists the message arrows longest first, so the scan prefers +// the long form. +var seqArrows = []string{"-->>", "-->", "->>", "-x", "--x", "->"} + +// parseSeqMessage reads FROM arrow TO: text. +func (d *seqDiagram) parseSeqMessage(line string, lineNo int) error { + bestAt, bestLen := -1, 0 + for _, a := range seqArrows { + if at := strings.Index(line, a); at >= 0 && (bestAt < 0 || at < bestAt || at == bestAt && len(a) > bestLen) { + if bestAt < 0 || at < bestAt || len(a) > bestLen { + bestAt, bestLen = at, len(a) + } + } + } + if bestAt < 0 { + return fmt.Errorf("diagram: sequence line %d: cannot parse %q", lineNo, line) + } + arrow := line[bestAt : bestAt+bestLen] + left := strings.TrimSpace(line[:bestAt]) + rest := strings.TrimSpace(line[bestAt+bestLen:]) + before, after, ok := strings.Cut(rest, ":") + if !ok { + return fmt.Errorf("diagram: sequence line %d: message needs a colon before its text", lineNo) + } + right := strings.TrimSpace(before) + text := strings.TrimSpace(after) + + // Activation flags cling to the participant names. + fromAct, fromDeact := false, false + if before, ok := strings.CutSuffix(left, "+"); ok { + fromAct, left = true, before + } else if before, ok := strings.CutSuffix(left, "-"); ok { + fromDeact, left = true, before + } + left = strings.TrimSpace(left) + toAct, toDeact := false, false + right = strings.TrimSpace(right) + if after, ok := strings.CutPrefix(right, "+"); ok { + toAct, right = true, after + } else if after, ok := strings.CutPrefix(right, "-"); ok { + toDeact, right = true, after + } + if before, ok := strings.CutSuffix(right, "+"); ok { + toAct, right = true, before + } else if before, ok := strings.CutSuffix(right, "-"); ok { + toDeact, right = true, before + } + right = strings.TrimSpace(right) + if left == "" || right == "" { + return fmt.Errorf("diagram: sequence line %d: message needs two participants", lineNo) + } + d.participant(left) + d.participant(right) + num := 0 + if d.autonumber { + num = d.countMessages() + 1 + } + d.items = append(d.items, &seqItem{ + kind: "msg", from: left, to: right, arrow: arrow, text: text, num: num, + }) + if fromAct { + d.activate(left, lineNo) + } + if toAct { + d.activate(right, lineNo) + } + if fromDeact { + d.deactivate(left, lineNo) + } + if toDeact { + d.deactivate(right, lineNo) + } + return nil +} + +func (d *seqDiagram) countMessages() int { + n := 0 + for _, it := range d.items { + if it.kind == "msg" { + n++ + } + } + return n +} + +// activations collects the open and closed activation spans while the +// items lay out; the parser records the changes as items. +type seqActivation struct { + id string + y1 int + y2 int + open bool +} + +func (d *seqDiagram) activate(id string, lineNo int) { + d.items = append(d.items, &seqItem{kind: "act", act: id, actOn: true, h: 8}) +} + +func (d *seqDiagram) deactivate(id string, lineNo int) { + d.items = append(d.items, &seqItem{kind: "act", act: id, actOn: false, h: 8}) +} + +// parseRectColour reads rgb(r,g,b) or #hex. +func parseRectColour(s string, lineNo int) (string, error) { + s = strings.TrimSpace(s) + if strings.HasPrefix(s, "rgb(") && strings.HasSuffix(s, ")") { + parts := strings.Split(strings.TrimSuffix(strings.TrimPrefix(s, "rgb("), ")"), ",") + if len(parts) != 3 { + return "", fmt.Errorf("diagram: sequence line %d: rgb takes three numbers", lineNo) + } + r, g, b := 0, 0, 0 + var err error + if _, err = fmt.Sscanf(strings.TrimSpace(parts[0]), "%d", &r); err != nil { + return "", fmt.Errorf("diagram: sequence line %d: rgb takes three numbers", lineNo) + } + if _, err = fmt.Sscanf(strings.TrimSpace(parts[1]), "%d", &g); err != nil { + return "", fmt.Errorf("diagram: sequence line %d: rgb takes three numbers", lineNo) + } + if _, err = fmt.Sscanf(strings.TrimSpace(parts[2]), "%d", &b); err != nil { + return "", fmt.Errorf("diagram: sequence line %d: rgb takes three numbers", lineNo) + } + return fmt.Sprintf("#%02x%02x%02x", clamp(r, 0, 255), clamp(g, 0, 255), clamp(b, 0, 255)), nil + } + if strings.HasPrefix(s, "#") && len(s) == 7 { + for i := 1; i < len(s); i++ { + c := s[i] + if !(c >= '0' && c <= '9' || c >= 'a' && c <= 'f' || c >= 'A' && c <= 'F') { + return "", fmt.Errorf("diagram: sequence line %d: bad colour %q", lineNo, s) + } + } + return s, nil + } + return "", fmt.Errorf("diagram: sequence line %d: rect needs rgb(r,g,b) or #rrggbb", lineNo) +} + +func writeSequence(d *seqDiagram) []byte { + // Lanes. + laneWidth := 150 + for _, p := range d.participants { + laneWidth = max(laneWidth, textWidth(p.name, 14)+80) + } + width := seqMargin*2 + laneWidth*len(d.participants) + for i, p := range d.participants { + p.x = seqMargin + laneWidth*i + laneWidth/2 + } + + // Rows. + y := seqMargin + seqHeaderH + seqRowGap + for _, it := range d.items { + switch it.kind { + case "note": + it.h = 16 + 18*len(labelLines(it.text)) + case "msg": + it.h = 38 + } + it.y = y + y += it.h + 10 + } + bottom := y + 10 + if len(d.items) == 0 { + bottom = seqMargin + seqHeaderH + seqRowGap + 40 + } + height := bottom + seqMargin + + svg := newSVGBuilder(width, height) + + // Frame backgrounds, then lifelines, headers, activations, notes and + // messages, then frame outlines. + for _, f := range d.frames { + if f.colour == "" { + continue + } + x0 := seqMargin + f.depth*12 + fw := width - 2*seqMargin - 2*f.depth*12 + top := d.items[f.first].y + 4 + low := d.items[f.last].y + d.items[f.last].h + svg.rect(x0, top, fw, low-top, 6, fmt.Sprintf(` fill="%s" fill-opacity="0.18" stroke="none"`, f.colour)) + } + + lifelineBottom := bottom + for _, p := range d.participants { + svg.line(p.x, seqMargin+seqHeaderH, p.x, lifelineBottom, ` stroke="#999" stroke-dasharray="5 5"`) + } + + // Activations: the act items, in row order, open and close the spans. + var activations []seqActivation + open := map[string][]int{} + for _, it := range d.items { + if it.kind != "act" { + continue + } + y := it.y + it.h/2 + if it.actOn { + open[it.act] = append(open[it.act], y) + } else if len(open[it.act]) > 0 { + start := open[it.act][len(open[it.act])-1] + open[it.act] = open[it.act][:len(open[it.act])-1] + activations = append(activations, seqActivation{id: it.act, y1: start, y2: y}) + } + } + for _, a := range activations { + x := d.participants[d.index[a.id]].x + svg.rect(x-5, a.y1, 10, max(a.y2-a.y1, 8), 0, ` fill="#e8e8e8" stroke="#666" stroke-width="1"`) + } + + // Participant headers. + for _, p := range d.participants { + attrs := ` fill="#eef2f7" stroke="#333" stroke-width="1.5"` + if p.actor { + attrs = ` fill="#e8f0e8" stroke="#333" stroke-width="2.5"` + } + w := max(laneWidth-24, textWidth(p.name, 14)+30) + svg.rect(p.x-w/2, seqHeaderTop, w, seqHeaderH, 8, attrs) + svg.text(point{p.x, seqHeaderTop + 27}, p.name, "middle", ` font-size="14" font-weight="bold"`) + } + + // Messages. + for _, it := range d.items { + if it.kind != "msg" { + continue + } + xA := d.participants[d.index[it.from]].x + xB := d.participants[d.index[it.to]].x + y := it.y + 28 + text := it.text + if it.num > 0 { + text = fmt.Sprintf("%d: %s", it.num, text) + } + attrs := ` stroke="#333" stroke-width="1.8"` + if strings.HasPrefix(it.arrow, "--") { + attrs += ` stroke-dasharray="7 5"` + } + if xA == xB { + svg.path(fmt.Sprintf("M %d %d H %d V %d H %d", xA+8, y-14, xA+64, y+4, xA+14), attrs) + if strings.HasSuffix(it.arrow, "x") { + svg.line(xA+14, y-4, xA+26, y+12, ` stroke="#333" stroke-width="1.8"`) + svg.line(xA+26, y-4, xA+14, y+12, ` stroke="#333" stroke-width="1.8"`) + } else { + svg.polygon([]point{{xA + 14, y + 4}, {xA + 25, y - 1}, {xA + 25, y + 9}}, ` fill="#333"`) + } + svg.text(point{xA + 74, y - 6}, text, "start", ` font-size="13"`) + continue + } + svg.line(xA, y, xB, y, attrs) + s := 1 + if xB < xA { + s = -1 + } + tip := xB + if strings.HasSuffix(it.arrow, "x") { + svg.line(tip-5*s, y-6, tip+4*s, y+6, ` stroke="#333" stroke-width="1.8"`) + svg.line(tip+4*s, y-6, tip-5*s, y+6, ` stroke="#333" stroke-width="1.8"`) + } else if strings.HasSuffix(it.arrow, ">") { + svg.polygon([]point{{tip, y}, {tip - 11*s, y - 5}, {tip - 11*s, y + 5}}, ` fill="#333"`) + } + midX := (xA + xB) / 2 + svg.text(point{midX, y - 7}, text, "middle", ` font-size="13"`) + } + + // Notes. + for _, it := range d.items { + if it.kind != "note" { + continue + } + lines := labelLines(it.text) + w := 0 + for _, l := range lines { + w = max(w, textWidth(l, 13)) + } + w += 26 + h := it.h - 4 + var x int + anchor := "start" + switch it.side { + case "left": + x = d.participants[d.index[it.from]].x - 14 - w + if it.over != "" { + x = d.participants[d.index[it.over]].x - 14 - w + } + case "right": + x = d.participants[d.index[it.from]].x + 14 + default: + xA := d.participants[d.index[it.from]].x + xB := xA + if it.over != "" { + xB = d.participants[d.index[it.over]].x + } + x = min(xA, xB) - 30 + w = abs(xB-xA) + 60 + anchor = "middle" + } + svg.rect(x, it.y+2, w, h, 4, ` fill="#fffbe0" stroke="#999"`) + for k, l := range lines { + tx := x + 12 + if anchor == "middle" { + tx = x + w/2 + } + svg.text(point{tx, it.y + 22 + k*18}, l, anchor, ` font-size="13"`) + } + } + + // Dividers. + for _, it := range d.items { + if it.kind != "divider" { + continue + } + y := it.y + 17 + svg.line(seqMargin, y, width-seqMargin, y, ` stroke="#999" stroke-dasharray="6 5"`) + w := textWidth(it.text, 13) + 24 + svg.rect((width-w)/2, y-11, w, 22, 10, ` fill="#eee" stroke="#999"`) + svg.text(point{width / 2, y + 4}, it.text, "middle", ` font-size="13"`) + } + + // Frame outlines and label tabs. + for _, f := range d.frames { + x0 := seqMargin + f.depth*12 + fw := width - 2*seqMargin - 2*f.depth*12 + top := d.items[f.first].y + 4 + low := d.items[f.last].y + d.items[f.last].h + svg.rect(x0, top, fw, low-top, 6, ` fill="none" stroke="#888"`) + if f.label != "" { + tw := min(textWidth(f.label, 13)+16, fw) + svg.rect(x0, top, tw, 22, 0, ` fill="#eee" stroke="#888"`) + svg.text(point{x0 + 8, top + 15}, f.label, "start", ` font-size="13"`) + } + } + return svg.finish() +} diff --git a/internal/diagram/svg.go b/internal/diagram/svg.go new file mode 100644 index 0000000..45c9857 --- /dev/null +++ b/internal/diagram/svg.go @@ -0,0 +1,145 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package diagram + +import ( + "fmt" + "strings" +) + +// point is a coordinate in the layout. Everything is an integer, so the +// same layout always writes the same bytes. +type point struct { + x, y int +} + +// svgBuilder writes SVG elements in the order they are given. +type svgBuilder struct { + b strings.Builder + width int + height int +} + +func newSVGBuilder(width, height int) *svgBuilder { + s := &svgBuilder{width: width, height: height} + fmt.Fprintf(&s.b, ``, width, height) + return s +} + +// rect draws a rectangle, rounded when rx is positive. +func (s *svgBuilder) rect(x, y, w, h, rx int, attrs string) { + if w < 0 || h < 0 { + return + } + if rx > 0 { + fmt.Fprintf(&s.b, ``, x, y, w, h, rx, attrs) + } else { + fmt.Fprintf(&s.b, ``, x, y, w, h, attrs) + } +} + +func (s *svgBuilder) line(x1, y1, x2, y2 int, attrs string) { + fmt.Fprintf(&s.b, ``, x1, y1, x2, y2, attrs) +} + +func (s *svgBuilder) path(d, attrs string) { + s.b.WriteString(``) +} + +// text writes one line of text. The anchor is middle, start or end. +func (s *svgBuilder) text(p point, str, anchor, attrs string) { + s.b.WriteString(``) + s.b.WriteString(escapeXML(str)) + s.b.WriteString(``) +} + +func (s *svgBuilder) polygon(pts []point, attrs string) { + if len(pts) < 3 { + return + } + s.b.WriteString(``) +} + +func (s *svgBuilder) finish() []byte { + s.b.WriteString("") + return []byte(s.b.String()) +} + +var xmlEscaper = strings.NewReplacer("&", "&", "<", "<", ">", ">", `"`, """) + +func escapeXML(s string) string { + return xmlEscaper.Replace(s) +} + +// textWidth estimates the width a string needs at the given font size. +// Characters outside the basic plane are wide: the estimate stays on the +// generous side so labels keep their padding. +func textWidth(s string, size int) int { + w := 0 + for _, r := range s { + if r > 0x2E7F { + w += size + } else { + w += (size*6 + 4) / 10 + } + } + return w +} + +// labelLines splits a label into its lines. +func labelLines(s string) []string { + if s == "" { + return []string{""} + } + normalised := strings.ReplaceAll(s, "
", "
") + normalised = strings.ReplaceAll(normalised, "
", "
") + lines := strings.Split(normalised, "
") + if len(lines) == 0 { + return []string{""} + } + return lines +} + +// styleString joins style pairs into a style attribute, declaration order +// preserved. +func styleString(pairs []stylePair) string { + if len(pairs) == 0 { + return "" + } + var b strings.Builder + b.WriteString(` style="`) + for i, p := range pairs { + if i > 0 { + b.WriteString(";") + } + b.WriteString(p.key) + b.WriteString(":") + b.WriteString(p.value) + } + b.WriteString(`"`) + return b.String() +} diff --git a/internal/diagram/testdata/fc-basic.svg b/internal/diagram/testdata/fc-basic.svg new file mode 100644 index 0000000..148a73f --- /dev/null +++ b/internal/diagram/testdata/fc-basic.svg @@ -0,0 +1 @@ +StartEnd \ No newline at end of file diff --git a/internal/diagram/testdata/fc-branching.svg b/internal/diagram/testdata/fc-branching.svg new file mode 100644 index 0000000..bb1f36e --- /dev/null +++ b/internal/diagram/testdata/fc-branching.svg @@ -0,0 +1 @@ +ABCDEF \ No newline at end of file diff --git a/internal/diagram/testdata/fc-classes.svg b/internal/diagram/testdata/fc-classes.svg new file mode 100644 index 0000000..aca4c91 --- /dev/null +++ b/internal/diagram/testdata/fc-classes.svg @@ -0,0 +1 @@ +FirstB \ No newline at end of file diff --git a/internal/diagram/testdata/fc-cycle.svg b/internal/diagram/testdata/fc-cycle.svg new file mode 100644 index 0000000..c54a5ea --- /dev/null +++ b/internal/diagram/testdata/fc-cycle.svg @@ -0,0 +1 @@ +ABCD \ No newline at end of file diff --git a/internal/diagram/testdata/fc-direction-line.svg b/internal/diagram/testdata/fc-direction-line.svg new file mode 100644 index 0000000..1b7722c --- /dev/null +++ b/internal/diagram/testdata/fc-direction-line.svg @@ -0,0 +1 @@ +AB \ No newline at end of file diff --git a/internal/diagram/testdata/fc-edge-kinds.svg b/internal/diagram/testdata/fc-edge-kinds.svg new file mode 100644 index 0000000..7c809f4 --- /dev/null +++ b/internal/diagram/testdata/fc-edge-kinds.svg @@ -0,0 +1 @@ +ABCDEFG \ No newline at end of file diff --git a/internal/diagram/testdata/fc-edge-labels.svg b/internal/diagram/testdata/fc-edge-labels.svg new file mode 100644 index 0000000..90845a0 --- /dev/null +++ b/internal/diagram/testdata/fc-edge-labels.svg @@ -0,0 +1 @@ +ABCDEtextpipeddotted textthick text \ No newline at end of file diff --git a/internal/diagram/testdata/fc-graph-alias.svg b/internal/diagram/testdata/fc-graph-alias.svg new file mode 100644 index 0000000..a00b46b --- /dev/null +++ b/internal/diagram/testdata/fc-graph-alias.svg @@ -0,0 +1 @@ +AlphaBeta \ No newline at end of file diff --git a/internal/diagram/testdata/fc-lr.svg b/internal/diagram/testdata/fc-lr.svg new file mode 100644 index 0000000..4175d6d --- /dev/null +++ b/internal/diagram/testdata/fc-lr.svg @@ -0,0 +1 @@ +ABC \ No newline at end of file diff --git a/internal/diagram/testdata/fc-multiline.svg b/internal/diagram/testdata/fc-multiline.svg new file mode 100644 index 0000000..b21820c --- /dev/null +++ b/internal/diagram/testdata/fc-multiline.svg @@ -0,0 +1 @@ +first linesecond lineB \ No newline at end of file diff --git a/internal/diagram/testdata/fc-quoted-labels.svg b/internal/diagram/testdata/fc-quoted-labels.svg new file mode 100644 index 0000000..b76d1a6 --- /dev/null +++ b/internal/diagram/testdata/fc-quoted-labels.svg @@ -0,0 +1 @@ +Label with (brackets) and \"quotes\"'single' \ No newline at end of file diff --git a/internal/diagram/testdata/fc-shapes.svg b/internal/diagram/testdata/fc-shapes.svg new file mode 100644 index 0000000..de50af4 --- /dev/null +++ b/internal/diagram/testdata/fc-shapes.svg @@ -0,0 +1 @@ +RectRoundDiamondCircleSubroutineAsymmetricStadiumHexagon \ No newline at end of file diff --git a/internal/diagram/testdata/fc-single.svg b/internal/diagram/testdata/fc-single.svg new file mode 100644 index 0000000..94d3d35 --- /dev/null +++ b/internal/diagram/testdata/fc-single.svg @@ -0,0 +1 @@ +Lonely \ No newline at end of file diff --git a/internal/diagram/testdata/fc-subgraph-direction.svg b/internal/diagram/testdata/fc-subgraph-direction.svg new file mode 100644 index 0000000..f5ef1f8 --- /dev/null +++ b/internal/diagram/testdata/fc-subgraph-direction.svg @@ -0,0 +1 @@ +InnerABCD \ No newline at end of file diff --git a/internal/diagram/testdata/fc-subgraph.svg b/internal/diagram/testdata/fc-subgraph.svg new file mode 100644 index 0000000..1d30706 --- /dev/null +++ b/internal/diagram/testdata/fc-subgraph.svg @@ -0,0 +1 @@ +Group OneABCD \ No newline at end of file diff --git a/internal/diagram/testdata/sq-activation.svg b/internal/diagram/testdata/sq-activation.svg new file mode 100644 index 0000000..e634981 --- /dev/null +++ b/internal/diagram/testdata/sq-activation.svg @@ -0,0 +1 @@ +ABrequestresponse \ No newline at end of file diff --git a/internal/diagram/testdata/sq-alt.svg b/internal/diagram/testdata/sq-alt.svg new file mode 100644 index 0000000..63e3f54 --- /dev/null +++ b/internal/diagram/testdata/sq-alt.svg @@ -0,0 +1 @@ +ABcheckproceedrefusealt yeselse no \ No newline at end of file diff --git a/internal/diagram/testdata/sq-arrows.svg b/internal/diagram/testdata/sq-arrows.svg new file mode 100644 index 0000000..fd5be01 --- /dev/null +++ b/internal/diagram/testdata/sq-arrows.svg @@ -0,0 +1 @@ +ABsolid arrowdashed arrowsolid crossdashed crosssolid opendashed open \ No newline at end of file diff --git a/internal/diagram/testdata/sq-autonumber.svg b/internal/diagram/testdata/sq-autonumber.svg new file mode 100644 index 0000000..703a14c --- /dev/null +++ b/internal/diagram/testdata/sq-autonumber.svg @@ -0,0 +1 @@ +AB1: first2: second3: self \ No newline at end of file diff --git a/internal/diagram/testdata/sq-basic.svg b/internal/diagram/testdata/sq-basic.svg new file mode 100644 index 0000000..8f9a326 --- /dev/null +++ b/internal/diagram/testdata/sq-basic.svg @@ -0,0 +1 @@ +AliceBobHelloHi \ No newline at end of file diff --git a/internal/diagram/testdata/sq-break.svg b/internal/diagram/testdata/sq-break.svg new file mode 100644 index 0000000..310ddb4 --- /dev/null +++ b/internal/diagram/testdata/sq-break.svg @@ -0,0 +1 @@ +ABstopbreak failure \ No newline at end of file diff --git a/internal/diagram/testdata/sq-critical.svg b/internal/diagram/testdata/sq-critical.svg new file mode 100644 index 0000000..d0c8f60 --- /dev/null +++ b/internal/diagram/testdata/sq-critical.svg @@ -0,0 +1 @@ +ABworkskipcritical lockedoption unlocked \ No newline at end of file diff --git a/internal/diagram/testdata/sq-divider.svg b/internal/diagram/testdata/sq-divider.svg new file mode 100644 index 0000000..2b0f208 --- /dev/null +++ b/internal/diagram/testdata/sq-divider.svg @@ -0,0 +1 @@ +ABbeforeaftersection break \ No newline at end of file diff --git a/internal/diagram/testdata/sq-loop-opt.svg b/internal/diagram/testdata/sq-loop-opt.svg new file mode 100644 index 0000000..f4d000f --- /dev/null +++ b/internal/diagram/testdata/sq-loop-opt.svg @@ -0,0 +1 @@ +ABticktockloop each roundopt maybe \ No newline at end of file diff --git a/internal/diagram/testdata/sq-nested.svg b/internal/diagram/testdata/sq-nested.svg new file mode 100644 index 0000000..1e52c98 --- /dev/null +++ b/internal/diagram/testdata/sq-nested.svg @@ -0,0 +1 @@ +ABxyalt outerloop innerelse other \ No newline at end of file diff --git a/internal/diagram/testdata/sq-notes.svg b/internal/diagram/testdata/sq-notes.svg new file mode 100644 index 0000000..4707141 --- /dev/null +++ b/internal/diagram/testdata/sq-notes.svg @@ -0,0 +1 @@ +ABon the lefton the rightalonespanning \ No newline at end of file diff --git a/internal/diagram/testdata/sq-par.svg b/internal/diagram/testdata/sq-par.svg new file mode 100644 index 0000000..6ab4a25 --- /dev/null +++ b/internal/diagram/testdata/sq-par.svg @@ -0,0 +1 @@ +ABCDonetwopar leftand right \ No newline at end of file diff --git a/internal/diagram/testdata/sq-participants.svg b/internal/diagram/testdata/sq-participants.svg new file mode 100644 index 0000000..e660ad9 --- /dev/null +++ b/internal/diagram/testdata/sq-participants.svg @@ -0,0 +1 @@ +AliceBobChiforward \ No newline at end of file diff --git a/internal/diagram/testdata/sq-rect.svg b/internal/diagram/testdata/sq-rect.svg new file mode 100644 index 0000000..b538959 --- /dev/null +++ b/internal/diagram/testdata/sq-rect.svg @@ -0,0 +1 @@ +ABinsidealso inside \ No newline at end of file diff --git a/internal/diagram/testdata/sq-self.svg b/internal/diagram/testdata/sq-self.svg new file mode 100644 index 0000000..86ccefb --- /dev/null +++ b/internal/diagram/testdata/sq-self.svg @@ -0,0 +1 @@ +Athinkrethink \ No newline at end of file diff --git a/internal/markdown/autolink.go b/internal/markdown/autolink.go new file mode 100644 index 0000000..a4f170b --- /dev/null +++ b/internal/markdown/autolink.go @@ -0,0 +1,222 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package markdown + +import "bytes" + +// The GFM extended autolinks: bare www., http://, https:// and ftp:// +// addresses, and bare email addresses, recognised without angle brackets. +// The rules follow the GFM specification literally. + +// isAutolinkStartByte reports whether the previous character allows an +// extended www or URL autolink to begin here: beginning of the inline +// source, whitespace, or one of the delimiting characters. +func isAutolinkStartByte(prev byte) bool { + return prev == 0 || isSpaceTab(prev) || prev == '\n' || + prev == '*' || prev == '_' || prev == '~' || prev == '(' +} + +// isEmailLocalByte reports whether c may appear in the local part of an +// extended email autolink. +func isEmailLocalByte(c byte) bool { + return isAlnum(c) || c == '.' || c == '-' || c == '_' || c == '+' +} + +// isEmailDomainByte reports whether c may appear in a domain segment of an +// extended email autolink. +func isEmailDomainByte(c byte) bool { + return isAlnum(c) || c == '-' || c == '_' +} + +// isEmailPrevByte reports whether the character before a candidate would +// continue a longer email address, which disqualifies the candidate: the +// local part must be maximal. +func isEmailPrevByte(prev byte) bool { + return isAlnum(prev) || prev == '.' || prev == '-' || prev == '_' || + prev == '+' || prev == '@' +} + +// scanExtendedAutolink recognises an extended autolink at i, returning +// the length, the link text and the destination. The prev byte is the +// character before i; a www or URL candidate may only begin where the +// GFM spec allows it, and an email candidate must start a maximal local +// part. +func scanExtendedAutolink(src []byte, i int, prev byte) (n int, text, dest string, ok bool) { + if isAutolinkStartByte(prev) { + if n, end, ok := scanWWWOrURL(src, i); ok { + text := string(src[i:end]) + dest := text + if src[i] == 'w' { + dest = "http://" + text + } + return n, text, dest, true + } + } + if !isEmailPrevByte(prev) { + if n, end, ok := scanEmailAutolink(src, i); ok { + addr := string(src[i:end]) + return n, addr, "mailto:" + addr, true + } + } + return 0, "", "", false +} + +// scanWWWOrURL recognises a bare www address or one with an explicit +// http, https or ftp scheme. It returns the length and the end offset. +func scanWWWOrURL(src []byte, i int) (int, int, bool) { + var schemeLen int + switch { + case bytes.HasPrefix(src[i:], []byte("www.")): + schemeLen = 4 + case bytes.HasPrefix(src[i:], []byte("http://")): + schemeLen = 7 + case bytes.HasPrefix(src[i:], []byte("https://")): + schemeLen = 8 + case bytes.HasPrefix(src[i:], []byte("ftp://")): + schemeLen = 6 + default: + return 0, 0, false + } + domainEnd, ok := scanValidDomain(src, i+schemeLen) + if !ok { + return 0, 0, false + } + // Zero or more non-space, non-< characters follow the domain. + end := domainEnd + for end < len(src) && src[end] != ' ' && src[end] != '\t' && + src[end] != '\n' && src[end] != '<' { + end++ + } + end = validateAutolinkPath(src, i, end) + if end <= i+schemeLen { + return 0, 0, false + } + return end - i, end, true +} + +// scanValidDomain reads a GFM valid domain at i: one or more segments of +// alphanumerics, underscores and hyphens separated by periods, with at +// least one period, and no underscore in the last two segments. It +// returns the offset after the domain. +func scanValidDomain(src []byte, i int) (int, bool) { + var starts, ends []int + j := i + for { + start := j + for j < len(src) && (isAlnum(src[j]) || src[j] == '-' || src[j] == '_') { + j++ + } + if j == start { + return 0, false + } + starts = append(starts, start) + ends = append(ends, j) + // A period only continues the domain when a segment follows it; + // a trailing period belongs to whatever comes after. + if j+1 < len(src) && src[j] == '.' && + (isAlnum(src[j+1]) || src[j+1] == '-' || src[j+1] == '_') { + j++ + continue + } + break + } + if len(ends) < 2 { + return 0, false + } + for k := len(ends) - 2; k < len(ends); k++ { + for c := starts[k]; c < ends[k]; c++ { + if src[c] == '_' { + return 0, false + } + } + } + return j, true +} + +// validateAutolinkPath applies the extended autolink path validation to +// the candidate src[begin:end]: trailing punctuation is trimmed, an +// unbalanced closing parenthesis is trimmed, and a trailing entity +// reference is excluded. +func validateAutolinkPath(src []byte, begin, end int) int { + for end > begin { + switch src[end-1] { + case '?', '!', '.', ',', ':', '*', '_', '~': + end-- + continue + } + break + } + for end > begin && src[end-1] == ')' { + opens, closes := 0, 0 + for k := begin; k < end; k++ { + switch src[k] { + case '(': + opens++ + case ')': + closes++ + } + } + if closes <= opens { + break + } + end-- + } + if end > begin && src[end-1] == ';' { + if amp := bytes.LastIndexByte(src[begin:end], '&'); amp >= 0 { + entity := src[begin+amp+1 : end-1] + valid := len(entity) > 0 + for _, c := range entity { + if !isAlnum(c) { + valid = false + break + } + } + if valid { + end = begin + amp + } + } + } + return end +} + +// scanEmailAutolink recognises an extended email autolink at i: a local +// part of alphanumerics and .-_+, an @, and a domain of alphanumerics, +// hyphens and underscores separated by at least one period, whose last +// character is not a hyphen or underscore. A trailing period is not part +// of the address. +func scanEmailAutolink(src []byte, i int) (int, int, bool) { + j := i + for j < len(src) && isEmailLocalByte(src[j]) { + j++ + } + if j == i || j >= len(src) || src[j] != '@' { + return 0, 0, false + } + j++ + segments := 0 + for { + start := j + for j < len(src) && isEmailDomainByte(src[j]) { + j++ + } + if j == start { + return 0, 0, false + } + segments++ + // A period only continues the domain when a segment follows it; + // a trailing period is not part of the address. + if j+1 < len(src) && src[j] == '.' && isEmailDomainByte(src[j+1]) { + j++ + continue + } + break + } + if segments < 2 { + return 0, 0, false + } + if last := src[j-1]; last == '-' || last == '_' { + return 0, 0, false + } + return j - i, j, true +} diff --git a/internal/markdown/corpus_test.go b/internal/markdown/corpus_test.go new file mode 100644 index 0000000..2ba95ab --- /dev/null +++ b/internal/markdown/corpus_test.go @@ -0,0 +1,165 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package markdown + +import "testing" + +// blockCorpus is the project's own hand-written corpus. Every case states +// an input and the exact HTML the engine produces for it. +var blockCorpus = []struct { + name string + input string + want string +}{ + // ATX headings. + {"atx level 1", "# foo", "

foo

\n"}, + {"atx level 6", "###### foo", "
foo
\n"}, + {"atx seven hashes", "####### foo", "

####### foo

\n"}, + {"atx needs a space", "#foo", "

#foo

\n"}, + {"atx closing sequence", "### foo ###", "

foo

\n"}, + {"atx closing sequence mid content", "### foo ### ###", "

foo ###

\n"}, + {"atx hash without space stays", "# foo#", "

foo#

\n"}, + {"atx indented three", " # foo", "

foo

\n"}, + {"atx indented four is code", " # foo", "
# foo\n
\n"}, + {"atx bare hash", "#", "

\n"}, + {"atx empty content", "###", "

\n"}, + + // Setext headings. + {"setext level 1", "Foo\n===", "

Foo

\n"}, + {"setext level 2", "Foo\n---", "

Foo

\n"}, + {"setext multi line", "Foo\nBar\n---", "

Foo\nBar

\n"}, + {"setext takes whole paragraph", "Foo\nbar\n---", "

Foo\nbar

\n"}, + {"setext indented underline", "Foo\n ===", "

Foo

\n"}, + {"setext dashes with spaces stay thematic", "Foo\n- - -", "

Foo

\n
\n"}, + {"setext cannot be lazy", "> Foo\n===", "
\n

Foo\n===

\n
\n"}, + {"setext inside quote", "> foo\n> ---", "
\n

foo

\n
\n"}, + + // Thematic breaks. + {"thematic stars", "***", "
\n"}, + {"thematic underscores", "___", "
\n"}, + {"thematic spaced dashes", "- - -", "
\n"}, + {"thematic interrupts paragraph", "foo\n***", "

foo

\n
\n"}, + {"thematic long", "---------------------------------------", "
\n"}, + {"thematic trailing spaces", "*** ", "
\n"}, + {"thematic indented three", " ***", "
\n"}, + + // Indented code blocks. + {"indented code", " code", "
code\n
\n"}, + {"indented code two lines", " code\n more", "
code\nmore\n
\n"}, + {"indented code keeps internal blank", " a\n\n b", "
a\n\nb\n
\n"}, + {"indented code strips trailing blanks", " a\n\n\n", "
a\n
\n"}, + {"tab indents code", "\tcode", "
code\n
\n"}, + {"indented code cannot interrupt paragraph", "foo\n bar", "

foo\nbar

\n"}, + + // Fenced code blocks. + {"fenced backticks", "```\ncode\n```", "
code\n
\n"}, + {"fenced tildes", "~~~\ncode\n~~~", "
code\n
\n"}, + {"fenced info class", "```ruby\nx = 1\n```", "
x = 1\n
\n"}, + {"fenced longer close", "````\n```\n````", "
```\n
\n"}, + {"fenced short close is content", "````\ncode\n```", "
code\n```\n
\n"}, + {"fenced indent stripping", " ```\n foo\nbar\n```", "
foo\nbar\n
\n"}, + {"fenced empty", "```\n```", "
\n"}, + {"fenced unclosed", "```\ncode", "
code\n
\n"}, + {"tilde info may hold backticks", "~~~ js `x`\ncode\n~~~", "
code\n
\n"}, + {"fenced blank lines kept", "```\na\n\nb\n```", "
a\n\nb\n
\n"}, + + // HTML blocks. + {"html type six", "
\nfoo\n
", "
\nfoo\n
\n"}, + {"html comment", "", "\n"}, + {"html processing instruction", "", "\n"}, + {"html declaration ends at bracket", "\nfoo", "\n

foo

\n"}, + {"html cdata", "", "\n"}, + {"html script block", "", "\n"}, + {"html blank ends six", "
\nfoo\n\nbar", "
\nfoo\n

bar

\n"}, + {"html six interrupts paragraph", "Foo\n
", "

Foo

\n
\n"}, + {"html textarea", "", "\n"}, + {"html seven complete tag", "\nfoo", "\nfoo\n"}, + + // Block quotes. + {"quote basic", "> foo", "
\n

foo

\n
\n"}, + {"quote two lines", "> foo\n> bar", "
\n

foo\nbar

\n
\n"}, + {"quote without space", ">foo", "
\n

foo

\n
\n"}, + {"quote lazy", "> foo\nbar", "
\n

foo\nbar

\n
\n"}, + {"quote blank closes paragraph", "> foo\n\nbar", "
\n

foo

\n
\n

bar

\n"}, + {"quote empty", ">", "
\n
\n"}, + {"quote nested", "> > foo", "
\n
\n

foo

\n
\n
\n"}, + {"quote heading", "> # foo", "
\n

foo

\n
\n"}, + {"quote list", "> - foo", "
\n
    \n
  • foo
  • \n
\n
\n"}, + {"quote thematic interrupt", "> foo\n---", "
\n

foo

\n
\n
\n"}, + {"quote blank line inside", "> foo\n>\n> bar", "
\n

foo

\n

bar

\n
\n"}, + {"quote blank line between", "> a\n\n> b", "
\n

a

\n
\n
\n

b

\n
\n"}, + {"quote tab content", ">\tfoo", "
\n

foo

\n
\n"}, + + // Lists. + {"list bullet", "- foo", "
    \n
  • foo
  • \n
\n"}, + {"list star", "* foo", "
    \n
  • foo
  • \n
\n"}, + {"list plus", "+ foo", "
    \n
  • foo
  • \n
\n"}, + {"list tight items", "- foo\n- bar", "
    \n
  • foo
  • \n
  • bar
  • \n
\n"}, + {"list loose items", "- foo\n\n- bar", "
    \n
  • \n

    foo

    \n
  • \n
  • \n

    bar

    \n
  • \n
\n"}, + {"list blank splits items loose", "- foo\n- bar\n\n- baz", "
    \n
  • \n

    foo

    \n
  • \n
  • \n

    bar

    \n
  • \n
  • \n

    baz

    \n
  • \n
\n"}, + {"list bullet change splits", "* foo\n+ bar", "
    \n
  • foo
  • \n
\n
    \n
  • bar
  • \n
\n"}, + {"ordered list", "1. foo\n2. bar", "
    \n
  1. foo
  2. \n
  3. bar
  4. \n
\n"}, + {"ordered start", "3. foo", "
    \n
  1. foo
  2. \n
\n"}, + {"ordered paren delimiter", "1) foo", "
    \n
  1. foo
  2. \n
\n"}, + {"ordered delimiter change splits", "1. foo\n1) bar", "
    \n
  1. foo
  2. \n
\n
    \n
  1. bar
  2. \n
\n"}, + {"list item two paragraphs loose", "- foo\n\n bar", "
    \n
  • \n

    foo

    \n

    bar

    \n
  • \n
\n"}, + {"list nested", "- foo\n - bar", "
    \n
  • foo\n
      \n
    • bar
    • \n
    \n
  • \n
\n"}, + {"list item continuation", "- foo\n bar", "
    \n
  • foo\nbar
  • \n
\n"}, + {"list item lazy", "- foo\nbar", "
    \n
  • foo\nbar
  • \n
\n"}, + {"list lazy carries on", "- foo\n bar\ncar", "
    \n
  • foo\nbar\ncar
  • \n
\n"}, + {"ordered nine digits", "123456789. foo", "
    \n
  1. foo
  2. \n
\n"}, + {"ordered ten digits not a list", "1234567890. foo", "

1234567890. foo

\n"}, + {"empty items", "- foo\n-\n- bar", "
    \n
  • foo
  • \n
  • \n
  • bar
  • \n
\n"}, + {"list interrupts paragraph", "foo\n- bar", "

foo

\n
    \n
  • bar
  • \n
\n"}, + {"ordered two cannot interrupt", "foo\n2. bar", "

foo\n2. bar

\n"}, + {"single dash setext under paragraph", "foo\n-", "

foo

\n"}, + {"bullet blank cannot interrupt", "foo\n+", "

foo\n+

\n"}, + {"list marker five spaces makes code", "- indented code", "
    \n
  • \n
    indented code\n
    \n
  • \n
\n"}, + {"item fence on marker line", "- ```\n foo\n ```", "
    \n
  • \n
    foo\n
    \n
  • \n
\n"}, + {"item fence after paragraph", "- foo\n ```\n bar\n ```", "
    \n
  • foo\n
    bar\n
    \n
  • \n
\n"}, + {"item paragraph indented content", "1. foo\n bar", "
    \n
  1. foo\nbar
  2. \n
\n"}, + {"ordered paragraph two lines", "1. A paragraph\n with two lines.", "
    \n
  1. A paragraph\nwith two lines.
  2. \n
\n"}, + {"second item less indented", "- a\n - b", "
    \n
  • a
  • \n
  • b
  • \n
\n"}, + {"lazy into quoted item", "> - foo\nbar", "
\n
    \n
  • foo\nbar
  • \n
\n
\n"}, + {"loose ordered with blank", "1. a\n\n 2. b", "
    \n
  1. \n

    a

    \n
  2. \n
  3. \n

    b

    \n
  4. \n
\n"}, + {"list item second paragraph after blank", "- a\n- b\n\n c", "
    \n
  • \n

    a

    \n
  • \n
  • \n

    b

    \n

    c

    \n
  • \n
\n"}, + + // Paragraphs. + {"paragraph single", "foo", "

foo

\n"}, + {"paragraph lines", "foo\nbar", "

foo\nbar

\n"}, + {"paragraph leading spaces", " foo", "

foo

\n"}, + {"paragraph escaping", "a < b & c", "

a < b & c

\n"}, + {"paragraph quotes escape", "say \"hi\"", "

say "hi"

\n"}, + {"blank lines separate", "foo\n\nbar", "

foo

\n

bar

\n"}, + {"leading blanks ignored", "\n\nfoo", "

foo

\n"}, + {"crlf normalised", "foo\r\nbar\r\n", "

foo\nbar

\n"}, + + // Link reference definitions. + {"reference definition alone", "[foo]: /url", ""}, + {"reference definition with title", "[foo]: /url \"title\"", ""}, + {"reference definition then text", "[foo]: /url\nused", "

used

\n"}, + {"reference definition on two lines", "[foo]:\n/url", ""}, + {"reference definition on three lines", "[foo]:\n/url\n\"title\"", ""}, + {"junk after destination", "[foo]: /url junk", "

[foo]: /url junk

\n"}, + {"two definitions", "[foo]: /a\n[bar]: /b", ""}, + {"definition indented", " [foo]: /url", ""}, + + // Documents. + {"empty input", "", ""}, + {"only blanks", "\n\n\n", ""}, + {"composite document", "# Title\n\nIntro text here.\n\n- one\n- two\n\n```go\nfmt.Println(1)\n```\n\n> quoted\n", + "

Title

\n

Intro text here.

\n
    \n
  • one
  • \n
  • two
  • \n
\n" + + "
fmt.Println(1)\n
\n
\n

quoted

\n
\n"}, +} + +func TestBlockCorpus(t *testing.T) { + for _, tc := range blockCorpus { + t.Run(tc.name, func(t *testing.T) { + got := string(RenderHTML([]byte(tc.input))) + if got != tc.want { + t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want) + } + }) + } +} diff --git a/internal/markdown/extension_corpus_test.go b/internal/markdown/extension_corpus_test.go new file mode 100644 index 0000000..cf478c4 --- /dev/null +++ b/internal/markdown/extension_corpus_test.go @@ -0,0 +1,93 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package markdown + +import "testing" + +func TestFootnotes(t *testing.T) { + cases := []struct { + name string + input string + want string + }{ + {"basic footnote", "Here is a reference.[^1]\n\n[^1]: Here is the note.", + "

Here is a reference." + + `1` + + "

\n" + + "
\n
    \n" + + "
  1. \n" + + "

    Here is the note." + + ` ` + "↩" + `` + + "

    \n" + + "
  2. \n
\n
\n"}, + {"unreferenced definition renders nothing", "[^1]: the note", ""}, + {"undefined reference stays text", "Text[^missing]", "

Text[^missing]

\n"}, + {"numbered by reference order", "[^b] and [^a]\n\n[^a]: A\n[^b]: B", + "

" + + `1` + " and " + + `2` + + "

\n" + + "
\n
    \n" + + "
  1. \n

    B" + + ` ` + "↩" + `` + + "

    \n
  2. \n" + + "
  3. \n

    A" + + ` ` + "↩" + `` + + "

    \n
  4. \n" + + "
\n
\n"}, + {"repeated reference ids", "[^a] again [^a]\n\n[^a]: A", + "

" + + `1` + " again " + + `1` + + "

\n" + + "
\n
    \n" + + "
  1. \n

    A" + + ` ` + "↩" + `` + + "

    \n
  2. \n
\n
\n"}, + {"multi block definition", "Ref.[^1]\n\n[^1]: first\n\n second", + "

Ref." + + `1` + + "

\n" + + "
\n
    \n" + + "
  1. \n

    first

    \n

    second" + + ` ` + "↩" + `` + + "

    \n
  2. \n
\n
\n"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + got := string(RenderHTML([]byte(tc.input))) + if got != tc.want { + t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want) + } + }) + } +} + +func TestDefinitionLists(t *testing.T) { + cases := []struct { + name string + input string + want string + }{ + {"definition basic", "Term\n: Definition", "
\n
Term
\n
Definition
\n
\n"}, + {"definition two entries", "Term 1\n: Def 1\n\nTerm 2\n: Def 2", + "
\n
Term 1
\n
\n

Def 1

\n
\n
Term 2
\n
\n

Def 2

\n
\n
\n"}, + {"definition two definitions", "Term\n: Def a\n: Def b", + "
\n
Term
\n
Def a
\n
Def b
\n
\n"}, + {"definition multiline terms", "Term 1\nTerm 2\n: Def", + "
\n
Term 1
\n
Term 2
\n
Def
\n
\n"}, + {"definition second paragraph", "Term\n: Def\n\n more", + "
\n
Term
\n
\n

Def

\n

more

\n
\n
\n"}, + {"definition inline content", "Term\n: A *bold* claim", + "
\n
Term
\n
A bold claim
\n
\n"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + got := string(RenderHTML([]byte(tc.input))) + if got != tc.want { + t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want) + } + }) + } +} diff --git a/internal/markdown/gfm_corpus_test.go b/internal/markdown/gfm_corpus_test.go new file mode 100644 index 0000000..4905d44 --- /dev/null +++ b/internal/markdown/gfm_corpus_test.go @@ -0,0 +1,75 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package markdown + +import "testing" + +// gfmCorpus holds the cases of the GitHub Flavored Markdown extensions: +// tables, strikethrough and task lists. +var gfmCorpus = []struct { + name string + input string + want string +}{ + // Tables, in the multi-line shape the GFM specification prints. + {"table basic", "| foo | bar |\n| --- | --- |\n| baz | bim |", + "\n\n\n\n\n\n\n\n\n\n\n\n\n
foobar
bazbim
\n"}, + {"table without boundary pipes", "foo | bar\n--- | ---\nbaz | bim", + "\n\n\n\n\n\n\n\n\n\n\n\n\n
foobar
bazbim
\n"}, + {"table alignment", "| a | b | c | d |\n| :- | :-: | -: | - |", + "\n\n\n\n\n\n\n\n\n
abcd
\n"}, + {"table without body", "| abc | def |\n| --- | --- |", + "\n\n\n\n\n\n\n
abcdef
\n"}, + {"table excess cell ignored", "| a | b |\n| --- | --- |\n| c | d | e |", + "\n\n\n\n\n\n\n\n\n\n\n\n\n
ab
cd
\n"}, + {"table missing cell empty", "| a | b |\n| --- | --- |\n| c |", + "\n\n\n\n\n\n\n\n\n\n\n\n\n
ab
c
\n"}, + {"table count mismatch stays paragraph", "| a | b |\n| --- |\n| c |", + "

| a | b |\n| --- |\n| c |

\n"}, + {"table escaped pipe", "| a \\| b | c |\n| --- | --- |", + "\n\n\n\n\n\n\n
a | bc
\n"}, + {"table cells hold inline", "| *a* | `b` |\n| --- | --- |", + "\n\n\n\n\n\n\n
ab
\n"}, + {"table code span escaped pipe", "| f\\|oo |\n| --- |\n| b `\\|` az |", + "\n\n\n\n\n\n\n\n\n\n\n
f|oo
b | az
\n"}, + {"table breaks at blank", "| a |\n| --- |\n| b |\n\nparagraph", + "\n\n\n\n\n\n\n\n\n\n\n
a
b
\n

paragraph

\n"}, + {"table interrupted by heading", "| a |\n| --- |\n| b |\n# h", + "\n\n\n\n\n\n\n\n\n\n\n
a
b
\n

h

\n"}, + {"table interrupts paragraph", "a | b\n--- | ---", + "\n\n\n\n\n\n\n
ab
\n"}, + {"table splits paragraph", "foo\na | b\n--- | ---", + "

foo

\n\n\n\n\n\n\n\n
ab
\n"}, + {"setext without pipes stays", "abc\n---", "

abc

\n"}, + + // Strikethrough. + {"strikethrough double tilde", "~~foo~~", "

foo

\n"}, + {"strikethrough single tilde", "~foo~", "

foo

\n"}, + {"strikethrough unmatched pair", "~~foo~", "

~foo

\n"}, + {"strikethrough intraword", "foo~~bar~~baz", "

foobarbaz

\n"}, + {"tilde spaced stays literal", "a ~ b", "

a ~ b

\n"}, + + // Task lists. + {"task checked", "- [x] foo", "
    \n
  • foo
  • \n
\n"}, + {"task unchecked", "- [ ] foo", "
    \n
  • foo
  • \n
\n"}, + {"task uppercase x", "- [X] foo", "
    \n
  • foo
  • \n
\n"}, + {"task needs a space", "- [x]foo", "
    \n
  • [x]foo
  • \n
\n"}, + {"task list of two", "- [x] foo\n- [ ] bar", + "
    \n
  • foo
  • \n
  • bar
  • \n
\n"}, + {"task nested", "- [x] foo\n - [ ] bar", + "
    \n
  • foo\n
      \n
    • bar
    • \n
    \n
  • \n
\n"}, + {"task in loose list", "- [x] foo\n\n- [ ] bar", + "
    \n
  • \n

    foo

    \n
  • \n
  • \n

    bar

    \n
  • \n
\n"}, +} + +func TestGFMCorpus(t *testing.T) { + for _, tc := range gfmCorpus { + t.Run(tc.name, func(t *testing.T) { + got := string(RenderHTML([]byte(tc.input))) + if got != tc.want { + t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want) + } + }) + } +} diff --git a/internal/markdown/inline.go b/internal/markdown/inline.go new file mode 100644 index 0000000..1b6e5a6 --- /dev/null +++ b/internal/markdown/inline.go @@ -0,0 +1,742 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package markdown + +import ( + "bytes" + "unicode" + "unicode/utf8" +) + +type inlineKind uint8 + +const ( + inlText inlineKind = iota + inlCode + inlRawHTML + inlEmph + inlStrong + inlStrikethrough + inlLink + inlImage + inlBreak + inlFootnoteRef +) + +// inline is one node of the inline tree. While parsing, the top level +// nodes form a doubly-linked chain; when emphasis or a link takes a range, +// the range becomes the children slice of the wrapping node. +type inline struct { + kind inlineKind + literal string // text, code content, raw HTML, soft break + dest string + title string + hasTitle bool + num int // footnote reference: ordinal + occurrence int // footnote reference: which reference to that ordinal + children []*inline + prev *inline + next *inline +} + +// footnoteTracker numbers footnote references in document order: a +// definition receives its ordinal at its first reference, and later +// references to the same definition count their occurrences. +type footnoteTracker struct { + defs map[string]*Node + ordinals map[string]int + seen map[string]int + order []*Node +} + +func newFootnoteTracker(defs []*Node) *footnoteTracker { + m := make(map[string]*Node, len(defs)) + for _, d := range defs { + if _, ok := m[d.label]; !ok { + m[d.label] = d + } + } + return &footnoteTracker{defs: m, ordinals: map[string]int{}, seen: map[string]int{}} +} + +// reference records one reference to the labelled footnote. +func (t *footnoteTracker) reference(label string) (ordinal, occurrence int, ok bool) { + if _, defined := t.defs[label]; !defined { + return 0, 0, false + } + t.seen[label]++ + if t.seen[label] == 1 { + t.ordinals[label] = len(t.order) + 1 + t.order = append(t.order, t.defs[label]) + } + return t.ordinals[label], t.seen[label], true +} + +// delimiter is one entry of the delimiter stack: a run of asterisks or +// underscores waiting to be matched, or an open link or image bracket. +type delimiter struct { + node *inline + char byte + isBracket bool + image bool + active bool + length int + origLen int + canOpen bool + canClose bool + srcPos int // bracket: index just after the opening literal + prev *delimiter + next *delimiter +} + +// parseInlines parses the inline content of a paragraph or heading against +// the document's link reference definitions and footnotes. +func parseInlines(content []byte, refs map[string]reference, footnotes *footnoteTracker) []*inline { + p := &inlineParser{src: bytes.TrimRight(content, " \t"), refs: refs, footnotes: footnotes} + p.parse() + return p.chain() +} + +type inlineParser struct { + src []byte + pos int + refs map[string]reference + footnotes *footnoteTracker + + first *inline + last *inline + firstDelim *delimiter + lastDelim *delimiter +} + +func (p *inlineParser) chain() []*inline { + var nodes []*inline + for n := p.first; n != nil; n = n.next { + nodes = append(nodes, n) + } + return nodes +} + +func (p *inlineParser) push(n *inline) *inline { + n.prev = p.last + n.next = nil + if p.last != nil { + p.last.next = n + } else { + p.first = n + } + p.last = n + return n +} + +func (p *inlineParser) removeNode(n *inline) { + if n.prev != nil { + n.prev.next = n.next + } else { + p.first = n.next + } + if n.next != nil { + n.next.prev = n.prev + } else { + p.last = n.prev + } +} + +func (p *inlineParser) pushDelim(d *delimiter) { + d.prev = p.lastDelim + d.next = nil + if p.lastDelim != nil { + p.lastDelim.next = d + } else { + p.firstDelim = d + } + p.lastDelim = d +} + +func (p *inlineParser) removeDelim(d *delimiter) { + if d.prev != nil { + d.prev.next = d.next + } else { + p.firstDelim = d.next + } + if d.next != nil { + d.next.prev = d.prev + } else { + p.lastDelim = d.prev + } +} + +// lastBracket returns the most recent open bracket on the stack. +func (p *inlineParser) lastBracket() *delimiter { + for d := p.lastDelim; d != nil; d = d.prev { + if d.isBracket { + return d + } + } + return nil +} + +func (p *inlineParser) parse() { + for p.pos < len(p.src) { + switch c := p.src[p.pos]; c { + case '\n': + p.handleNewline() + case '\\': + p.handleBackslash() + case '`': + p.handleBackticks() + case '<': + p.handleLessThan() + case '&': + p.handleAmpersand() + case '*', '_', '~': + p.handleDelimiterRun(c) + case '[': + if p.tryFootnoteRef() { + continue + } + p.pushBracket(false) + case '!': + if p.pos+1 < len(p.src) && p.src[p.pos+1] == '[' { + p.pushBracket(true) + } else { + p.push(&inline{kind: inlText, literal: "!"}) + p.pos++ + } + case ']': + p.handleCloseBracket() + default: + p.textRun() + } + } + p.processEmphasis(nil) +} + +// handleNewline ends the line: the whitespace before the line ending is +// stripped, and two or more spaces make the break hard. +func (p *inlineParser) handleNewline() { + spaces := 0 + if p.last != nil && p.last.kind == inlText { + lit := p.last.literal + end := len(lit) + for end > 0 && isSpaceTab(lit[end-1]) { + if lit[end-1] == ' ' { + spaces++ + } + end-- + } + if end == 0 { + p.removeNode(p.last) + } else { + p.last.literal = lit[:end] + } + } + if spaces >= 2 { + p.push(&inline{kind: inlBreak}) + } else { + p.push(&inline{kind: inlText, literal: "\n"}) + } + p.pos++ +} + +func (p *inlineParser) handleBackslash() { + if p.pos+1 < len(p.src) { + next := p.src[p.pos+1] + switch { + case next == '\n': + p.push(&inline{kind: inlBreak}) + p.pos += 2 + return + case isASCIIPunct(next): + p.push(&inline{kind: inlText, literal: string(next)}) + p.pos += 2 + return + } + } + p.push(&inline{kind: inlText, literal: `\`}) + p.pos++ +} + +// handleBackticks scans for a closing backtick string of equal length and +// emits the code span between them, or the literal opening run. +func (p *inlineParser) handleBackticks() { + openLen := fenceRun(p.src[p.pos:], '`') + i := p.pos + openLen + for i < len(p.src) { + if p.src[i] == '`' { + l := fenceRun(p.src[i:], '`') + if l == openLen { + p.push(&inline{kind: inlCode, literal: codeSpanContent(p.src[p.pos+openLen : i])}) + p.pos = i + l + return + } + i += l + continue + } + i++ + } + p.push(&inline{kind: inlText, literal: string(p.src[p.pos : p.pos+openLen])}) + p.pos += openLen +} + +// codeSpanContent converts line endings to spaces and strips the one-space +// margin a span carries at both ends when it is not all spaces. +func codeSpanContent(c []byte) string { + c = bytes.ReplaceAll(c, []byte("\n"), []byte(" ")) + if len(c) >= 2 && c[0] == ' ' && c[len(c)-1] == ' ' { + allSpaces := true + for _, b := range c { + if b != ' ' { + allSpaces = false + break + } + } + if !allSpaces { + c = c[1 : len(c)-1] + } + } + return string(c) +} + +func (p *inlineParser) handleLessThan() { + if text, dest, n, ok := scanAutolink(p.src[p.pos:]); ok { + p.push(&inline{kind: inlLink, dest: dest, children: []*inline{{kind: inlText, literal: text}}}) + p.pos += n + return + } + if n := scanRawHTML(p.src[p.pos:]); n > 0 { + p.push(&inline{kind: inlRawHTML, literal: string(p.src[p.pos : p.pos+n])}) + p.pos += n + return + } + p.push(&inline{kind: inlText, literal: "<"}) + p.pos++ +} + +func (p *inlineParser) handleAmpersand() { + if s, n, ok := scanEntity(p.src, p.pos); ok { + p.push(&inline{kind: inlText, literal: s}) + p.pos += n + return + } + p.push(&inline{kind: inlText, literal: "&"}) + p.pos++ +} + +// handleDelimiterRun records a run of asterisks or underscores and whether +// it may open or close emphasis under the flanking rules. +func (p *inlineParser) handleDelimiterRun(char byte) { + start := p.pos + end := start + fenceRun(p.src[start:], char) + beforeWS, beforePunct := classifyRune(runeBefore(p.src, start)) + afterWS, afterPunct := classifyRune(runeAfter(p.src, end)) + left := !afterWS && (!afterPunct || beforeWS || beforePunct) + right := !beforeWS && (!beforePunct || afterWS || afterPunct) + d := &delimiter{char: char, length: end - start, origLen: end - start} + if char == '_' { + d.canOpen = left && (!right || beforePunct) + d.canClose = right && (!left || afterPunct) + } else { + // asterisks and tildes flank the same way + d.canOpen, d.canClose = left, right + } + d.node = p.push(&inline{kind: inlText, literal: string(p.src[start:end])}) + p.pushDelim(d) + p.pos = end +} + +// tryFootnoteRef consumes a reference to a defined footnote and emits its +// marker. A reference to an undefined footnote stays bracket text. +func (p *inlineParser) tryFootnoteRef() bool { + if p.footnotes == nil { + return false + } + label, n, ok := scanFootnoteLabel(p.src[p.pos:]) + if !ok { + return false + } + num, occurrence, ok := p.footnotes.reference(normaliseLabel(label)) + if !ok { + return false + } + p.push(&inline{kind: inlFootnoteRef, num: num, occurrence: occurrence}) + p.pos += n + return true +} + +func (p *inlineParser) pushBracket(image bool) { + lit := "[" + if image { + lit = "![" + } + n := p.push(&inline{kind: inlText, literal: lit}) + p.pushDelim(&delimiter{ + node: n, char: '[', isBracket: true, image: image, active: true, + srcPos: p.pos + len(lit), + }) + p.pos += len(lit) +} + +// textRun consumes the run of ordinary characters up to the next special +// one, emitting extended autolinks and plain text in the order they come. +func (p *inlineParser) textRun() { + start := p.pos + for p.pos < len(p.src) { + c := p.src[p.pos] + if isInlineSpecial(c) { + break + } + var prev byte + if p.pos > 0 { + prev = p.src[p.pos-1] + } + if c == 'w' || c == 'h' || c == 'f' || + (!isEmailPrevByte(prev) && isEmailLocalByte(c)) { + if n, text, dest, ok := scanExtendedAutolink(p.src, p.pos, prev); ok { + if p.pos > start { + p.push(&inline{kind: inlText, literal: string(p.src[start:p.pos])}) + } + p.push(&inline{ + kind: inlLink, + dest: dest, + children: []*inline{{kind: inlText, literal: text}}, + }) + p.pos += n + start = p.pos + continue + } + } + p.pos++ + } + if p.pos > start { + p.push(&inline{kind: inlText, literal: string(p.src[start:p.pos])}) + } +} + +func isInlineSpecial(c byte) bool { + switch c { + case '\n', '\\', '`', '<', '&', '*', '_', '~', '[', ']', '!': + return true + } + return false +} + +// processEmphasis matches delimiter runs between the stack bottom and the +// top into emphasis and strong nodes, following the reference algorithm: +// closers walk forward, openers are searched backwards, a matching pair +// may not both be intraword delimiters whose run lengths add up against +// the rule of three, and the matched run lengths shrink from the inner +// sides. +func (p *inlineParser) processEmphasis(bottom *delimiter) { + var closer *delimiter + if bottom == nil { + closer = p.firstDelim + } else { + closer = bottom.next + } + for closer != nil { + if closer.isBracket || !closer.canClose { + closer = closer.next + continue + } + opener, found := p.findOpener(closer, bottom) + if !found { + if !closer.canOpen { + p.removeDelim(closer) + } + closer = closer.next + continue + } + use := 1 + kind := inlEmph + switch closer.char { + case '~': + if closer.length >= 2 && opener.length >= 2 { + use = 2 + } + kind = inlStrikethrough + default: + if closer.length >= 2 && opener.length >= 2 { + use = 2 + kind = inlStrong + } + } + var children []*inline + for n := opener.node.next; n != closer.node; n = n.next { + children = append(children, n) + } + node := &inline{kind: kind, children: children} + opener.node.next = node + node.prev = opener.node + node.next = closer.node + closer.node.prev = node + opener.node.literal = opener.node.literal[:len(opener.node.literal)-use] + closer.node.literal = closer.node.literal[use:] + opener.length -= use + closer.length -= use + for d := closer.prev; d != nil && d != opener; { + prev := d.prev + p.removeDelim(d) + d = prev + } + if opener.length == 0 { + p.removeNode(opener.node) + p.removeDelim(opener) + } + if closer.length == 0 { + next := closer.next + p.removeNode(closer.node) + p.removeDelim(closer) + closer = next + } + } + for p.lastDelim != nil && p.lastDelim != bottom { + p.removeDelim(p.lastDelim) + } +} + +// findOpener searches backwards from the closer for a run of the same +// character that may open, honouring the rule of three. +func (p *inlineParser) findOpener(closer, bottom *delimiter) (*delimiter, bool) { + for opener := closer.prev; opener != nil && opener != bottom; opener = opener.prev { + if opener.isBracket || opener.char != closer.char || !opener.canOpen { + continue + } + oddMatch := closer.char != '~' && + (closer.canOpen || opener.canClose) && + (opener.origLen+closer.origLen)%3 == 0 && + !(opener.origLen%3 == 0 && closer.origLen%3 == 0) + if !oddMatch { + return opener, true + } + } + return nil, false +} + +// handleCloseBracket tries to close the most recent open bracket as an +// inline link or image, as a reference link with an explicit, collapsed or +// empty label, and leaves the bracket as literal text otherwise. +func (p *inlineParser) handleCloseBracket() { + opener := p.lastBracket() + if opener == nil { + p.push(&inline{kind: inlText, literal: "]"}) + p.pos++ + return + } + if !opener.active { + p.removeDelim(opener) + p.push(&inline{kind: inlText, literal: "]"}) + p.pos++ + return + } + closerIdx := p.pos + p.pos++ + + var dest, title string + var hasTitle bool + matched := false + + if p.pos < len(p.src) && p.src[p.pos] == '(' { + save := p.pos + p.pos++ + if d, t, ht, ok := p.scanInlineSpec(); ok { + dest, title, hasTitle, matched = d, t, ht, true + } else { + p.pos = save + } + } + if !matched { + save := p.pos + label, ok := p.referenceLabel(opener, closerIdx) + if ok && len(bytes.TrimSpace(label)) > 0 && validLabel(label) { + if ref, exists := p.refs[normaliseLabel(string(label))]; exists { + dest, title, hasTitle, matched = ref.destination, ref.title, ref.hasTitle, true + } + } + if !matched { + p.pos = save + } + } + if !matched { + p.removeDelim(opener) + p.push(&inline{kind: inlText, literal: "]"}) + return + } + + p.processEmphasis(opener) + var children []*inline + for n := opener.node.next; n != nil; n = n.next { + children = append(children, n) + } + kind := inlLink + if opener.image { + kind = inlImage + } + node := &inline{kind: kind, dest: dest, title: title, hasTitle: hasTitle, children: children} + if opener.node.prev != nil { + opener.node.prev.next = node + } else { + p.first = node + } + node.prev = opener.node.prev + node.next = nil + p.last = node + p.removeDelim(opener) + if !opener.image { + // Links may not nest in links; image brackets stay open. + for d := opener.prev; d != nil; d = d.prev { + if d.isBracket && !d.image { + d.active = false + } + } + } +} + +// referenceLabel reads the label of a reference link after the closing +// bracket: an explicit label in brackets, the collapsed empty brackets, or +// the shortcut label taken from the link text itself. +func (p *inlineParser) referenceLabel(opener *delimiter, closerIdx int) ([]byte, bool) { + if p.pos < len(p.src) && p.src[p.pos] == '[' { + if p.pos+1 < len(p.src) && p.src[p.pos+1] == ']' { + p.pos += 2 + return p.src[opener.srcPos:closerIdx], true + } + j := p.pos + 1 + for j < len(p.src) { + if p.src[j] == '\\' && j+1 < len(p.src) { + j += 2 + continue + } + if p.src[j] == ']' { + break + } + j++ + } + if j < len(p.src) { + label := p.src[p.pos+1 : j] + p.pos = j + 1 + return label, true + } + return nil, false + } + return p.src[opener.srcPos:closerIdx], true +} + +// scanInlineSpec parses the destination and optional title of an inline +// link, starting after the opening parenthesis. +func (p *inlineParser) scanInlineSpec() (string, string, bool, bool) { + p.skipWhitespace() + if p.pos < len(p.src) && p.src[p.pos] == ')' { + p.pos++ + return "", "", false, true + } + d, n, ok := scanDestination(p.src[p.pos:]) + if !ok { + return "", "", false, false + } + dest := unescapeText(string(d)) + p.pos += n + p.skipWhitespace() + if p.pos < len(p.src) { + switch c := p.src[p.pos]; c { + case '"', '\'', '(': + title, ok := p.scanInlineTitle(c) + if !ok { + return "", "", false, false + } + p.skipWhitespace() + if p.pos < len(p.src) && p.src[p.pos] == ')' { + p.pos++ + return dest, unescapeText(title), true, true + } + return "", "", false, false + } + } + if p.pos < len(p.src) && p.src[p.pos] == ')' { + p.pos++ + return dest, "", false, true + } + return "", "", false, false +} + +// scanInlineTitle scans a title through its closing quote, line endings +// included, leaving the position past the closing quote. +func (p *inlineParser) scanInlineTitle(open byte) (string, bool) { + closer := open + if open == '(' { + closer = ')' + } + i := p.pos + 1 + for i < len(p.src) { + c := p.src[i] + if c == '\\' && i+1 < len(p.src) { + i += 2 + continue + } + if c == closer { + title := string(p.src[p.pos+1 : i]) + p.pos = i + 1 + return title, true + } + i++ + } + return "", false +} + +func (p *inlineParser) skipWhitespace() { + for p.pos < len(p.src) { + switch p.src[p.pos] { + case ' ', '\t', '\n': + p.pos++ + default: + return + } + } +} + +// runeBefore returns the rune that ends just before the position, or a +// null rune at the start, which counts as whitespace. +func runeBefore(src []byte, pos int) rune { + if pos == 0 { + return 0 + } + r, _ := utf8.DecodeLastRune(src[:pos]) + return r +} + +// runeAfter returns the rune that starts at the position, or a null rune +// at the end, which counts as whitespace. +func runeAfter(src []byte, pos int) rune { + if pos >= len(src) { + return 0 + } + r, _ := utf8.DecodeRune(src[pos:]) + return r +} + +// classifyRune reports whether the rune is whitespace and whether it is +// punctuation, under the CommonMark definitions: Unicode whitespace, and +// ASCII or Unicode punctuation or symbol characters. +func classifyRune(r rune) (space, punct bool) { + if r == 0 { + return true, false + } + if r < utf8.RuneSelf { + space = r == ' ' || r == '\t' || r == '\n' || r == '\v' || r == '\f' || r == '\r' + punct = isASCIIPunct(byte(r)) + return space, punct + } + return unicode.IsSpace(r), unicode.IsPunct(r) || unicode.IsSymbol(r) +} + +// isASCIIPunct reports whether c is an ASCII punctuation character. +func isASCIIPunct(c byte) bool { + switch c { + case '!', '"', '#', '$', '%', '&', '\'', '(', ')', '*', '+', ',', '-', + '.', '/', ':', ';', '<', '=', '>', '?', '@', '[', '\\', ']', '^', + '_', '`', '{', '|', '}', '~': + return true + } + return false +} diff --git a/internal/markdown/inline_corpus_test.go b/internal/markdown/inline_corpus_test.go new file mode 100644 index 0000000..0a2b457 --- /dev/null +++ b/internal/markdown/inline_corpus_test.go @@ -0,0 +1,147 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package markdown + +import "testing" + +// inlineCorpus is the project's own hand-written corpus for the inline +// layer, on top of a paragraph unless the case states otherwise. +var inlineCorpus = []struct { + name string + input string + want string +}{ + // Emphasis. + {"em asterisk", "*foo*", "

foo

\n"}, + {"strong asterisk", "**foo**", "

foo

\n"}, + {"em and strong", "***foo***", "

foo

\n"}, + {"strong with inner em", "**foo *bar* baz**", "

foo bar baz

\n"}, + {"intraword asterisk", "foo*bar*baz", "

foobarbaz

\n"}, + {"intraword underscore stays", "foo_bar_baz", "

foo_bar_baz

\n"}, + {"em underscore with spaces", "_foo bar_", "

foo bar

\n"}, + {"underscore cannot open after space", "_ foo_", "

_ foo_

\n"}, + {"literal asterisk when spaced", "a * foo *", "

a * foo *

\n"}, + {"escaped asterisk", "\\*not em\\*", "

*not em*

\n"}, + {"escaped backslash then em", "\\\\*foo*", "

\\foo

\n"}, + {"lone closer stays", "a *", "

a *

\n"}, + {"em inside word boundaries", "a*b*c", "

abc

\n"}, + {"em holding strong", "*foo**bar**baz*", "

foobarbaz

\n"}, + {"strong inside em with text", "***foo** bar*", "

foo bar

\n"}, + {"em inside strong at end", "**foo *bar***", "

foo bar

\n"}, + {"unmatched inner run stays", "*foo**bar*", "

foo**bar

\n"}, + {"intraword digits", "5*6*78", "

5678

\n"}, + {"underscore opens before punctuation", "_(bar)_", "

(bar)

\n"}, + {"intraword underscore before punctuation stays", "foo_(bar)_", "

foo_(bar)_

\n"}, + + // Code spans. + {"code span", "`foo`", "

foo

\n"}, + {"code span strips one space margin", "` foo `", "

foo

\n"}, + {"code span keeps margin with doubles", "`` foo ``", "

foo

\n"}, + {"code span double backticks", "``foo ` bar``", "

foo ` bar

\n"}, + {"code span has no escapes", "`foo\\`bar`", "

foo\\bar`

\n"}, + {"code span unmatched", "foo ` bar", "

foo ` bar

\n"}, + {"code span escapes markup", "`*em*`", "

*em*

\n"}, + + // Links and images. + {"inline link", "[foo](/uri)", "

foo

\n"}, + {"inline link with title", "[foo](/uri \"title\")", "

foo

\n"}, + {"inline link single quoted title", "[foo](/uri 'title')", "

foo

\n"}, + {"inline link empty destination", "[foo]()", "

foo

\n"}, + {"angle destination with space", "[foo]()", "

foo

\n"}, + {"trailing paren stays text", "[foo](bar))", "

foo)

\n"}, + {"em inside link", "[*foo*](/uri)", "

foo

\n"}, + {"image with title", "![foo](/url \"title\")", "

\"foo\"

\n"}, + {"image inside link", "[![alt](img)](page)", "

\"alt\"

\n"}, + {"no nested links", "[a [b](x)](y)", "

[a b](y)

\n"}, + {"undefined reference stays", "[foo]", "

[foo]

\n"}, + {"link destination escaped ampersand", "[a](/url?a=1&b=2)", "

a

\n"}, + + // Autolinks. + {"uri autolink", "", "

http://example.com

\n"}, + {"email autolink", "", "

foo@bar.example.com

\n"}, + {"not an autolink", "<3>", "

<3>

\n"}, + + // Extended autolinks (GFM). + {"extended www", "Visit www.commonmark.org for more.", + "

Visit www.commonmark.org for more.

\n"}, + {"extended www with path", "go to www.commonmark.org/help today", + "

go to www.commonmark.org/help today

\n"}, + {"extended trailing punctuation", "see www.example.com.", + "

see www.example.com.

\n"}, + {"extended paren balance", "www.example.com/query?q=(a+b)))", + "

www.example.com/query?q=(a+b)))

\n"}, + {"extended entity suffix", "www.example.com?q=x&hl;", + "

www.example.com?q=x&hl;

\n"}, + {"extended https", "open https://example.com/page", + "

open https://example.com/page

\n"}, + {"extended ftp", "ftp://files.example.org/pub", + "

ftp://files.example.org/pub

\n"}, + {"extended email", "write to a.b-c_d@example.com soon", + "

write to a.b-c_d@example.com soon

\n"}, + {"extended email trailing dot", "mail me at user@example.net.", + "

mail me at user@example.net.

\n"}, + {"plus before at only", "hello@mail+xyz.example is not, but hello+xyz@mail.example is", + "

hello@mail+xyz.example is not, but hello+xyz@mail.example is

\n"}, + {"underscore banned in last segments", "www.un_der_score.org stays text", + "

www.un_der_score.org stays text

\n"}, + {"no autolink mid word", "awww.example.com stays text", + "

awww.example.com stays text

\n"}, + + // Raw inline HTML and entities. + {"raw inline html", "a c d", "

a c d

\n"}, + {"html comment inline", "a b", "

a b

\n"}, + {"entity ampersand", "AT&T", "

AT&T

\n"}, + {"entity numeric", "#", "

#

\n"}, + {"entity hex", """, "

"

\n"}, + {"bare ampersand", "AT&T", "

AT&T

\n"}, + {"not an entity", "&x;", "

&x;

\n"}, + {"less than escaped", "a < b", "

a < b

\n"}, + + // Breaks. + {"soft break", "foo\nbar", "

foo\nbar

\n"}, + {"hard break spaces", "foo \nbar", "

foo
\nbar

\n"}, + {"hard break backslash", "foo\\\nbar", "

foo
\nbar

\n"}, + {"one trailing space is soft", "foo \nbar", "

foo\nbar

\n"}, + {"trailing spaces at end dropped", "foo ", "

foo

\n"}, +} + +func TestInlineCorpus(t *testing.T) { + for _, tc := range inlineCorpus { + t.Run(tc.name, func(t *testing.T) { + got := string(RenderHTML([]byte(tc.input))) + if got != tc.want { + t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want) + } + }) + } +} + +// Reference links need definitions from earlier blocks, so these cases +// carry multi-block inputs. +func TestReferenceLinks(t *testing.T) { + cases := []struct { + name string + input string + want string + }{ + {"explicit reference", "[foo][bar]\n\n[bar]: /url", "

foo

\n"}, + {"collapsed reference", "[foo][]\n\n[foo]: /url", "

foo

\n"}, + {"shortcut reference", "[foo]\n\n[foo]: /url", "

foo

\n"}, + {"reference with title", "[foo]\n\n[foo]: /url \"the title\"", "

foo

\n"}, + {"reference label case folded", "[Foo]\n\n[foo]: /url", "

Foo

\n"}, + {"image reference", "![foo]\n\n[foo]: /url", "

\"foo\"

\n"}, + {"shortcut takes whole text", "[foo *bar*]\n\n[foo *bar*]: /url", "

foo bar

\n"}, + {"inline beats reference", "[foo](/inline)\n\n[foo]: /ref", "

foo

\n"}, + {"link in heading", "# [foo](/uri)", "

foo

\n"}, + {"code span in heading", "## a `b` c", "

a b c

\n"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + got := string(RenderHTML([]byte(tc.input))) + if got != tc.want { + t.Errorf("input %q\ngot: %q\nwant: %q", tc.input, got, tc.want) + } + }) + } +} diff --git a/internal/markdown/markdown.go b/internal/markdown/markdown.go new file mode 100644 index 0000000..5cdd124 --- /dev/null +++ b/internal/markdown/markdown.go @@ -0,0 +1,22 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +// Package markdown renders Markdown to HTML with an engine of its own, +// built on the standard library alone. Parse builds the tree of blocks +// and RenderHTML serialises it; the grammar of the block structure follows +// CommonMark. Inline content is escaped plain text. +package markdown + +// RenderHTML parses source and renders it to HTML. The same input always +// produces byte-identical output. +func RenderHTML(source []byte) []byte { + return RenderHTMLNode(Parse(source)) +} + +// RenderHTMLNode renders a parsed document tree to HTML. +func RenderHTMLNode(doc *Node) []byte { + r := &renderer{refs: doc.refs, footnotes: newFootnoteTracker(doc.footnotes)} + r.blocks(doc.children) + r.footnoteSection() + return r.out +} diff --git a/internal/markdown/node.go b/internal/markdown/node.go new file mode 100644 index 0000000..302c0da --- /dev/null +++ b/internal/markdown/node.go @@ -0,0 +1,116 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package markdown + +// nodeKind identifies a block node of the document tree. +type nodeKind uint8 + +const ( + kindDocument nodeKind = iota + kindParagraph + kindHeading + kindCodeBlock + kindHTMLBlock + kindBlockquote + kindList + kindListItem + kindThematicBreak + kindTable + kindFootnoteDef + kindDefList + kindDefTerm + kindDefItem +) + +// listKind distinguishes bullet lists from ordered lists. +type listKind uint8 + +const ( + bulletList listKind = iota + orderedList +) + +// Column alignment of a table, carried per column. +const ( + alignNone uint8 = iota + alignLeft + alignCentre + alignRight +) + +// Node is one block of the parsed document. The zero value is a document +// root; every other kind is created by the parser. +type Node struct { + kind nodeKind + parent *Node + children []*Node + + // content holds the raw text of a leaf block: paragraph or heading + // inline text, code source, or the lines of an HTML block. + content []byte + + level int // heading level, 1 to 6 + + fenced bool // code block opened by a fence + fenceChar byte // fence character, '`' or '~' + fenceLength int // length of the opening fence + fenceOffset int // columns of indentation before the opening fence + info string + + htmlType int // HTML block start condition, 1 to 7 + + listKind listKind + bulletChar byte // bullet list: the marker character + delimiter byte // ordered list: '.' or ')' + start int // ordered list: number of the first item + tight bool + markerOffset int // item: indentation of the marker inside its container + padding int // item: columns from the marker start to the content + + // refs collects the link reference definitions of the document; it is + // carried by the root node only. + refs map[string]reference + + // footnotes collects the footnote definitions of the document, in + // document order; it is carried by the root node only. The renderer + // orders the rendered section by reference. + footnotes []*Node + + // label names a footnote definition. + label string + + // table columns and rows, cells as raw inline content. + align []uint8 + header [][]byte + rows [][][]byte + + task bool // item: the first paragraph begins with a task marker + taskDone bool + + startLine int + + lastLineBlank bool + lastLineChecked bool + finalised bool +} + +// reference is one link reference definition of the document. +type reference struct { + destination string + title string + hasTitle bool +} + +// canContain reports whether parent accepts child blocks of the given kind. +func canContain(parent, child nodeKind) bool { + switch parent { + case kindDocument, kindBlockquote, kindListItem, kindDefItem, kindFootnoteDef: + return child != kindDocument + case kindList: + return child == kindListItem + case kindDefList: + return child == kindDefTerm || child == kindDefItem || child == kindParagraph + } + return false +} diff --git a/internal/markdown/parse.go b/internal/markdown/parse.go new file mode 100644 index 0000000..ff961ae --- /dev/null +++ b/internal/markdown/parse.go @@ -0,0 +1,817 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package markdown + +import "bytes" + +const ( + tabStop = 4 + codeIndent = 4 +) + +// Parse parses Markdown source into a tree of block nodes. The source is +// trusted input: no sanitisation is applied, because whether the output may +// reach an audience is the consumer's policy. +func Parse(source []byte) *Node { + p := &parser{doc: &Node{kind: kindDocument, refs: map[string]reference{}}} + p.tip = p.doc + for _, line := range splitLines(normalise(source)) { + p.processLine(line) + } + p.closeUnmatched(p.doc) + p.finalise(p.doc) + return p.doc +} + +// parser holds the block parsing state. The offset, column, indent and +// blank fields describe the current line from the current offset onward, +// which moves as container prefixes are consumed; a tab may end up +// partially consumed, in which case offset rests on the tab and column +// counts only the consumed part of it. +type parser struct { + doc *Node + tip *Node + + line []byte + lineNo int + + offset int + column int + + firstNonspace int + firstNonspaceColumn int + indent int + blank bool + + partiallyConsumedTab bool + + // suppressBlankMark keeps the blank-line bookkeeping away from a line + // the parser consumed entirely, such as a closing code fence. + suppressBlankMark bool +} + +func (p *parser) processLine(line []byte) { + p.line = line + p.lineNo++ + p.offset = 0 + p.column = 0 + p.partiallyConsumedTab = false + p.findFirstNonspace() + + lastMatched := p.checkOpenBlocks() + container, opened := p.openNewBlocks(lastMatched) + p.addText(container, opened) + + switch p.tip.kind { + case kindHeading, kindThematicBreak: + p.finalise(p.tip) + } +} + +// checkOpenBlocks matches the line against the open block chain, from the +// document down to the tip, consuming the prefix of every block that +// continues. It returns the deepest matched block; the blocks below it stay +// open until addText closes or lazily continues them. +func (p *parser) checkOpenBlocks() *Node { + var chain []*Node + for n := p.tip; n != nil; n = n.parent { + chain = append(chain, n) + } + for i := len(chain) - 2; i >= 0; i-- { + p.findFirstNonspace() + if !p.continueBlock(chain[i]) { + return chain[i+1] + } + } + return chain[0] +} + +// continueBlock reports whether the open block continues on the current +// line, consuming its prefix when it does. +func (p *parser) continueBlock(n *Node) bool { + switch n.kind { + case kindBlockquote: + if p.blank || p.indent > 3 { + return false + } + if p.line[p.firstNonspace] != '>' { + return false + } + p.advanceOffset(p.firstNonspace+1-p.offset, false) + if p.offset < len(p.line) && isSpaceTab(p.line[p.offset]) { + p.advanceOffset(1, true) + } + return true + case kindListItem: + if p.blank { + // A blank line ends an item that never took content. + if len(n.children) == 0 { + return false + } + p.advanceOffset(p.firstNonspace-p.offset, false) + return true + } + if p.indent >= n.markerOffset+n.padding { + p.advanceOffset(n.markerOffset+n.padding, true) + return true + } + return false + case kindCodeBlock: + if !n.fenced { + switch { + case p.indent >= codeIndent: + p.advanceOffset(codeIndent, true) + return true + case p.blank: + p.advanceOffset(p.firstNonspace-p.offset, false) + return true + } + return false + } + if !p.blank && p.indent <= 3 && p.line[p.firstNonspace] == n.fenceChar { + length := fenceRun(p.line[p.firstNonspace:], n.fenceChar) + if length >= n.fenceLength && allSpaceTab(p.line[p.firstNonspace+length:]) { + // A closing fence ends the block; the rest of the line is + // nothing but whitespace. + p.advanceOffset(len(p.line)-p.offset, false) + p.finalise(n) + p.suppressBlankMark = true + return false + } + } + // A content line gives up to the opening fence's indentation. + for i := n.fenceOffset; i > 0 && p.offset < len(p.line) && isSpaceTab(p.line[p.offset]); i-- { + p.advanceOffset(1, true) + } + return true + case kindHTMLBlock: + // The tag-based kinds end at a blank line; the raw kinds run to + // their closing condition. + if n.htmlType == 6 || n.htmlType == 7 { + return !p.blank + } + return true + case kindParagraph: + return !p.blank + case kindTable: + return !p.blank + case kindFootnoteDef: + if p.blank { + p.advanceOffset(p.firstNonspace-p.offset, false) + return true + } + if p.indent >= codeIndent { + p.advanceOffset(codeIndent, true) + return true + } + return false + case kindDefItem: + if p.blank { + p.advanceOffset(p.firstNonspace-p.offset, false) + return true + } + if p.indent >= n.markerOffset+n.padding { + p.advanceOffset(n.markerOffset+n.padding, true) + return true + } + return false + } + return true +} + +// openNewBlocks starts new blocks on the line, beginning at the last +// matched container and opening containers until a leaf takes over. It +// returns the container the remaining text belongs to and whether anything +// was opened or changed on the line. +func (p *parser) openNewBlocks(container *Node) (*Node, bool) { + opened := false + for { + switch container.kind { + case kindCodeBlock, kindHTMLBlock: + return container, opened + } + p.findFirstNonspace() + if p.blank { + return container, opened + } + indented := p.indent >= codeIndent + + if !indented && p.line[p.firstNonspace] == '>' { + p.advanceOffset(p.firstNonspace+1-p.offset, false) + if p.offset < len(p.line) && isSpaceTab(p.line[p.offset]) { + p.advanceOffset(1, true) + } + container = p.addChild(container, kindBlockquote) + opened = true + continue + } + if !indented { + if level, ok := scanATX(p.line[p.firstNonspace:]); ok { + h := p.addChild(container, kindHeading) + h.level = level + p.advanceOffset(p.firstNonspace+level-p.offset, false) + return h, true + } + } + if !indented { + if char, length, ok := scanOpenFence(p.line[p.firstNonspace:]); ok { + code := p.addChild(container, kindCodeBlock) + code.fenced = true + code.fenceChar = char + code.fenceLength = length + code.fenceOffset = p.indent + p.advanceOffset(p.firstNonspace+length-p.offset, false) + return code, true + } + } + if !indented { + if t := scanHTMLBlockStart(p.line[p.firstNonspace:], container.kind == kindParagraph); t > 0 { + h := p.addChild(container, kindHTMLBlock) + h.htmlType = t + return h, true + } + } + if !indented && container.kind == kindParagraph { + if aligns, ok := scanTableDelimiter(p.line[p.firstNonspace:]); ok { + if table := p.tryOpenTable(container, aligns); table != nil { + p.advanceOffset(len(p.line)-p.offset, false) + return table, true + } + } + } + if !indented { + if label, markerLen, ok := scanFootnoteDefStart(p.line[p.firstNonspace:]); ok { + def := p.addChild(container, kindFootnoteDef) + def.label = normaliseLabel(label) + def.padding = codeIndent + p.advanceOffset(p.firstNonspace+markerLen-p.offset, false) + container = def + opened = true + continue + } + } + if !indented { + if container.kind == kindParagraph || container.kind == kindDefList { + if scanDefMarker(p.line[p.firstNonspace:]) { + if item := p.tryOpenDefItem(container); item != nil { + container = item + opened = true + continue + } + } + } + } + if !indented && container.kind == kindParagraph { + if level, ok := scanSetext(p.line[p.firstNonspace:]); ok { + // Reference definitions leave the paragraph first; the + // heading forms only over what remains of it, and a + // paragraph the definitions emptied turns the underline + // back into plain text. + p.extractReferences(container) + if len(container.content) == 0 { + parent := container.parent + parent.children = parent.children[:len(parent.children)-1] + p.tip = parent + return parent, true + } + container.kind = kindHeading + container.level = level + p.advanceOffset(len(p.line)-p.offset, false) + return container, true + } + } + if !indented && isThematicBreak(p.line[p.firstNonspace:]) { + p.addChild(container, kindThematicBreak) + p.advanceOffset(len(p.line)-p.offset, false) + return p.tip, true + } + if !indented { + if data, markerLen, ok := p.parseListMarker(container.kind == kindParagraph); ok { + if container.kind != kindList || !listsMatch(container, data) { + l := p.addChild(container, kindList) + l.listKind = data.listKind + l.bulletChar = data.bulletChar + l.delimiter = data.delimiter + l.start = data.start + container = l + } + item := p.addChild(container, kindListItem) + item.markerOffset = p.indent + p.advanceOffset(p.firstNonspace+markerLen-p.offset, false) + saveOffset, saveColumn, saveTab := p.offset, p.column, p.partiallyConsumedTab + for p.offset < len(p.line) && isSpaceTab(p.line[p.offset]) { + p.advanceOffset(1, true) + } + cols := p.column - saveColumn + blankItem := p.offset >= len(p.line) + padding := markerLen + cols + if blankItem || cols >= 5 || cols < 1 { + padding = markerLen + 1 + } + p.offset, p.column, p.partiallyConsumedTab = saveOffset, saveColumn, saveTab + item.padding = padding + p.advanceOffset(padding-markerLen, true) + container = item + opened = true + continue + } + } + return container, opened + } +} + +// listData carries what a list marker says about the list it belongs to. +type listData struct { + listKind listKind + bulletChar byte + delimiter byte + start int +} + +// listsMatch reports whether a new marker continues the given list. +func listsMatch(l *Node, d listData) bool { + if l.listKind != d.listKind { + return false + } + if d.listKind == bulletList { + return l.bulletChar == d.bulletChar + } + return l.delimiter == d.delimiter +} + +// parseListMarker recognises a list marker at the first non-space +// character. A marker interrupting a paragraph must carry content, and an +// ordered one must number 1. +func (p *parser) parseListMarker(interrupts bool) (listData, int, bool) { + s := p.line[p.firstNonspace:] + if len(s) == 0 { + return listData{}, 0, false + } + var d listData + i := 0 + switch c := s[0]; { + case c == '-' || c == '+' || c == '*': + d.listKind = bulletList + d.bulletChar = c + i = 1 + case c >= '0' && c <= '9': + start := 0 + for i < len(s) && i < 9 && s[i] >= '0' && s[i] <= '9' { + start = start*10 + int(s[i]-'0') + i++ + } + if i >= len(s) || (s[i] != '.' && s[i] != ')') { + return listData{}, 0, false + } + if interrupts && start != 1 { + return listData{}, 0, false + } + d.listKind = orderedList + d.delimiter = s[i] + d.start = start + i++ + default: + return listData{}, 0, false + } + if i < len(s) && !isSpaceTab(s[i]) { + return listData{}, 0, false + } + if interrupts { + j := i + for j < len(s) && isSpaceTab(s[j]) { + j++ + } + if j >= len(s) { + return listData{}, 0, false + } + } + return d, i, true +} + +// tryOpenTable turns the last line of an open paragraph into the header of +// a table whose delimiter row is on the current line. The paragraph keeps +// its earlier lines, or disappears when the header was all of it. The +// table opens only when the header and the delimiter row agree on the +// number of columns. +func (p *parser) tryOpenTable(para *Node, aligns []uint8) *Node { + content := para.content + if len(content) == 0 || content[len(content)-1] != '\n' { + return nil + } + lastNewline := bytes.LastIndexByte(content[:len(content)-1], '\n') + 1 + headerLine := content[lastNewline : len(content)-1] + cells := splitTableRow(headerLine) + if len(cells) != len(aligns) { + return nil + } + parent := para.parent + if lastNewline > 0 { + para.content = content[:lastNewline] + p.finalise(para) + } else { + parent.children = parent.children[:len(parent.children)-1] + p.tip = parent + } + table := p.addChild(parent, kindTable) + table.align = aligns + table.header = cells + return table +} + +// tryOpenDefItem turns the line's definition marker into a definition +// inside a definition list. When the container is a paragraph, the +// paragraph's lines become the terms of a new entry; when it is a +// definition list, the marker adds another definition to the entry. The +// returned item is the container for the definition's content, with the +// position placed at the content. +func (p *parser) tryOpenDefItem(container *Node) *Node { + var dl *Node + if container.kind == kindParagraph { + dl = p.defListFromTerms(container) + if dl == nil { + return nil + } + } else { + dl = container + } + item := p.addChild(dl, kindDefItem) + item.markerOffset = p.indent + p.advanceOffset(p.firstNonspace+1-p.offset, false) + saveOffset, saveColumn, saveTab := p.offset, p.column, p.partiallyConsumedTab + for p.offset < len(p.line) && isSpaceTab(p.line[p.offset]) { + p.advanceOffset(1, true) + } + cols := p.column - saveColumn + blankItem := p.offset >= len(p.line) + padding := 1 + cols + if blankItem || cols >= 5 || cols < 1 { + padding = 2 + } + p.offset, p.column, p.partiallyConsumedTab = saveOffset, saveColumn, saveTab + item.padding = padding + p.advanceOffset(padding-1, true) + return item +} + +// defListFromTerms turns a paragraph of terms into definition terms of a +// definition list, continuing the list when one is already open beside the +// paragraph. +func (p *parser) defListFromTerms(para *Node) *Node { + if len(para.content) == 0 || para.content[len(para.content)-1] != '\n' { + return nil + } + lines := splitLines(para.content) + if len(lines) == 0 { + return nil + } + parent := para.parent + parent.children = parent.children[:len(parent.children)-1] + p.tip = parent + var dl *Node + switch { + case parent.kind == kindDefList: + dl = parent + case len(parent.children) > 0 && parent.children[len(parent.children)-1].kind == kindDefList: + dl = parent.children[len(parent.children)-1] + default: + dl = p.addChild(parent, kindDefList) + } + for _, line := range lines { + dt := p.addChild(dl, kindDefTerm) + dt.content = line + } + return dl +} + +// addText places the remaining text of the line: into the open paragraph +// when it continues, lazily or not, or into the container that accepts +// lines, or into a new block otherwise. It also records which blocks the +// blank line terminates, which decides list tightness. +func (p *parser) addText(container *Node, opened bool) { + p.findFirstNonspace() + + if p.suppressBlankMark { + p.suppressBlankMark = false + return + } + + if p.blank && len(container.children) > 0 { + container.children[len(container.children)-1].lastLineBlank = true + } + lastBlank := p.blank && + container.kind != kindBlockquote && + container.kind != kindHeading && + container.kind != kindThematicBreak && + !(container.kind == kindCodeBlock && container.fenced) && + !(container.kind == kindListItem && len(container.children) == 0 && container.startLine == p.lineNo) + container.lastLineBlank = lastBlank + for n := container.parent; n != nil; n = n.parent { + n.lastLineBlank = false + } + + maybeLazy := p.tip.kind == kindParagraph + if maybeLazy && !opened && !p.blank { + p.advanceOffset(p.firstNonspace-p.offset, false) + p.appendLine(p.tip, true) + return + } + + p.closeUnmatched(container) + + switch { + case container.kind == kindCodeBlock: + p.appendLine(container, true) + case container.kind == kindHTMLBlock: + p.appendLine(container, true) + if htmlBlockEnds(container.htmlType, p.line[p.offset:]) { + p.finalise(container) + } + case p.blank: + // A blank line adds no content. + case container.kind == kindTable: + row := splitTableRow(p.line[p.offset:]) + for len(row) < len(container.header) { + row = append(row, nil) + } + container.rows = append(container.rows, row[:len(container.header)]) + case container.kind == kindParagraph: + p.advanceOffset(p.firstNonspace-p.offset, false) + p.appendLine(container, true) + case container.kind == kindHeading: + container.content = append(container.content, chopClosingHashes(p.line[p.firstNonspace:])...) + default: + if p.indent >= codeIndent && !maybeLazy { + code := p.addChild(container, kindCodeBlock) + p.advanceOffset(codeIndent, true) + p.appendLine(code, true) + } else { + if container.kind == kindListItem && len(container.children) == 0 && container.startLine == p.lineNo { + if checked, n, ok := scanTaskMarker(p.line[p.firstNonspace:]); ok { + p.advanceOffset(p.firstNonspace+n-p.offset, false) + container.task = true + container.taskDone = checked + p.findFirstNonspace() + } + } + para := p.addChild(container, kindParagraph) + p.advanceOffset(p.firstNonspace-p.offset, false) + p.appendLine(para, true) + } + } +} + +// addChild attaches a new block below parent, closing open blocks that +// cannot contain it, and makes it the tip. +func (p *parser) addChild(parent *Node, kind nodeKind) *Node { + for !canContain(parent.kind, kind) { + p.finalise(parent) + parent = parent.parent + } + n := &Node{kind: kind, parent: parent, startLine: p.lineNo} + parent.children = append(parent.children, n) + p.tip = n + return n +} + +// closeUnmatched closes every open block below the given container. +func (p *parser) closeUnmatched(container *Node) { + for p.tip != container { + p.finalise(p.tip) + } +} + +// finalise closes a block and its children, trimming content and computing +// derived data such as list tightness. +func (p *parser) finalise(n *Node) { + if n.finalised { + return + } + n.finalised = true + for _, c := range n.children { + p.finalise(c) + } + switch n.kind { + case kindParagraph: + p.extractReferences(n) + if len(n.content) == 0 && n.parent != nil { + n.parent.children = n.parent.children[:len(n.parent.children)-1] + } + case kindHeading: + if bytes.HasSuffix(n.content, []byte("\n")) { + n.content = n.content[:len(n.content)-1] + } + case kindCodeBlock: + if n.fenced { + if i := bytes.IndexByte(n.content, '\n'); i >= 0 { + n.info = unescapeText(string(bytes.TrimSpace(n.content[:i]))) + n.content = n.content[i+1:] + } else { + n.info = unescapeText(string(bytes.TrimSpace(n.content))) + n.content = nil + } + } else { + n.content = trimTrailingBlankLines(n.content) + } + case kindList, kindDefList: + n.tight = !blocksAreLoose(n.children) + case kindDocument: + p.gatherFootnotes(n) + } + if p.tip == n { + p.tip = n.parent + } +} + +// trimTrailingBlankLines removes the trailing blank lines of an indented +// code block. Every content line carries its newline, so the result of a +// non-empty block ends with exactly one. +func trimTrailingBlankLines(c []byte) []byte { + for len(c) > 0 { + end := len(c) - 1 // the final newline + start := bytes.LastIndexByte(c[:end], '\n') + 1 + if !allSpaceTab(c[start:end]) { + return c + } + c = c[:start] + } + return c +} + +// blocksAreLoose reports whether any two sibling blocks, or any two blocks +// of one container-like block, are separated by a blank line. +func blocksAreLoose(nodes []*Node) bool { + for i, n := range nodes { + if endsWithBlank(n) && i+1 < len(nodes) { + return true + } + switch n.kind { + case kindListItem, kindDefItem: + for j, child := range n.children { + lastItem := i+1 == len(nodes) + lastChild := j+1 == len(n.children) + if endsWithBlank(child) && (!lastItem || !lastChild) { + return true + } + } + } + } + return false +} + +// endsWithBlank reports whether the block, or the last block inside a +// container chain, was followed by a blank line. +func endsWithBlank(n *Node) bool { + if n.lastLineChecked { + return n.lastLineBlank + } + n.lastLineChecked = true + switch n.kind { + case kindList, kindListItem, kindDefList, kindDefItem, kindFootnoteDef: + if len(n.children) > 0 { + return endsWithBlank(n.children[len(n.children)-1]) + } + } + return n.lastLineBlank +} + +// gatherFootnotes lifts every footnote definition out of the tree, keeping +// the first definition of a label. +func (p *parser) gatherFootnotes(doc *Node) { + seen := map[string]bool{} + var defs []*Node + var walk func(n *Node) + walk = func(n *Node) { + keep := n.children[:0] + for _, c := range n.children { + if c.kind == kindFootnoteDef { + if !seen[c.label] { + seen[c.label] = true + defs = append(defs, c) + } + continue + } + walk(c) + keep = append(keep, c) + } + n.children = keep + } + walk(doc) + doc.footnotes = defs +} + +// appendLine appends the remaining line to the block's content. A tab +// partially consumed while skipping indentation becomes the spaces it +// stood for. +func (p *parser) appendLine(n *Node, newline bool) { + if p.partiallyConsumedTab { + p.offset++ + for i := tabStop - p.column%tabStop; i > 0; i-- { + n.content = append(n.content, ' ') + } + } + n.content = append(n.content, p.line[p.offset:]...) + if newline { + n.content = append(n.content, '\n') + } +} + +// advanceOffset moves into the line by count bytes, or columns when +// columns is set, expanding tabs to tab stops and leaving a tab partially +// consumed when the count stops inside it. +func (p *parser) advanceOffset(count int, columns bool) { + for count > 0 && p.offset < len(p.line) { + c := p.line[p.offset] + if c != '\t' { + p.partiallyConsumedTab = false + p.offset++ + p.column++ + count-- + continue + } + toTab := tabStop - p.column%tabStop + if columns { + p.partiallyConsumedTab = toTab > count + steps := min(count, toTab) + p.column += steps + if !p.partiallyConsumedTab { + p.offset++ + } + count -= steps + } else { + p.partiallyConsumedTab = false + p.offset++ + p.column += toTab + count-- + } + } +} + +// findFirstNonspace locates the first non-space character from the offset +// onward, computing the column of that character and the indentation of +// the line relative to the offset. The line is blank when the first +// non-space character does not exist. +func (p *parser) findFirstNonspace() { + toTab := tabStop - p.column%tabStop + p.firstNonspace = p.offset + p.firstNonspaceColumn = p.column + for p.firstNonspace < len(p.line) { + c := p.line[p.firstNonspace] + switch c { + case ' ': + p.firstNonspace++ + p.firstNonspaceColumn++ + toTab-- + if toTab == 0 { + toTab = tabStop + } + case '\t': + p.firstNonspace++ + p.firstNonspaceColumn += toTab + toTab = tabStop + default: + p.indent = p.firstNonspaceColumn - p.column + p.blank = false + return + } + } + p.indent = p.firstNonspaceColumn - p.column + p.blank = true +} + +// normalise prepares source for parsing: line endings become newlines and +// a null byte becomes the replacement character. +func normalise(source []byte) []byte { + out := make([]byte, 0, len(source)) + for i := 0; i < len(source); i++ { + switch c := source[i]; c { + case '\r': + if i+1 < len(source) && source[i+1] == '\n' { + i++ + } + out = append(out, '\n') + case 0: + out = append(out, '\xef', '\xbf', '\xbd') + default: + out = append(out, c) + } + } + return out +} + +// splitLines splits normalised source into lines without their newlines. +// A trailing newline produces no empty final line. +func splitLines(source []byte) [][]byte { + var lines [][]byte + start := 0 + for i, c := range source { + if c == '\n' { + lines = append(lines, source[start:i]) + start = i + 1 + } + } + if start < len(source) { + lines = append(lines, source[start:]) + } + return lines +} diff --git a/internal/markdown/parse_test.go b/internal/markdown/parse_test.go new file mode 100644 index 0000000..5568f3f --- /dev/null +++ b/internal/markdown/parse_test.go @@ -0,0 +1,69 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package markdown + +import ( + "bytes" + "testing" +) + +func TestDeterministicOutput(t *testing.T) { + source := []byte("# Title\n\nText with < and &.\n\n- one\n - nested\n\n> quoted\n\n" + + "```go\nx := 1\n```\n\n[ref]: /url \"title\"\n") + first := RenderHTML(source) + for i := range 10 { + if next := RenderHTML(source); !bytes.Equal(first, next) { + t.Fatalf("render %d differs:\nfirst: %q\nnext: %q", i, first, next) + } + } +} + +func TestReferenceNormalisation(t *testing.T) { + doc := Parse([]byte("[Foo Bar]: /first\n[FOO\t bar]: /second\n")) + if len(doc.refs) != 1 { + t.Fatalf("got %d references, want 1", len(doc.refs)) + } + ref, ok := doc.refs["foo bar"] + if !ok { + t.Fatalf("no reference under the normalised label \"foo bar\"") + } + if ref.destination != "/first" { + t.Errorf("destination = %q, want \"/first\": the first definition wins", ref.destination) + } +} + +func TestReferenceTitleKept(t *testing.T) { + doc := Parse([]byte("[foo]: /url 'the title'\n")) + ref, ok := doc.refs["foo"] + if !ok { + t.Fatal("no reference recorded") + } + if !ref.hasTitle || ref.title != "the title" { + t.Errorf("title = %q with hasTitle %v, want \"the title\" with hasTitle", ref.title, ref.hasTitle) + } +} + +func TestJunkAfterTitleIsNotADefinition(t *testing.T) { + doc := Parse([]byte("[foo]: /url \"title\" ok\n")) + if len(doc.refs) != 0 { + t.Errorf("got %d references, want 0", len(doc.refs)) + } + if len(doc.children) != 1 || doc.children[0].kind != kindParagraph { + t.Fatalf("the line should stay a paragraph") + } +} + +func TestBacktickInfoStringRejectsFence(t *testing.T) { + doc := Parse([]byte("``` aaa ```\n")) + if len(doc.children) != 1 || doc.children[0].kind != kindParagraph { + t.Fatalf("the line should stay a paragraph, got %d children", len(doc.children)) + } +} + +func TestParseEmptyDocument(t *testing.T) { + doc := Parse(nil) + if doc.kind != kindDocument || len(doc.children) != 0 { + t.Errorf("Parse(nil) should give an empty document") + } +} diff --git a/internal/markdown/refdef.go b/internal/markdown/refdef.go new file mode 100644 index 0000000..9fe4177 --- /dev/null +++ b/internal/markdown/refdef.go @@ -0,0 +1,253 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package markdown + +import ( + "bytes" + "strings" +) + +// extractReferences strips leading link reference definitions from a +// closed paragraph and records them in the document. The first definition +// of a label wins. What remains of the paragraph keeps a single trailing +// newline removed. +func (p *parser) extractReferences(n *Node) { + for { + def, consumed, ok := parseReference(n.content) + if !ok || consumed <= 0 { + break + } + if _, exists := p.doc.refs[def.label]; !exists { + p.doc.refs[def.label] = def.reference + } + n.content = n.content[consumed:] + } + if bytes.HasSuffix(n.content, []byte("\n")) { + n.content = n.content[:len(n.content)-1] + } +} + +// refDef is a parsed link reference definition: the normalised label under +// which it is recorded and the reference it defines. +type refDef struct { + label string + reference +} + +// parseReference parses one link reference definition from the start of +// the paragraph content, possibly spanning lines. It returns the +// definition, the number of bytes consumed and whether a definition was +// there at all. +func parseReference(c []byte) (refDef, int, bool) { + // The label: brackets around one to 999 characters, no unescaped + // bracket inside. + if len(c) == 0 || c[0] != '[' { + return refDef{}, 0, false + } + i := 1 + labelEnd := -1 + for i < len(c) { + ch := c[i] + if ch == '\\' && i+1 < len(c) { + i += 2 + continue + } + if ch == ']' { + labelEnd = i + break + } + i++ + } + if labelEnd < 0 { + return refDef{}, 0, false + } + label := c[1:labelEnd] + if len(label) < 1 || len(label) > 999 || len(bytes.TrimSpace(label)) == 0 || !validLabel(label) { + return refDef{}, 0, false + } + i = labelEnd + 1 + if i >= len(c) || c[i] != ':' { + return refDef{}, 0, false + } + i++ + + // Up to one line ending may sit between the colon and the destination. + i, _, ok := skipSpaceOneNewline(c, i) + if !ok { + return refDef{}, 0, false + } + dest, n, ok := scanDestination(c[i:]) + if !ok { + return refDef{}, 0, false + } + i += n + + // A definition without a title needs the rest of the destination's + // line to be blank. + titlelessEnd := -1 + if lineEnd := bytes.IndexByte(c[i:], '\n'); lineEnd < 0 { + if allSpaceTab(c[i:]) { + titlelessEnd = len(c) + } + } else if allSpaceTab(c[i : i+lineEnd]) { + titlelessEnd = i + lineEnd + 1 + } + + // A title, when present, sits after at least one character of + // whitespace, with at most one line ending between it and the + // destination, and nothing but whitespace may follow it. + if j, skipped, ok := skipSpaceOneNewline(c, i); ok && skipped > 0 && j < len(c) && (c[j] == '"' || c[j] == '\'' || c[j] == '(') { + open := c[j] + closer := open + if open == '(' { + closer = ')' + } + k := j + 1 + for k < len(c) { + ch := c[k] + if ch == '\\' && k+1 < len(c) { + k += 2 + continue + } + if open == '(' && ch == '(' { + break + } + if ch == closer { + after := k + 1 + for after < len(c) && isSpaceTab(c[after]) { + after++ + } + if after >= len(c) { + return def(label, dest, c[j+1:k], len(c)) + } + if c[after] == '\n' { + return def(label, dest, c[j+1:k], after+1) + } + break + } + k++ + } + } + + if titlelessEnd < 0 { + return refDef{}, 0, false + } + return def(label, dest, nil, titlelessEnd) +} + +// def builds the result of a parsed definition. +func def(label, dest, title []byte, consumed int) (refDef, int, bool) { + r := reference{destination: unescapeText(string(dest))} + if title != nil { + r.title = unescapeText(string(title)) + r.hasTitle = true + } + return refDef{label: normaliseLabel(string(label)), reference: r}, consumed, true +} + +// skipSpaceOneNewline skips spaces, tabs and at most one newline, stopping +// at the first other character or the end. It reports how much it skipped +// and fails on a second line ending. +func skipSpaceOneNewline(c []byte, i int) (int, int, bool) { + start := i + newlines := 0 + for i < len(c) { + switch c[i] { + case ' ', '\t': + i++ + case '\n': + newlines++ + if newlines > 1 { + return i, i - start, false + } + i++ + default: + return i, i - start, true + } + } + return i, i - start, true +} + +// scanDestination parses a link destination: a run in angle brackets with +// no line ending inside, or a bare run without whitespace in which +// parentheses stay balanced. +func scanDestination(c []byte) ([]byte, int, bool) { + if len(c) > 0 && c[0] == '<' { + i := 1 + for i < len(c) { + ch := c[i] + if ch == '\\' && i+1 < len(c) { + i += 2 + continue + } + if ch == '>' { + return c[1:i], i + 1, true + } + if ch == '<' || ch == '\n' { + return nil, 0, false + } + i++ + } + return nil, 0, false + } + i := 0 + depth := 0 + for i < len(c) { + ch := c[i] + if ch == '\\' && i+1 < len(c) { + i += 2 + continue + } + if ch == '(' { + depth++ + i++ + continue + } + if ch == ')' { + if depth == 0 { + break + } + depth-- + i++ + continue + } + if isSpaceTab(ch) || ch == '\n' { + break + } + i++ + } + if depth != 0 || i == 0 { + return nil, 0, false + } + return c[:i], i, true +} + +// validLabel reports whether the raw label text carries no unescaped open +// bracket, which a link label may not contain. +func validLabel(label []byte) bool { + for i := 0; i < len(label); i++ { + switch label[i] { + case '\\': + i++ + case '[': + return false + } + } + return true +} + +// labelFolder carries the case fold pairs the standard library's lower +// casing does not perform, which the Unicode case fold CommonMark names +// does. +var labelFolder = strings.NewReplacer( + "ß", "ss", "ff", "ff", "fi", "fi", "fl", "fl", + "ffi", "ffi", "ffl", "ffl", "ſt", "st", "st", "st", +) + +// normaliseLabel brings a link label to the form definitions and uses are +// compared under: surrounding and repeated whitespace collapsed to single +// spaces, then case folded. +func normaliseLabel(s string) string { + return labelFolder.Replace(strings.ToLower(strings.Join(strings.Fields(s), " "))) +} diff --git a/internal/markdown/render.go b/internal/markdown/render.go new file mode 100644 index 0000000..50a9536 --- /dev/null +++ b/internal/markdown/render.go @@ -0,0 +1,336 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package markdown + +import ( + "strconv" + "strings" +) + +// renderer serialises the block tree. Every block ends its output with a +// newline; a paragraph inside a tight list item is the one exception, where +// the text stands without its own wrapper. +type renderer struct { + out []byte + refs map[string]reference + footnotes *footnoteTracker +} + +func (r *renderer) puts(s string) { + r.out = append(r.out, s...) +} + +func (r *renderer) blocks(nodes []*Node) { + for _, n := range nodes { + r.block(n) + } +} + +func (r *renderer) block(n *Node) { + switch n.kind { + case kindParagraph: + task := taskInput(n) + if tightParent(n) { + r.puts(task) + r.inlineContent(n.content) + if !lastChild(n) { + r.puts("\n") + } + return + } + r.puts("

") + r.puts(task) + r.inlineContent(n.content) + r.puts("

\n") + case kindHeading: + level := strconv.Itoa(n.level) + r.puts("") + r.inlineContent(n.content) + r.puts("\n") + case kindCodeBlock: + r.puts("
")
+		r.puts(escapeHTML(string(n.content)))
+		r.puts("
\n") + case kindHTMLBlock: + r.out = append(r.out, n.content...) + case kindBlockquote: + r.puts("
\n") + r.blocks(n.children) + r.puts("
\n") + case kindList: + switch n.listKind { + case bulletList: + r.puts("
    \n") + r.blocks(n.children) + r.puts("
\n") + case orderedList: + if n.start != 1 { + r.puts(`
    ` + "\n") + } else { + r.puts("
      \n") + } + r.blocks(n.children) + r.puts("
    \n") + } + case kindListItem: + r.itemLike("li", n) + case kindDefItem: + r.itemLike("dd", n) + case kindDefTerm: + r.puts("
    ") + r.inlineContent(n.content) + r.puts("
    \n") + case kindDefList: + r.puts("
    \n") + r.blocks(n.children) + r.puts("
    \n") + case kindThematicBreak: + r.puts("
    \n") + case kindTable: + r.puts("\n\n\n") + for i, cell := range n.header { + r.puts("") + r.inlineContent(cell) + r.puts("\n") + } + r.puts("\n\n") + if len(n.rows) > 0 { + r.puts("\n") + for _, row := range n.rows { + r.puts("\n") + for i, cell := range row { + r.puts("") + r.inlineContent(cell) + r.puts("\n") + } + r.puts("\n") + } + r.puts("\n") + } + r.puts("
    \n") + } +} + +// itemLike renders a container of a list-shaped structure: an item of a +// list or a definition of a definition list. An empty container never +// gains a newline, whatever the tightness. +func (r *renderer) itemLike(tag string, n *Node) { + firstIsParagraph := len(n.children) > 0 && n.children[0].kind == kindParagraph + r.puts("<" + tag + ">") + if len(n.children) > 0 && (!n.parent.tight || !firstIsParagraph) { + r.puts("\n") + } + r.blocks(n.children) + r.puts("\n") +} + +// taskInput renders the checkbox of a task list item, at the head of the +// item's first paragraph. +func taskInput(n *Node) string { + if n.parent == nil || n.parent.kind != kindListItem || !n.parent.task { + return "" + } + siblings := n.parent.children + if siblings[0] != n { + return "" + } + if n.parent.taskDone { + return ` ` + } + return ` ` +} + +// alignAttr writes the alignment attribute of a table column. The +// attribute value keeps the spelling the HTML vocabulary defines. +func (r *renderer) alignAttr(a uint8) { + switch a { + case alignLeft: + r.puts(` align="left"`) + case alignCentre: + r.puts(` align="center"`) + case alignRight: + r.puts(` align="right"`) + } +} + +// inlineContent parses and renders the inline content of a leaf block. +func (r *renderer) inlineContent(content []byte) { + for _, n := range parseInlines(content, r.refs, r.footnotes) { + r.renderInline(n) + } +} + +func (r *renderer) renderInline(n *inline) { + switch n.kind { + case inlText: + r.puts(escapeHTML(n.literal)) + case inlCode: + r.puts("") + r.puts(escapeHTML(n.literal)) + r.puts("") + case inlRawHTML: + r.puts(n.literal) + case inlEmph: + r.puts("") + r.inlineNodes(n.children) + r.puts("") + case inlStrong: + r.puts("") + r.inlineNodes(n.children) + r.puts("") + case inlStrikethrough: + r.puts("") + r.inlineNodes(n.children) + r.puts("") + case inlLink: + r.puts(`") + r.inlineNodes(n.children) + r.puts("") + case inlImage: + r.puts(`` + escapeHTML(plainText(n.children)) + `") + case inlBreak: + r.puts("
    \n") + case inlFootnoteRef: + id := "fnref-" + strconv.Itoa(n.num) + if n.occurrence > 1 { + id += "-" + strconv.Itoa(n.occurrence) + } + num := strconv.Itoa(n.num) + r.puts(`` + num + ``) + } +} + +// footnoteSection renders the definitions of every referenced footnote, in +// the order of their first reference. The back reference lands at the end +// of the definition's last paragraph. +func (r *renderer) footnoteSection() { + if len(r.footnotes.order) == 0 { + return + } + r.puts("
    \n
      \n") + for i, def := range r.footnotes.order { + num := strconv.Itoa(i + 1) + r.puts(`
    1. ` + "\n") + backref := ` ` + "↩" + `` + for j, c := range def.children { + if c.kind == kindParagraph && j+1 == len(def.children) { + r.puts("

      ") + r.inlineContent(c.content) + r.puts(backref) + r.puts("

      \n") + continue + } + r.block(c) + } + r.puts("
    2. \n") + } + r.puts("
    \n
    \n") +} + +func (r *renderer) inlineNodes(nodes []*inline) { + for _, n := range nodes { + r.renderInline(n) + } +} + +// plainText renders inline nodes without markup, for the alt text of an +// image. +func plainText(nodes []*inline) string { + var b strings.Builder + for _, n := range nodes { + switch n.kind { + case inlText, inlCode, inlRawHTML: + b.WriteString(n.literal) + case inlBreak: + b.WriteByte('\n') + default: + b.WriteString(plainText(n.children)) + } + } + return b.String() +} + +// urlSafe marks the bytes that stay literal in an escaped destination. +const urlSafe = "!#$%()*+,-./:;=?@_~$" + +// escapeURL escapes a link destination for an href or src attribute: the +// ampersand and the apostrophe become entities, the bytes outside the safe +// set become percent escapes. +func escapeURL(s string) string { + var b strings.Builder + for i := 0; i < len(s); i++ { + c := s[i] + switch { + case c == '&': + b.WriteString("&") + case c == '\'': + b.WriteString("'") + case isAlnum(c) || strings.IndexByte(urlSafe, c) >= 0: + b.WriteByte(c) + default: + b.WriteByte('%') + b.WriteByte("0123456789ABCDEF"[c>>4]) + b.WriteByte("0123456789ABCDEF"[c&0x0f]) + } + } + return b.String() +} + +// tightParent reports whether the node is a paragraph directly inside an +// item of a tight list or a definition of a tight definition list. +func tightParent(n *Node) bool { + if n.parent == nil || n.parent.parent == nil { + return false + } + switch { + case n.parent.kind == kindListItem && n.parent.parent.kind == kindList: + return n.parent.parent.tight + case n.parent.kind == kindDefItem && n.parent.parent.kind == kindDefList: + return n.parent.parent.tight + } + return false +} + +// lastChild reports whether the node is the last child of its parent. +func lastChild(n *Node) bool { + siblings := n.parent.children + return siblings[len(siblings)-1] == n +} + +// infoWord returns the first word of a code block's info string. +func infoWord(info string) string { + fields := strings.Fields(info) + if len(fields) == 0 { + return "" + } + return fields[0] +} + +var htmlEscaper = strings.NewReplacer( + "&", "&", + "<", "<", + ">", ">", + `"`, """, +) + +// escapeHTML escapes plain text for HTML output. +func escapeHTML(s string) string { + return htmlEscaper.Replace(s) +} diff --git a/internal/markdown/scanners.go b/internal/markdown/scanners.go new file mode 100644 index 0000000..b49e1d7 --- /dev/null +++ b/internal/markdown/scanners.go @@ -0,0 +1,752 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package markdown + +import ( + "bytes" + "html" + "strings" +) + +func isSpaceTab(c byte) bool { return c == ' ' || c == '\t' } + +// isTagSpace marks the whitespace an inline HTML tag may contain between +// its parts, line endings included. +func isTagSpace(c byte) bool { return c == ' ' || c == '\t' || c == '\n' } + +func isAlpha(c byte) bool { + return c >= 'a' && c <= 'z' || c >= 'A' && c <= 'Z' +} + +func isAlnum(c byte) bool { + return isAlpha(c) || c >= '0' && c <= '9' +} + +// allSpaceTab reports whether s is empty or holds only spaces and tabs. +func allSpaceTab(s []byte) bool { + for _, c := range s { + if !isSpaceTab(c) { + return false + } + } + return true +} + +// scanATX recognises an ATX heading opener: one to six hashes followed by +// a space, a tab or the end of the line. It returns the heading level. +func scanATX(s []byte) (int, bool) { + n := 0 + for n < len(s) && s[n] == '#' { + n++ + } + if n == 0 || n > 6 { + return 0, false + } + if n < len(s) && !isSpaceTab(s[n]) { + return 0, false + } + return n, true +} + +// chopClosingHashes removes an ATX heading's closing sequence of hashes +// together with the whitespace around it. +func chopClosingHashes(s []byte) []byte { + end := len(s) + for end > 0 && isSpaceTab(s[end-1]) { + end-- + } + h := end + for h > 0 && s[h-1] == '#' { + h-- + } + if h == end { + return s[:end] + } + if h > 0 && !isSpaceTab(s[h-1]) { + return s[:end] + } + for h > 0 && isSpaceTab(s[h-1]) { + h-- + } + return s[:h] +} + +// scanSetext recognises a setext heading underline: a run of equals or +// dashes followed by nothing but whitespace. It returns the heading level. +func scanSetext(s []byte) (int, bool) { + if len(s) == 0 { + return 0, false + } + c := s[0] + if c != '=' && c != '-' { + return 0, false + } + i := 0 + for i < len(s) && s[i] == c { + i++ + } + if !allSpaceTab(s[i:]) { + return 0, false + } + if c == '=' { + return 1, true + } + return 2, true +} + +// isThematicBreak reports whether the line is a thematic break: three or +// more matching dashes, asterisks or underscores with optional whitespace +// between them. +func isThematicBreak(s []byte) bool { + if len(s) == 0 { + return false + } + c := s[0] + if c != '-' && c != '*' && c != '_' { + return false + } + count := 0 + for _, ch := range s { + switch ch { + case c: + count++ + case ' ', '\t': + default: + return false + } + } + return count >= 3 +} + +// fenceRun counts the leading run of the fence character. +func fenceRun(s []byte, c byte) int { + n := 0 + for n < len(s) && s[n] == c { + n++ + } + return n +} + +// scanOpenFence recognises a code fence opener: three or more backticks or +// tildes. The info string of a backtick fence may not contain a backtick, +// so such a line is not a fence at all. +func scanOpenFence(s []byte) (byte, int, bool) { + if len(s) == 0 { + return 0, 0, false + } + c := s[0] + if c != '`' && c != '~' { + return 0, 0, false + } + n := fenceRun(s, c) + if n < 3 { + return 0, 0, false + } + if c == '`' && bytes.IndexByte(s[n:], '`') >= 0 { + return 0, 0, false + } + return c, n, true +} + +// blockTagNames are the tag names whose open or closing tag starts an HTML +// block of the sixth kind. +var blockTagNames = map[string]bool{ + "address": true, "article": true, "aside": true, "base": true, + "basefont": true, "blockquote": true, "body": true, "caption": true, + "center": true, "col": true, "colgroup": true, "dd": true, + "details": true, "dialog": true, "dir": true, "div": true, + "dl": true, "dt": true, "fieldset": true, "figcaption": true, + "figure": true, "footer": true, "form": true, "frame": true, + "frameset": true, "h1": true, "h2": true, "h3": true, "h4": true, + "h5": true, "h6": true, "head": true, "header": true, "hr": true, + "html": true, "iframe": true, "legend": true, "li": true, + "link": true, "main": true, "menu": true, "menuitem": true, + "nav": true, "noframes": true, "ol": true, "optgroup": true, + "option": true, "p": true, "param": true, "search": true, + "section": true, "source": true, "summary": true, "table": true, + "tbody": true, "td": true, "tfoot": true, "th": true, "thead": true, + "title": true, "tr": true, "track": true, "ul": true, +} + +// scanHTMLBlockStart recognises an HTML block opener and returns its start +// condition, 1 to 7, or 0. The seventh condition, a complete tag alone on +// the line, may not interrupt a paragraph. +func scanHTMLBlockStart(s []byte, inParagraph bool) int { + if len(s) == 0 || s[0] != '<' { + return 0 + } + r := s[1:] + for _, name := range [...]string{"script", "pre", "style", "textarea"} { + if len(r) >= len(name) && bytes.EqualFold(r[:len(name)], []byte(name)) { + after := r[len(name):] + if len(after) == 0 || after[0] == ' ' || after[0] == '\t' || after[0] == '>' { + return 1 + } + } + } + if bytes.HasPrefix(r, []byte("!--")) { + return 2 + } + if len(r) > 0 && r[0] == '?' { + return 3 + } + if bytes.HasPrefix(r, []byte("![CDATA[")) { + return 5 + } + if len(r) > 1 && r[0] == '!' && isAlpha(r[1]) { + return 4 + } + if typeSixStart(r) { + return 6 + } + if !inParagraph && completeTag(r) { + return 7 + } + return 0 +} + +// typeSixStart reports whether r, the line after '<', opens an HTML block +// of the sixth kind: an optional slash, a known tag name and a boundary. +func typeSixStart(r []byte) bool { + i := 0 + if i < len(r) && r[i] == '/' { + i++ + } + start := i + for i < len(r) && isAlpha(r[i]) { + i++ + } + if i == start || !blockTagNames[strings.ToLower(string(r[start:i]))] { + return false + } + rest := r[i:] + if len(rest) == 0 || isSpaceTab(rest[0]) || rest[0] == '>' { + return true + } + return len(rest) >= 2 && rest[0] == '/' && rest[1] == '>' +} + +// completeTag reports whether r is a complete open or closing tag followed +// by nothing but whitespace, per the HTML grammar CommonMark quotes. +func completeTag(r []byte) bool { + n := tagLength(r) + return n > 0 && allSpaceTab(r[n:]) +} + +// tagLength returns the length of the open or closing tag at the start of +// r, including the final '>', or 0 when r does not begin with one. +func tagLength(r []byte) int { + i := 0 + closing := false + if i < len(r) && r[i] == '/' { + closing = true + i++ + } + if i >= len(r) || !isAlpha(r[i]) { + return 0 + } + for i < len(r) && (isAlnum(r[i]) || r[i] == '-') { + i++ + } + if closing { + for i < len(r) && isTagSpace(r[i]) { + i++ + } + if i < len(r) && r[i] == '>' { + return i + 1 + } + return 0 + } + for { + j := i + for j < len(r) && isTagSpace(r[j]) { + j++ + } + if j < len(r) && r[j] == '>' { + return j + 1 + } + if j+1 < len(r) && r[j] == '/' && r[j+1] == '>' { + return j + 2 + } + if j == i || j >= len(r) { + return 0 + } + i = j + if i >= len(r) || !(isAlpha(r[i]) || r[i] == '_' || r[i] == ':') { + return 0 + } + for i < len(r) && (isAlnum(r[i]) || r[i] == '_' || r[i] == ':' || r[i] == '.' || r[i] == '-') { + i++ + } + k := i + for k < len(r) && isTagSpace(r[k]) { + k++ + } + if k < len(r) && r[k] == '=' { + k++ + for k < len(r) && isTagSpace(r[k]) { + k++ + } + if k >= len(r) { + return 0 + } + switch r[k] { + case '"', '\'': + q := r[k] + k++ + for k < len(r) && r[k] != q { + k++ + } + if k >= len(r) { + return 0 + } + k++ + case '<', '>', '`', '=': + return 0 + default: + start := k + for k < len(r) && !isTagSpace(r[k]) && r[k] != '"' && r[k] != '\'' && r[k] != '=' && r[k] != '<' && r[k] != '>' && r[k] != '`' { + k++ + } + if k == start { + return 0 + } + } + i = k + } + } +} + +// scanRawHTML returns the length of the raw HTML construct at the start of +// s: a comment, a processing instruction, a declaration, a CDATA section, +// or an open or closing tag. +func scanRawHTML(s []byte) int { + if len(s) == 0 || s[0] != '<' { + return 0 + } + if bytes.HasPrefix(s, []byte(". The text may not start with +// '>' or '->' and may not end with '-'; the empty spellings are accepted. +func scanHTMLComment(s []byte) int { + if len(s) >= 5 && s[4] == '>' { + return 5 + } + if len(s) >= 6 && s[4] == '-' && s[5] == '>' { + return 6 + } + for i := 4; i < len(s); i++ { + if !bytes.HasPrefix(s[i:], []byte("-->")) { + continue + } + text := s[4:i] + if len(text) == 0 { + return i + 3 + } + if text[len(text)-1] == '-' { + return 0 + } + if text[0] == '>' || (len(text) > 1 && text[0] == '-' && text[1] == '>') { + return 0 + } + return i + 3 + } + return 0 +} + +// scanAutolink recognises a URI autolink or an email autolink at the start +// of s, returning its text, its destination and its length. +func scanAutolink(s []byte) (text, dest string, n int, ok bool) { + if len(s) == 0 || s[0] != '<' { + return "", "", 0, false + } + // An email autolink: local part, one at sign, and a domain of labels. + i := 1 + local := i + for i < len(s) && isEmailByte(s[i]) { + i++ + } + if i > local && i < len(s) && s[i] == '@' { + if end, domOK := scanEmailDomain(s, i+1); domOK && end < len(s) && s[end] == '>' { + addr := string(s[1:end]) + return addr, "mailto:" + addr, end + 1, true + } + } + // A URI autolink: scheme, colon, and a destination without whitespace + // or angle brackets. + i = 1 + schemeEnd := -1 + if i < len(s) && isAlpha(s[i]) { + i++ + for i < len(s) && i <= 32 && (isAlnum(s[i]) || s[i] == '+' || s[i] == '-' || s[i] == '.') { + i++ + } + if i < len(s) && s[i] == ':' && i >= 3 { + schemeEnd = i + } + } + if schemeEnd < 0 { + return "", "", 0, false + } + i = schemeEnd + 1 + for i < len(s) && s[i] != '>' { + if s[i] <= ' ' || s[i] == '<' || s[i] == '>' { + return "", "", 0, false + } + i++ + } + if i >= len(s) || i == schemeEnd+1 { + return "", "", 0, false + } + uri := string(s[1:i]) + return uri, uri, i + 1, true +} + +func isEmailByte(c byte) bool { + return isAlnum(c) || strings.IndexByte(".!#$%&'*+/=?^_`{|}~-", c) >= 0 +} + +// scanEmailDomain scans a domain of dot-separated labels, where a label +// starts and ends with an alphanumeric, may hold dashes inside, and is at +// most 63 bytes long. +func scanEmailDomain(s []byte, i int) (int, bool) { + end := 0 + for { + if i >= len(s) || !isAlnum(s[i]) { + return 0, false + } + start := i + i++ + for i < len(s) && (isAlnum(s[i]) || s[i] == '-') { + i++ + } + for i > start+1 && s[i-1] == '-' { + i-- + } + if i-start > 63 { + return 0, false + } + end = i + if i < len(s) && s[i] == '.' { + i++ + continue + } + return end, true + } +} + +// scanEntity recognises an HTML entity at pos and returns its decoded +// text, the position after it, and whether one was there. Numeric +// references follow CommonMark strictly: a malformed or out-of-range +// number is no entity at all, while null and surrogate code points decode +// to the replacement character. +func scanEntity(src []byte, pos int) (string, int, bool) { + return scanEntityAt(string(src[pos:])) +} + +func scanEntityAt(s string) (string, int, bool) { + if len(s) < 3 || s[0] != '&' { + return "", 0, false + } + if s[1] == '#' { + i := 2 + base := 10 + if i < len(s) && (s[i] == 'x' || s[i] == 'X') { + base = 16 + i++ + } + start := i + value := 0 + for i < len(s) { + d := digitValue(s[i]) + if d < 0 || d >= base { + break + } + value = value*base + d + if value > 0x10FFFF { + return "", 0, false + } + i++ + } + if i == start || i >= len(s) || s[i] != ';' { + return "", 0, false + } + i++ + r := rune(value) + if r == 0 || (r >= 0xD800 && r <= 0xDFFF) { + r = 0xFFFD + } + return string(r), i, true + } + limit := min(len(s), 33) + for i := 1; i < limit; i++ { + c := s[i] + if c == ';' { + slice := s[:i+1] + if decoded := html.UnescapeString(slice); decoded != slice { + return decoded, i + 1, true + } + return "", 0, false + } + if !isAlnum(c) { + return "", 0, false + } + } + return "", 0, false +} + +func digitValue(c byte) int { + switch { + case c >= '0' && c <= '9': + return int(c - '0') + case c >= 'a' && c <= 'f': + return int(c-'a') + 10 + case c >= 'A' && c <= 'F': + return int(c-'A') + 10 + } + return -1 +} + +// unescapeText resolves backslash escapes and entities, as link +// destinations, titles and code info strings are read. +func unescapeText(s string) string { + if !strings.ContainsAny(s, "\\&") { + return s + } + var b strings.Builder + for i := 0; i < len(s); { + c := s[i] + if c == '\\' && i+1 < len(s) && isASCIIPunct(s[i+1]) { + b.WriteByte(s[i+1]) + i += 2 + continue + } + if c == '&' { + if decoded, n, ok := scanEntityAt(s[i:]); ok { + b.WriteString(decoded) + i += n + continue + } + } + b.WriteByte(c) + i++ + } + return b.String() +} + +// htmlBlockEnds reports whether the line ends an HTML block of the given +// start condition. The sixth and seventh conditions end on a blank line, +// which the parser handles without this check. +func htmlBlockEnds(t int, line []byte) bool { + switch t { + case 1: + lower := bytes.ToLower(line) + return bytes.Contains(lower, []byte("")) || + bytes.Contains(lower, []byte("")) || + bytes.Contains(lower, []byte("")) || + bytes.Contains(lower, []byte("")) + case 2: + return bytes.Contains(line, []byte("-->")) + case 3: + return bytes.Contains(line, []byte("?>")) + case 4: + return bytes.Contains(line, []byte(">")) + case 5: + return bytes.Contains(line, []byte("]]>")) + } + return false +} + +// scanTableDelimiter recognises a table delimiter row: at least one pipe, +// and cells of dashes with optional flanking colons. It returns the +// alignment of every column, which also gives the column count. +func scanTableDelimiter(line []byte) ([]uint8, bool) { + hasPipe := false + for i := 0; i < len(line); i++ { + if line[i] == '\\' { + i++ + continue + } + if line[i] == '|' { + hasPipe = true + break + } + } + if !hasPipe { + return nil, false + } + cells := splitTableRow(line) + if len(cells) == 0 { + return nil, false + } + aligns := make([]uint8, len(cells)) + for i, cell := range cells { + a, ok := parseAlignCell(cell) + if !ok { + return nil, false + } + aligns[i] = a + } + return aligns, true +} + +// parseAlignCell reads one delimiter cell: dashes with an optional leading +// and trailing colon. +func parseAlignCell(cell []byte) (uint8, bool) { + i := 0 + left := false + if i < len(cell) && cell[i] == ':' { + left = true + i++ + } + dashes := 0 + for i < len(cell) && cell[i] == '-' { + dashes++ + i++ + } + right := false + if i < len(cell) && cell[i] == ':' { + right = true + i++ + } + if dashes == 0 || i != len(cell) { + return alignNone, false + } + switch { + case left && right: + return alignCentre, true + case left: + return alignLeft, true + case right: + return alignRight, true + } + return alignNone, true +} + +// splitTableRow splits a row into trimmed cells on unescaped pipes. The +// empty cells produced by leading and trailing boundary pipes are dropped. +// An escaped pipe resolves to a plain pipe here, before the inline parser +// runs, so a code span in a cell never shows the backslash. +func splitTableRow(line []byte) [][]byte { + trimmed := bytes.TrimSpace(line) + var cells [][]byte + var cur []byte + flush := func() { + cells = append(cells, bytes.TrimSpace(cur)) + cur = nil + } + for i := 0; i < len(trimmed); { + switch c := trimmed[i]; { + case c == '\\' && i+1 < len(trimmed) && trimmed[i+1] == '|': + cur = append(cur, '|') + i += 2 + case c == '|': + flush() + i++ + default: + cur = append(cur, c) + i++ + } + } + flush() + if len(cells) > 1 && len(cells[0]) == 0 { + cells = cells[1:] + } + if len(cells) > 1 && len(cells[len(cells)-1]) == 0 { + cells = cells[:len(cells)-1] + } + return cells +} + +// scanTaskMarker recognises a task list item marker: brackets around a +// space or an x, followed by a space and content. +func scanTaskMarker(s []byte) (checked bool, n int, ok bool) { + if len(s) < 4 || s[0] != '[' || s[2] != ']' || !isSpaceTab(s[3]) { + return false, 0, false + } + switch s[1] { + case ' ': + case 'x', 'X': + checked = true + default: + return false, 0, false + } + for i := 3; i < len(s); i++ { + if !isSpaceTab(s[i]) { + return checked, 4, true + } + } + return false, 0, false +} + +// scanFootnoteLabel reads the bracketed label of a footnote, the "[^label]" +// spelling, returning the label and the length of the whole bracket. The +// label holds no whitespace and no brackets. +func scanFootnoteLabel(s []byte) (string, int, bool) { + if len(s) < 4 || s[0] != '[' || s[1] != '^' { + return "", 0, false + } + i := 2 + for i < len(s) { + switch s[i] { + case ']': + if i == 2 { + return "", 0, false + } + return string(s[2:i]), i + 1, true + case '[', ' ', '\t', '\n': + return "", 0, false + } + i++ + } + return "", 0, false +} + +// scanFootnoteDefStart recognises a footnote definition opener, the +// "[^label]:" spelling, returning the label and the length of the marker. +func scanFootnoteDefStart(s []byte) (string, int, bool) { + label, n, ok := scanFootnoteLabel(s) + if !ok { + return "", 0, false + } + if n >= len(s) || s[n] != ':' { + return "", 0, false + } + if n+1 < len(s) && !isSpaceTab(s[n+1]) { + return "", 0, false + } + return label, n + 1, true +} + +// scanDefMarker recognises a definition list marker: a colon followed by a +// space or the end of the line. +func scanDefMarker(s []byte) bool { + return len(s) > 0 && s[0] == ':' && (len(s) == 1 || isSpaceTab(s[1])) +} diff --git a/internal/mathml/command.go b/internal/mathml/command.go new file mode 100644 index 0000000..15248a9 --- /dev/null +++ b/internal/mathml/command.go @@ -0,0 +1,570 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package mathml + +import ( + "strings" + "unicode" +) + +func isDigitByte(c byte) bool { return c >= '0' && c <= '9' } + +// isIdentifierRune reports whether the rune names a variable: a letter in +// any script. +func isIdentifierRune(r rune) bool { + return unicode.IsLetter(r) +} + +func spaceNode(width string) *node { + return &node{kind: "mspace", attrs: []attribute{{"width", width}}} +} + +// fenced wraps content in stretchy fence delimiters. +func fenced(open string, content *node, close string) *node { + row := el("mrow") + row.children = []*node{fenceNode(open), content, fenceNode(close)} + return row +} + +func fenceNode(char string) *node { + return text("mo", char, attribute{"fence", "true"}, attribute{"stretchy", "true"}) +} + +// wrapRow packs nodes into one row; an empty list becomes an empty row. +func wrapRow(nodes []*node) *node { + if len(nodes) == 1 { + return nodes[0] + } + row := el("mrow") + row.children = nodes + return row +} + +// command parses one backslash command and whatever it takes with it. The +// cursor sits on the command token. +func (p *parser) command() *node { + t := p.cur() + name := t.text[1:] + p.pos++ + // The single-character escapes: braces, the reserved characters and + // the double bar. + switch name { + case "{", "}", "%", "$", "#", "&", "_", "|", "<", ">", ",", ":", ";", "!", " ": + if w, ok := spaces[name]; ok { + return spaceNode(w) + } + if name == "|" { + return text("mo", "‖") + } + return p.withVariant(el("mo"), name) + } + switch name { + case "frac", "dfrac", "tfrac", "cfrac": + num := p.argument() + den := p.argument() + return el("mfrac", num, den) + case "binom", "dbinom", "tbinom": + num := p.argument() + den := p.argument() + return fenced("(", elA("mfrac", []attribute{{"linethickness", "0em"}}, num, den), ")") + case "genfrac": + return p.genfrac() + case "sqrt": + return p.sqrt() + case "substack": + raw, ok := p.rawBraced() + if !ok { + return errorNode(t.text) + } + return p.substack(raw) + case "text", "textrm", "textnormal", "mbox", "textmd", "textsc", "textsl": + return p.textArgument() + case "textbf", "textup": + return p.textArgument(attribute{"fontweight", "bold"}) + case "textit", "emph": + return p.textArgument(attribute{"fontstyle", "italic"}) + case "textsf": + return p.textArgument(attribute{"mathvariant", "sans-serif"}) + case "texttt": + return p.textArgument(attribute{"mathvariant", "monospace"}) + case "operatorname", "operatornamewithlimits": + raw, ok := p.rawBraced() + if !ok { + return errorNode(t.text) + } + return identifierMulti(raw, name == "operatornamewithlimits") + case "operatorname*": + raw, ok := p.rawBraced() + if !ok { + return errorNode(t.text) + } + return identifierMulti(raw, true) + case "mathchoice": + // four arguments, of which the parser renders the second, the + // text style one; the display, script and scriptscript variants + // of MathML Core are the renderer's business + _, _ = p.argument(), p.argument() + if n := p.argument(); n != nil { + _, _ = p.argument(), p.argument() + return n + } + return errorNode(t.text) + case "phantom", "hphantom", "vphantom": + arg := p.argument() + inner := el("mphantom") + inner.children = arg.children + inner.text = arg.text + return inner + case "overset", "stackrel": + over := p.argument() + base := p.argument() + return el("mover", base, over) + case "underset": + under := p.argument() + base := p.argument() + return el("munder", base, under) + case "pmod": + arg := p.argument() + row := el("mrow") + row.children = []*node{ + identifierMulti("mod", false), spaceNode("0.2778em"), arg, + } + return fenced("(", row, ")") + case "left": + return p.leftRight() + case "big", "Big", "bigg", "Bigg", + "bigl", "Bigl", "biggl", "Biggl", + "bigr", "Bigr", "biggr", "Biggr", + "bigm", "Bigm", "biggm", "Biggm": + return p.bigDelimiter() + case "middle": + return text("mo", "", attribute{"fence", "true"}, attribute{"stretchy", "true"}) + case "pod": + return fenced("(", p.argument(), ")") + case "xrightarrow", "xleftarrow", "xleftrightarrow", "xhookrightarrow", + "xhookleftarrow", "xRightarrow", "xLeftarrow", "xLeftrightarrow", + "xrightleftharpoons", "xmapsto", "xtwoheadleftarrow", "xtwoheadrightarrow", + "xleftharpoonup", "xrightharpoonup", "xleftharpoondown", "xrightharpoondown", + "xleftrightharpoons", "xtofrom", "xlongequal": + return p.xArrow(name) + case "displaystyle", "textstyle", "scriptstyle", "scriptscriptstyle": + return p.styleDeclaration(name) + case "newcommand", "renewcommand", "providecommand", "def", "gdef", "DeclareMathOperator", "DeclareMathOperator*": + return p.macroDefinition(name) + case "begin": + return p.environment() + case "end": + return errorNode(t.text) + case "limits", "nolimits": + return errorNode(t.text) + case "not": + return p.not() + case "hspace", "mspace": + return p.spaceArgument() + case "color", "textcolor": + return p.color(name) + } + if s, ok := symbols[name]; ok { + return p.symbolNode(t.text, s) + } + if functions[name] { + return identifierMulti(name, movableFunctions[name]) + } + if v, ok := styles[name]; ok { + return p.styleArgument(v) + } + if accent, ok := accents[name]; ok { + return p.accent(accent) + } + if w, ok := spaces[name]; ok { + return spaceNode(w) + } + return errorNode(t.text) +} + +// symbolNode renders a table symbol, upright under a variant switch. +func (p *parser) symbolNode(source string, s symbol) *node { + if s.mi { + return p.identifier(s.char) + } + n := p.withVariant(el("mo"), s.char) + if s.movable { + n.attrs = append(n.attrs, attribute{"movablelimits", "true"}) + } + return n +} + +// identifierMulti renders a name as an upright identifier, which a +// multi-character mi is by default. The movable flag marks the limit +// operators. +func identifierMulti(name string, movable bool) *node { + if movable { + return text("mi", name, attribute{"movablelimits", "true"}) + } + return text("mi", name) +} + +// textArgument reads a text command's argument verbatim. +func (p *parser) textArgument(attrs ...attribute) *node { + raw, ok := p.rawBraced() + if !ok { + return errorNode(`\` + "text") + } + return &node{kind: "mtext", text: raw, attrs: attrs} +} + +// styleArgument parses the argument under a forced variant. +func (p *parser) styleArgument(variant string) *node { + save := p.variant + if variant == "italic" { + p.variant = "" + } else { + p.variant = variant + } + arg := p.argument() + p.variant = save + return arg +} + +// accent puts its mark over or under the argument. +func (p *parser) accent(a struct { + char string + under bool +}) *node { + base := p.argument() + mark := text("mo", a.char, attribute{"stretchy", "false"}) + if a.under { + return el("munder", base, mark) + } + if a.char == "\u203e" || a.char == "\u23de" || a.char == "_" || a.char == "\u23e1" || a.char == "\u23df" { + mark = text("mo", a.char, attribute{"stretchy", "true"}) + } + return el("mover", base, mark) +} + +// sqrt parses a radical with its optional index. +func (p *parser) sqrt() *node { + if p.at(tokChar) && p.cur().text == "[" { + p.pos++ + var index []*node + for !p.at(tokEOF) && !(p.at(tokChar) && p.cur().text == "]") { + if n := p.atom(); n != nil { + index = append(index, n) + } + } + if p.at(tokChar) && p.cur().text == "]" { + p.pos++ + return el("mroot", p.argument(), wrapRow(index)) + } + return errorNode(`\sqrt`) + } + return el("msqrt", p.argument()) +} + +// genfrac parses the six arguments of \genfrac: the delimiters, the rule +// thickness, the style and the numerator and denominator. +func (p *parser) genfrac() *node { + open := p.delimiterArg() + close := p.delimiterArg() + thick, hasThick := p.bracketArg() + style, _ := p.bracketArg() + num := p.argument() + den := p.argument() + frac := el("mfrac") + if hasThick { + frac.attrs = append(frac.attrs, attribute{"linethickness", thick}) + } + frac.children = []*node{num, den} + if style >= "2" { + wrapped := elA("mstyle", []attribute{{"scriptlevel", "1"}}) + wrapped.children = []*node{frac} + frac = wrapped + } + if open == "" && close == "" { + return frac + } + if open == "" { + open = "." + } + if close == "" { + close = "." + } + row := el("mrow") + row.children = []*node{p.fenceOf(open), frac, p.fenceOf(close)} + return row +} + +// bracketArg reads an optional bracketed argument, reporting whether one +// was there. +func (p *parser) bracketArg() (string, bool) { + if !p.at(tokChar) || p.cur().text != "[" { + return "", false + } + p.pos++ + var b strings.Builder + for { + t := p.cur() + if t.kind == tokEOF { + return "", true + } + if t.kind == tokChar && t.text == "]" { + p.pos++ + return b.String(), true + } + b.WriteString(t.text) + p.pos++ + } +} + +// delimiterArg reads a \genfrac delimiter argument: a character, a +// command or an empty group. +func (p *parser) delimiterArg() string { + t := p.cur() + switch { + case t.kind == tokLBrace: + p.pos++ + if p.at(tokRBrace) { + p.pos++ + return "" + } + d, _ := p.delimiter() + return d + case t.kind == tokChar || t.kind == tokCommand: + d, _ := p.delimiter() + return d + } + return "" +} + +// leftRight parses a stretchy delimited row. +func (p *parser) leftRight() *node { + start := p.pos + open, ok := p.delimiter() + if !ok { + return errorNode(`\left`) + } + var content []*node + for { + t := p.cur() + if t.kind == tokEOF { + p.pos = len(p.toks) - 1 + return errorNode(string(p.src[start-1:])) + } + if t.kind == tokCommand && t.text == `\right` { + p.pos++ + break + } + if n := p.atom(); n != nil { + content = append(content, n) + } + } + close, _ := p.delimiter() + row := el("mrow") + row.children = append([]*node{p.fenceOf(open)}, content...) + row.children = append(row.children, p.fenceOf(close)) + return row +} + +// delimiter reads one delimiter: a character or a table command, with the +// dot meaning invisible. The second result reports whether a delimiter +// was there at all. +func (p *parser) delimiter() (string, bool) { + t := p.cur() + switch t.kind { + case tokChar: + p.pos++ + if t.text == "." { + return "", true + } + return t.text, true + case tokCommand: + name := t.text[1:] + if name == "|" { + p.pos++ + return "‖", true + } + if s, ok := symbols[name]; ok { + p.pos++ + return s.char, true + } + if _, ok := spaces[name]; ok { + p.pos++ + return "", true + } + } + return "", false +} + +// fenceOf makes the fence for a delimiter name; an empty name is the +// invisible fence. +func (p *parser) fenceOf(d string) *node { + if d == "" { + return text("mo", "", attribute{"fence", "true"}) + } + if d == "." { + return text("mo", "", attribute{"fence", "true"}) + } + return fenceNode(d) +} + +// bigDelimiter renders \big and its relatives around one delimiter. +func (p *parser) bigDelimiter() *node { + t := p.cur() + d, ok := p.delimiter() + if !ok { + return errorNode(t.text) + } + return text("mo", d, attribute{"stretchy", "true"}) +} + +// xArrowNames gives the shaft of every extensible arrow command. +var xArrowNames = map[string]string{ + "xrightarrow": "→", + "xleftarrow": "←", + "xleftrightarrow": "↔", + "xhookrightarrow": "↪", + "xhookleftarrow": "↩", + "xRightarrow": "⇒", + "xLeftarrow": "⇐", + "xLeftrightarrow": "⇔", + "xrightleftharpoons": "⇌", + "xmapsto": "↦", + "xtwoheadleftarrow": "↞", + "xtwoheadrightarrow": "↠", + "xleftharpoonup": "↼", + "xrightharpoonup": "⇀", + "xleftharpoondown": "↽", + "xrightharpoondown": "⇁", + "xleftrightharpoons": "⇋", + "xtofrom": "⇄", + "xlongequal": "=", +} + +// xArrow parses an extensible arrow: an optional underscript in brackets, +// then the overscript in braces. +func (p *parser) xArrow(name string) *node { + char := xArrowNames[name] + arrow := text("mo", char, attribute{"stretchy", "true"}) + var under *node + if c, ok := p.bracketArg(); ok && c != "" { + under = &node{kind: "mrow", text: c} + } + over := p.argument() + if under == nil { + return el("mover", arrow, over) + } + return el("munderover", arrow, under, over) +} + +// styleDeclaration wraps the rest of the current group in an mstyle. +func (p *parser) styleDeclaration(name string) *node { + rest := p.sequence(false) + switch name { + case "displaystyle": + n := elA("mstyle", []attribute{{"displaystyle", "true"}}) + n.children = rest + return n + case "textstyle": + n := elA("mstyle", []attribute{{"displaystyle", "false"}}) + n.children = rest + return n + case "scriptstyle": + n := elA("mstyle", []attribute{{"scriptlevel", "1"}}) + n.children = rest + return n + default: + n := elA("mstyle", []attribute{{"scriptlevel", "2"}}) + n.children = rest + return n + } +} + +// not combines the negation slash with the relation that follows. +func (p *parser) not() *node { + t := p.cur() + if t.kind == tokCommand { + name := t.text[1:] + if s, ok := symbols[name]; ok { + p.pos++ + return text("mo", s.char+"̸") + } + } + return text("mo", "¬") +} + +// spaceArgument reads the argument of \hspace and \mspace. +func (p *parser) spaceArgument() *node { + t := p.cur() + if t.kind == tokLBrace { + p.pos++ + var b strings.Builder + for { + c := p.cur() + if c.kind == tokEOF || c.kind == tokRBrace { + break + } + b.WriteString(c.text) + p.pos++ + } + if p.at(tokRBrace) { + p.pos++ + } + if validWidth(b.String()) { + return spaceNode(b.String()) + } + return errorNode(t.text) + } + return errorNode(t.text) +} + +// validWidth accepts a number with a CSS length unit. +func validWidth(s string) bool { + i := 0 + for i < len(s) && (isDigitByte(s[i]) || s[i] == '.' || s[i] == '-') { + i++ + } + if i == 0 { + return false + } + switch s[i:] { + case "em", "ex", "px", "pt", "cm", "mm", "in", "mu", "%": + return true + } + return false +} + +// color wraps its argument or the rest of the group in a coloured style. +func (p *parser) color(name string) *node { + c, ok := p.rawBraced() + if !ok || !validColour(c) { + return errorNode(`\` + name) + } + n := elA("mstyle", []attribute{{"mathcolor", c}}) + if name == "textcolor" { + n.children = []*node{p.argument()} + return n + } + n.children = p.sequence(false) + return n +} + +// validColour accepts the colour names and the hex forms. +func validColour(s string) bool { + switch s { + case "red", "green", "blue", "cyan", "magenta", "yellow", "black", + "white", "gray", "grey", "orange", "purple", "brown", "pink", + "olive", "violet", "teal", "navy", "darkgray", "lightgray": + return true + } + if len(s) == 7 && s[0] == '#' || len(s) == 4 && s[0] == '#' { + for i := 1; i < len(s); i++ { + c := s[i] + if !isDigitByte(c) && !(c >= 'a' && c <= 'f') && !(c >= 'A' && c <= 'F') { + return false + } + } + return true + } + return false +} diff --git a/internal/mathml/environments.go b/internal/mathml/environments.go new file mode 100644 index 0000000..2114940 --- /dev/null +++ b/internal/mathml/environments.go @@ -0,0 +1,252 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package mathml + +import "strings" + +// envSpec describes a table environment: its delimiters, its column +// alignment, whether it renders in script size, whether it takes a column +// specification and whether it takes a number argument first. +type envSpec struct { + fences [2]string + align string + scripted bool + colSpec bool + number bool +} + +var environments = map[string]envSpec{ + "matrix": {}, "pmatrix": {fences: [2]string{"(", ")"}}, + "bmatrix": {fences: [2]string{"[", "]"}}, + "Bmatrix": {fences: [2]string{"{", "}"}}, + "vmatrix": {fences: [2]string{"|", "|"}}, + "Vmatrix": {fences: [2]string{"‖", "‖"}}, + "matrix*": {}, + "pmatrix*": {fences: [2]string{"(", ")"}}, + "bmatrix*": {fences: [2]string{"[", "]"}}, + "Bmatrix*": {fences: [2]string{"{", "}"}}, + "vmatrix*": {fences: [2]string{"|", "|"}}, + "Vmatrix*": {fences: [2]string{"‖", "‖"}}, + "smallmatrix": {scripted: true}, + "cases": {fences: [2]string{"{", ""}, align: "left"}, + "rcases": {fences: [2]string{"", "}"}, align: "left"}, + "dcases": {fences: [2]string{"{", ""}, align: "left"}, + "drcases": {fences: [2]string{"", "}"}, align: "left"}, + "aligned": {align: "right left"}, + "align": {align: "right left"}, + "align*": {align: "right left"}, + "alignedat": {align: "right left", number: true}, + "alignat": {align: "right left", number: true}, + "alignat*": {align: "right left", number: true}, + "split": {align: "right left"}, + "gather": {}, "gather*": {}, + "equation": {}, "equation*": {}, + "array": {colSpec: true}, + "darray": {colSpec: true}, + "subarray": {colSpec: true, scripted: true}, +} + +// environment parses a whole \begin{name}...\end{name} construct. The +// cursor sits just after the \begin token. +func (p *parser) environment() *node { + begin := p.toks[p.pos-1] + name, ok := p.envName() + if !ok { + return errorNode(begin.text) + } + spec, supported := environments[name] + if !supported { + return p.degradeEnvironment(begin) + } + + if spec.colSpec { + raw, ok := p.rawBraced() + if !ok { + return errorNode(begin.text) + } + _, supported = parseColSpec(raw) + if !supported { + return p.degradeEnvironment(begin) + } + } else if spec.number { + if _, ok := p.rawBraced(); !ok { + return errorNode(begin.text) + } + } + rows := p.tableRows() + if !p.atCommand("end") { + p.pos = len(p.toks) - 1 + return errorNode(string(p.src[begin.start:])) + } + endStart := p.toks[p.pos].start + p.pos++ + endName, ok := p.envName() + if !ok || endName != name { + p.pos = len(p.toks) - 1 + return errorNode(string(p.src[begin.start:])) + } + _ = endStart + table := buildTable(rows, spec) + if spec.scripted { + inner := elA("mstyle", []attribute{{"scriptlevel", "1"}}) + inner.children = []*node{table} + table = inner + } + if spec.fences == [2]string{"", ""} { + return table + } + row := el("mrow") + row.children = []*node{p.fenceOf(spec.fences[0]), table, p.fenceOf(spec.fences[1])} + return row +} + +// envName reads the environment name in braces. +func (p *parser) envName() (string, bool) { + raw, ok := p.rawBraced() + if !ok || raw == "" { + return "", false + } + return raw, true +} + +// degradeEnvironment consumes a whole unsupported environment, up to and +// including its matching \end, and degrades it as its verbatim source. +func (p *parser) degradeEnvironment(begin token) *node { + depth := 1 + end := len(p.src) + i := p.pos + for ; i < len(p.toks); i++ { + t := p.toks[i] + if t.kind != tokCommand { + continue + } + switch t.text { + case `\begin`: + depth++ + case `\end`: + depth-- + if depth == 0 { + end = t.end + // include the name argument of \end + if i+2 < len(p.toks) && p.toks[i+1].kind == tokLBrace && p.toks[i+2].kind == tokRBrace { + end = p.toks[i+2].end + i += 2 + } + i++ + p.pos = i + return errorNode(string(p.src[begin.start:end])) + } + } + } + p.pos = len(p.toks) - 1 + return errorNode(string(p.src[begin.start:])) +} + +// tableRows parses the rows of a table, each row a slice of cells, until +// the \end or the end of input. +func (p *parser) tableRows() [][]*node { + var rows [][]*node + for { + var cells []*node + for { + nodes := p.sequence(true) + cell := el("mtd") + cell.children = nodes + cells = append(cells, cell) + if p.at(tokAmpersand) { + p.pos++ + continue + } + break + } + rows = append(rows, cells) + if isRowEnd(p.cur()) { + p.pos++ + p.rowSpacing() + if p.at(tokEOF) || p.atCommand("end") { + break + } + continue + } + break + } + return rows +} + +// rowSpacing skips the optional bracket after a row separator. +func (p *parser) rowSpacing() { + if p.at(tokChar) && p.cur().text == "[" { + for { + t := p.cur() + p.pos++ + if t.kind == tokEOF || t.kind == tokChar && t.text == "]" { + return + } + } + } +} + +// parseColSpec reads an array column specification: alignment letters and +// vertical rules, nothing else. +func parseColSpec(raw string) (int, bool) { + count := 0 + for _, c := range raw { + switch c { + case 'l', 'c', 'r': + count++ + case '|', ' ', '\t': + default: + return 0, false + } + } + if count == 0 { + return 0, false + } + return count, true +} + +// buildTable assembles the mtable with its alignment attributes. The +// "right left" alignment alternates over the widest row. +func buildTable(rows [][]*node, spec envSpec) *node { + table := el("mtable") + cols := 0 + for _, row := range rows { + if len(row) > cols { + cols = len(row) + } + } + switch { + case spec.align == "left" && cols > 0: + table.attrs = append(table.attrs, attribute{"columnalign", "left"}) + case spec.align == "right left" && cols > 0: + var b strings.Builder + for i := range cols { + if i > 0 { + b.WriteString(" ") + } + if i%2 == 0 { + b.WriteString("right") + } else { + b.WriteString("left") + } + } + table.attrs = append(table.attrs, attribute{"columnalign", b.String()}) + } + for _, row := range rows { + tr := el("mtr") + tr.children = row + table.children = append(table.children, tr) + } + return table +} + +// substack renders the rows of a \substack argument in script size. +func (p *parser) substack(raw string) *node { + q := newParser([]byte(raw), false) + rows := q.tableRows() + table := buildTable(rows, envSpec{}) + inner := elA("mstyle", []attribute{{"scriptlevel", "1"}}) + inner.children = []*node{table} + return inner +} diff --git a/internal/mathml/macros.go b/internal/mathml/macros.go new file mode 100644 index 0000000..17a8078 --- /dev/null +++ b/internal/mathml/macros.go @@ -0,0 +1,221 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package mathml + +import "strings" + +// macro is a user-defined command: the number of arguments its body takes +// and the body itself as tokens. +type macro struct { + args int + body []token +} + +// The depth limits one call site from nesting forever; the budget bounds +// the whole parse, so a self-splicing macro can never outrun the parser. +const ( + macroExpansionDepth = 64 + macroExpansionBudget = 10000 +) + +// expandMacros replaces the macro call at the cursor with its expanded +// body, ready for the parser to read. A call beyond the limits degrades +// to its own source. +func (p *parser) expandMacros() { + for range macroExpansionDepth { + t := p.toks[p.pos] + if t.kind != tokCommand { + return + } + if p.expansions >= macroExpansionBudget { + p.degradeAt(p.pos, t) + return + } + m, ok := p.macros[t.text[1:]] + if !ok { + return + } + j := p.pos + 1 + args := make([][]token, 0, m.args) + for range m.args { + arg, next, ok := p.argTokens(j) + if !ok { + break + } + args = append(args, arg) + j = next + } + if len(args) != m.args { + p.degradeAt(p.pos, t) + return + } + var body []token + for i := 0; i < len(m.body); i++ { + bt := m.body[i] + if bt.kind == tokChar && bt.text == "#" && i+1 < len(m.body) && + m.body[i+1].kind == tokChar && len(m.body[i+1].text) == 1 && + isDigitByte(m.body[i+1].text[0]) { + k := int(m.body[i+1].text[0] - '0') + if k >= 1 && k <= len(args) { + body = append(body, args[k-1]...) + } + i++ + continue + } + body = append(body, bt) + } + // The spliced tokens carry the call site as their position, so a + // construct that fails inside a macro degrades at the call. + for i := range body { + body[i].start = t.start + body[i].end = t.end + } + spliced := make([]token, 0, len(p.toks)-(j-p.pos)+len(body)) + spliced = append(spliced, p.toks[:p.pos]...) + spliced = append(spliced, body...) + spliced = append(spliced, p.toks[j:]...) + p.toks = spliced + p.expansions++ + } + t := p.toks[p.pos] + p.degradeAt(p.pos, t) +} + +// degradeAt replaces one token with a degraded token. +func (p *parser) degradeAt(i int, t token) { + p.toks[i] = token{kind: tokDegraded, text: t.text, start: t.start, end: t.end} +} + +// argTokens reads one macro argument from position j: a braced group with +// its braces, or a single token. +func (p *parser) argTokens(j int) ([]token, int, bool) { + if j >= len(p.toks) || p.toks[j].kind == tokEOF { + return nil, j, false + } + if p.toks[j].kind != tokLBrace { + return []token{p.toks[j]}, j + 1, true + } + depth := 0 + for k := j; k < len(p.toks); k++ { + switch p.toks[k].kind { + case tokLBrace: + depth++ + case tokRBrace: + depth-- + if depth == 0 { + group := make([]token, k+1-j) + copy(group, p.toks[j:k+1]) + return group, k + 1, true + } + case tokEOF: + return nil, j, false + } + } + return nil, j, false +} + +// macroDefinition registers a \newcommand or \def style definition and +// produces no output. The cursor sits just after the definition command. +func (p *parser) macroDefinition(kind string) *node { + source := `\` + kind + if kind == "DeclareMathOperator" || kind == "DeclareMathOperator*" { + name := p.defName() + if name == "" { + return errorNode(source) + } + body, ok := p.rawBraced() + if !ok { + return errorNode(source) + } + wrap := `\operatorname{` + body + `}` + if strings.HasSuffix(kind, "*") { + wrap = `\operatorname*{` + body + `}` + } + p.macros[name] = macro{body: tokenise([]byte(wrap))} + return nil + } + if kind == "def" || kind == "gdef" { + return p.tecDefinition(source) + } + name := p.defName() + if name == "" { + return errorNode(source) + } + args := 0 + if count, ok := p.bracketArg(); ok && count != "" { + n := 0 + for i := 0; i < len(count); i++ { + if !isDigitByte(count[i]) { + return errorNode(source) + } + n = n*10 + int(count[i]-'0') + } + if n > 9 { + return errorNode(source) + } + args = n + } + body, ok := p.rawBraced() + if !ok { + return errorNode(source) + } + p.macros[name] = macro{args: args, body: tokenise([]byte(body))} + return nil +} + +// tecDefinition registers a \def, whose parameter text names undelimited +// arguments with #1 up to #9. +func (p *parser) tecDefinition(source string) *node { + name := p.defName() + if name == "" { + return errorNode(source) + } + args := 0 + for { + t := p.cur() + if t.kind == tokChar && t.text == "#" { + p.pos++ + d := p.cur() + if d.kind != tokChar || len(d.text) != 1 || !isDigitByte(d.text[0]) { + return errorNode(source) + } + if int(d.text[0]-'0') != args+1 { + return errorNode(source) + } + args++ + p.pos++ + continue + } + break + } + body, ok := p.rawBraced() + if !ok { + return errorNode(source) + } + p.macros[name] = macro{args: args, body: tokenise([]byte(body))} + return nil +} + +// defName reads the name a definition declares: a braced command or a +// bare command. +func (p *parser) defName() string { + if p.at(tokLBrace) { + p.pos++ + if p.at(tokCommand) { + name := p.cur().text[1:] + p.pos++ + if p.at(tokRBrace) { + p.pos++ + return name + } + } + return "" + } + if p.at(tokCommand) { + name := p.cur().text[1:] + p.pos++ + return name + } + return "" +} diff --git a/internal/mathml/mathml.go b/internal/mathml/mathml.go new file mode 100644 index 0000000..774d1c5 --- /dev/null +++ b/internal/mathml/mathml.go @@ -0,0 +1,97 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +// Package mathml converts TeX mathematics to MathML Core with an engine of +// its own, built on the standard library alone. The grammar covers the +// standard command surface that maps to MathML: the complete symbol tables, +// fractions, scripts, operators with movable limits, stretchy delimiters, +// the amsmath environments and bounded macros. +// +// A construct outside the mappable surface is neither dropped nor +// mistranslated: it stays in the output as its verbatim source inside an +// merror element, so the author sees exactly what was not understood. +package mathml + +import "strings" + +// Render converts the TeX source to a MathML Core math element. The block +// form sets display="block". The same source always produces +// byte-identical output. +func Render(source []byte, display bool) []byte { + p := newParser(source, display) + body := p.parseAll() + var b strings.Builder + b.WriteString(`") + writeNodes(&b, body) + b.WriteString("") + return []byte(b.String()) +} + +// node is one element of the MathML tree. Attributes keep the order the +// parser gave them, which keeps the output deterministic. +type node struct { + kind string + text string + attrs []attribute + children []*node +} + +type attribute struct { + key string + value string +} + +func el(kind string, children ...*node) *node { + return &node{kind: kind, children: children} +} + +// elA builds an element that carries attributes. +func elA(kind string, attrs []attribute, children ...*node) *node { + return &node{kind: kind, attrs: attrs, children: children} +} + +func text(kind, text string, attrs ...attribute) *node { + return &node{kind: kind, text: text, attrs: attrs} +} + +// errorNode degrades a span of source to its verbatim text in a marked +// element. +func errorNode(source string) *node { + return &node{kind: "merror", children: []*node{{kind: "mtext", text: source}}} +} + +// writeNodes renders nodes without indentation; the output is one line, +// which keeps it a single inline unit in the surrounding HTML. +func writeNodes(b *strings.Builder, nodes []*node) { + for _, n := range nodes { + writeNode(b, n) + } +} + +func writeNode(b *strings.Builder, n *node) { + b.WriteString("<") + b.WriteString(n.kind) + for _, a := range n.attrs { + b.WriteString(" ") + b.WriteString(a.key) + b.WriteString(`="`) + escapeXML(b, a.value) + b.WriteString(`"`) + } + b.WriteString(">") + escapeXML(b, n.text) + writeNodes(b, n.children) + b.WriteString("") +} + +var xmlEscaper = strings.NewReplacer("&", "&", "<", "<", ">", ">", `"`, """) + +func escapeXML(b *strings.Builder, s string) { + xmlEscaper.WriteString(b, s) +} diff --git a/internal/mathml/mathml_test.go b/internal/mathml/mathml_test.go new file mode 100644 index 0000000..d0bd2e4 --- /dev/null +++ b/internal/mathml/mathml_test.go @@ -0,0 +1,189 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package mathml + +import ( + "bytes" + "encoding/xml" + "testing" +) + +// corpus is the project's own hand-written corpus: every case states a TeX +// input and the exact MathML Core the engine produces for it. +var corpus = []struct { + name string + input string + want string +}{ + {"single letter", "x", `x`}, + {"digits", "42", `42`}, + {"decimal", "1.5", `1.5`}, + {"sum", "a+b", `a+b`}, + {"greek", `\alpha + \beta`, `α+β`}, + {"uppercase greek", `\Gamma`, `Γ`}, + {"relation", `a \leq b`, `a≤b`}, + {"operator with limits", `\sum_{i=1}^{n} i`, + `∑i=1ni`}, + {"integral keeps side scripts", `\int_0^1 x`, + `∫01x`}, + {"fraction", `\frac{a}{b}`, `ab`}, + {"fraction shorthand", `\frac12`, `12`}, + {"binom", `\binom{n}{k}`, + `(nk)`}, + {"over infix", `a \over b`, `ab`}, + {"choose infix", `a \choose b`, + `(ab)`}, + {"sqrt", `\sqrt{x}`, `x`}, + {"root with index", `\sqrt[3]{x}`, `x3`}, + {"sub and sup", `x_1^2`, `x12`}, + {"single digit script", `x^10`, `x10`}, + {"group", `{xy}`, `xy`}, + {"left right", `\left( x \right)`, + `(x)`}, + {"left right invisible", `\left. x \right)`, + `x)`}, + {"function", `\sin x`, `sinx`}, + {"limit under", `\lim_{x \to 0}`, + `limx→0`}, + {"accent", `\hat{x}`, `x^`}, + {"vec accent", `\vec{v}`, `v→`}, + {"overline", `\overline{x}`, `x‾`}, + {"text", `\text{if}`, `if`}, + {"text keeps spaces", `\text{hello world}`, `hello world`}, + {"mathrm", `\mathrm{d}x`, `dx`}, + {"mathbb", `\mathbb{R}`, `R`}, + {"operatorname", `\operatorname{sgn}`, `sgn`}, + {"spacing", `a \, b`, `ab`}, + {"quad", `a \quad b`, `ab`}, + {"prime", `x'`, `x′`}, + {"escaped brace", `\{x\}`, `{x}`}, + {"overset", `\overset{a}{b}`, `ba`}, + {"matrix", `\begin{matrix} a & b \\ c & d \end{matrix}`, + `abcd`}, + {"pmatrix", `\begin{pmatrix} a \\ b \end{pmatrix}`, + `(ab)`}, + {"cases", `\begin{cases} a & b \\ c & d \end{cases}`, + `{abcd`}, + {"aligned", `\begin{aligned} a &= b \\ c &= d \end{aligned}`, + `a=bc=d`}, + {"array spec", `\begin{array}{c|l} a & b \end{array}`, + `ab`}, + {"macro", `\newcommand{\R}{\mathbb{R}} \R`, + `R`}, + {"macro with argument", `\newcommand{\ip}[2]{\langle #1, #2 \rangle} \ip{a}{b}`, + `⟨a,b⟩`}, + {"def", `\def\dx{\mathrm{d}x} \dx`, + `dx`}, + {"escaping in text", `\text{a < b & c}`, + `a < b & c`}, + {"unicode letter", `λ`, `λ`}, + {"pmod", `x \pmod n`, + `x(modn)`}, + {"colon relation precomposed", `f \coloneqq g`, + `f≔g`}, + {"colon relation linear", `f \coloneq g`, + `f:−g`}, + {"colon relation double", `f \Coloneqq g`, + `f∷=g`}, + {"eqqcolon", `a \eqqcolon b`, + `a≕b`}, +} + +func TestCorpus(t *testing.T) { + for _, tc := range corpus { + t.Run(tc.name, func(t *testing.T) { + got := string(Render([]byte(tc.input), false)) + want := `` + tc.want + `` + if got != want { + t.Errorf("input %q\ngot: %s\nwant: %s", tc.input, got, want) + } + }) + } +} + +// degrade holds the inputs whose constructs lie outside the mappable +// surface: nothing may disappear, everything stays as verbatim source in +// a marked element. +var degrade = []struct { + name string + input string +}{ + {"unknown command", `\tikz{x}`}, + {"unsupported environment", `\begin{tikzpicture} \draw (0,0); \end{tikzpicture}`}, + {"unclosed group", `{x`}, + {"unclosed environment", `\begin{matrix} a \end{pmatrix}`}, + {"reserved character", `a # b`}, + {"stray ampersand", `a & b`}, + {"stray row end", `a \\ b`}, + {"recursive macro", `\newcommand{\x}{\x}\x`}, + {"boxed", `\boxed{x}`}, + {"sideset", `\sideset{_a^b}{_c^d}\sum`}, + {"tag", `\tag{1} x`}, +} + +func TestDegradation(t *testing.T) { + for _, tc := range degrade { + t.Run(tc.name, func(t *testing.T) { + out := Render([]byte(tc.input), false) + if !bytes.Contains(out, []byte("")) { + t.Errorf("input %q produced no merror:\n%s", tc.input, out) + } + if !wellFormed(out) { + t.Errorf("input %q produced malformed XML:\n%s", tc.input, out) + } + }) + } +} + +func TestVerbatimSourceKept(t *testing.T) { + out := string(Render([]byte(`a + \unknowncmd b`), false)) + want := `\unknowncmd` + if !bytes.Contains([]byte(out), []byte(want)) { + t.Errorf("degraded construct lost its source:\n%s", out) + } +} + +func TestDeterministicOutput(t *testing.T) { + source := []byte(`\frac{1}{2}\sqrt[3]{x}\begin{pmatrix} a & b \\ c & d \end{pmatrix}\sum_{i=1}^{n} i`) + first := Render(source, true) + for range 5 { + if next := Render(source, true); !bytes.Equal(first, next) { + t.Fatal("Render of the same source differs between calls") + } + } +} + +// wellFormed checks that the output parses as XML. +func wellFormed(out []byte) bool { + dec := xml.NewDecoder(bytes.NewReader(out)) + for { + _, err := dec.Token() + if err != nil { + return err.Error() == "EOF" + } + } +} + +func TestAllCorpusWellFormed(t *testing.T) { + for _, tc := range corpus { + if out := Render([]byte(tc.input), false); !wellFormed(out) { + t.Errorf("input %q produced malformed XML:\n%s", tc.input, out) + } + } +} + +func TestDisplayBlock(t *testing.T) { + if got := string(Render([]byte("x"), true)); got != `x` { + t.Errorf("display form = %s", got) + } +} + +func TestMacroDepthBounded(t *testing.T) { + // A macro that doubles itself grows past any depth limit; the parser + // must degrade rather than hang or exhaust memory. + out := Render([]byte(`\newcommand{\a}{\a\a}`+"\n"+`\a`), false) + if !bytes.Contains(out, []byte("")) { + t.Errorf("unbounded macro produced no merror:\n%s", out) + } +} diff --git a/internal/mathml/parse.go b/internal/mathml/parse.go new file mode 100644 index 0000000..f86e198 --- /dev/null +++ b/internal/mathml/parse.go @@ -0,0 +1,326 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package mathml + +import "strings" + +// parser walks the token list. The variant field is the mathvariant the +// current style command imposes on the atoms it covers. +type parser struct { + src []byte + toks []token + pos int + display bool + variant string + macros map[string]macro + expansions int +} + +func newParser(src []byte, display bool) *parser { + return &parser{src: src, toks: tokenise(src), display: display, macros: map[string]macro{}} +} + +func (p *parser) cur() token { + p.expandMacros() + return p.toks[p.pos] +} + +func (p *parser) at(kind tokenKind) bool { return p.cur().kind == kind } + +func (p *parser) atCommand(name string) bool { + t := p.cur() + return t.kind == tokCommand && t.text == `\`+name +} + +// parseAll parses the whole source into nodes. A boundary with nothing to +// bound degrades as the stray it is. +func (p *parser) parseAll() []*node { + nodes := p.sequence(true) + for p.cur().kind != tokEOF { + if isBoundary(p.cur()) { + t := p.cur() + p.pos++ + nodes = append(nodes, errorNode(t.text)) + continue + } + nodes = append(nodes, p.sequence(true)...) + } + return nodes +} + +// isBoundary reports whether the token ends a sequence: a cell separator, +// a row separator, an environment close or a closing brace. +func isBoundary(t token) bool { + if t.kind == tokRBrace || t.kind == tokAmpersand || isRowEnd(t) { + return true + } + return t.kind == tokCommand && t.text == `\end` +} + +func isRowEnd(t token) bool { + return t.kind == tokCommand && t.text == `\\` +} + +// sequence parses atoms until a boundary or the end. When allowOver is +// set, the TeX infix constructs \over, \atop and \choose may appear and +// take everything parsed so far as their numerator. +func (p *parser) sequence(allowOver bool) []*node { + var nodes []*node + for { + t := p.cur() + if t.kind == tokEOF || isBoundary(t) { + break + } + if t.kind == tokDegraded { + nodes = append(nodes, errorNode(t.text)) + p.pos++ + continue + } + if t.kind == tokCommand && isOverCommand(t.text) { + p.pos++ + if !allowOver { + nodes = append(nodes, errorNode(t.text)) + continue + } + num := wrapRow(nodes) + den := wrapRow(p.sequence(false)) + nodes = []*node{p.overNode(t.text, num, den)} + continue + } + if n := p.atom(); n != nil { + nodes = append(nodes, n) + } + } + return nodes +} + +func isOverCommand(text string) bool { + switch text { + case `\over`, `\atop`, `\choose`, `\overwithdelims`, `\atopwithdelims`, `\abovewithdelims`: + return true + } + return false +} + +func (p *parser) overNode(cmd string, num, den *node) *node { + switch cmd { + case `\choose`: + return fenced("(", elA("mfrac", []attribute{{"linethickness", "0em"}}, num, den), ")") + case `\atop`: + return elA("mfrac", []attribute{{"linethickness", "0em"}}, num, den) + default: + return el("mfrac", num, den) + } +} + +// atom parses one unit with its scripts, or a whole command construct. +func (p *parser) atom() *node { + t := p.cur() + switch t.kind { + case tokLBrace: + return p.braceGroup() + case tokChar: + return p.scripts(p.charAtom(t)) + case tokCommand: + n := p.command() + if n == nil { + return nil + } + return p.scripts(n) + } + p.pos++ + return errorNode(t.text) +} + +// braceGroup parses a braced group, degrading the whole span when the +// closing brace never comes. A group of one node is that node: the braces +// only grouped. +func (p *parser) braceGroup() *node { + open := p.cur() + p.pos++ + nodes := p.sequence(true) + if p.at(tokRBrace) { + p.pos++ + if len(nodes) == 1 { + return nodes[0] + } + if len(nodes) == 0 { + return el("mrow") + } + row := el("mrow") + row.children = nodes + return row + } + p.pos = len(p.toks) - 1 + return errorNode(string(p.src[open.start:])) +} + +// charAtom maps one character token to its element. The reserved TeX +// characters degrade rather than pass as content. +func (p *parser) charAtom(t token) *node { + c := t.text + switch c { + case "#", "$", "%": + p.pos++ + return errorNode(c) + case "~": + p.pos++ + return spaceNode("0.25em") + case "'": + p.pos++ + return text("mo", "′") + } + r := []rune(c)[0] + switch { + case isDigitByte(c[0]) && len(c) == 1: + p.pos++ + return p.numberRun(c) + case isIdentifierRune(r): + p.pos++ + return p.identifier(c) + default: + p.pos++ + return p.withVariant(el("mo"), c) + } +} + +// numberRun collects the digits and decimal points that follow. +func (p *parser) numberRun(first string) *node { + var b strings.Builder + b.WriteString(first) + for { + t := p.cur() + if t.kind != tokChar || len(t.text) != 1 || !isDigitByte(t.text[0]) && t.text != "." { + break + } + b.WriteString(t.text) + p.pos++ + } + return p.withVariant(el("mn"), b.String()) +} + +// identifier renders one letter, upright when the variant says so. +func (p *parser) identifier(c string) *node { + return p.withVariant(el("mi"), c) +} + +// withVariant fills a leaf node with text, applying the active variant. +func (p *parser) withVariant(n *node, c string) *node { + n.text = c + if p.variant != "" && (n.kind == "mi" || n.kind == "mn" || n.kind == "mo") { + if !(n.kind == "mi" && p.variant == "italic") { + n.attrs = append(n.attrs, attribute{"mathvariant", p.variant}) + } + } + return n +} + +// scripts attaches the sub and superscript runs that follow a base. A +// movable base puts its scripts under and over in display style; inline, +// the movablelimits attribute lets the renderer decide. +func (p *parser) scripts(base *node) *node { + movable := baseMovable(base) + if p.atCommand("limits") { + p.pos++ + movable = true + } else if p.atCommand("nolimits") { + p.pos++ + movable = false + } + var sub, sup *node + for { + t := p.cur() + if t.kind == tokUnderscore && sub == nil { + p.pos++ + sub = p.argument() + continue + } + if t.kind == tokCaret && sup == nil { + p.pos++ + sup = p.argument() + continue + } + break + } + under := movable && p.display + switch { + case sub == nil && sup == nil: + return base + case under && sub != nil && sup != nil: + return el("munderover", base, sub, sup) + case under && sub != nil: + return el("munder", base, sub) + case under && sup != nil: + return el("mover", base, sup) + case sub != nil && sup != nil: + return el("msubsup", base, sub, sup) + case sub != nil: + return el("msub", base, sub) + default: + return el("msup", base, sup) + } +} + +// baseMovable reports whether the base carries movable limits. +func baseMovable(base *node) bool { + if base.kind != "mo" { + return false + } + for _, a := range base.attrs { + if a.key == "movablelimits" { + return a.value == "true" + } + } + return false +} + +// argument reads one macro argument or script: a braced group or a single +// token. A single digit stays one digit, the way TeX reads x^10. +func (p *parser) argument() *node { + t := p.cur() + switch { + case t.kind == tokLBrace: + return p.braceGroup() + case t.kind == tokChar && len(t.text) == 1 && isDigitByte(t.text[0]): + p.pos++ + return p.withVariant(el("mn"), t.text) + case t.kind == tokChar || t.kind == tokCommand: + n := p.atom() + if n == nil { + return el("mrow") + } + return n + } + return errorNode(t.text) +} + +// rawBraced reads a braced group from the source as verbatim text, +// keeping the spaces. It fails when the closing brace is missing. +func (p *parser) rawBraced() (string, bool) { + t := p.cur() + if t.kind != tokLBrace { + return "", false + } + depth := 0 + for i := t.start; i < len(p.src); { + switch p.src[i] { + case '{': + depth++ + case '}': + depth-- + if depth == 0 { + text := string(p.src[t.start+1 : i]) + // consume the tokens the span covers + for p.pos < len(p.toks) && p.toks[p.pos].start <= i { + p.pos++ + } + return text, true + } + case '\\': + i++ // an escaped character never opens or closes + } + i++ + } + return "", false +} diff --git a/internal/mathml/symbols.go b/internal/mathml/symbols.go new file mode 100644 index 0000000..d649d88 --- /dev/null +++ b/internal/mathml/symbols.go @@ -0,0 +1,349 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package mathml + +// symbol names the MathML element a bare command becomes and whether its +// scripts sit under and over it. +type symbol struct { + char string + mi bool + movable bool +} + +func op(char string) symbol { return symbol{char: char} } +func mov(char string) symbol { return symbol{char: char, movable: true} } +func id(char string) symbol { return symbol{char: char, mi: true} } + +// symbols holds the commands that stand for one character: Greek letters, +// letterlike symbols, binary operators, relations, arrows, delimiters, big +// operators and miscellaneous symbols. +var symbols = map[string]symbol{ + // Greek letters, lowercase. + "alpha": id("α"), "beta": id("β"), "gamma": id("γ"), "delta": id("δ"), + "epsilon": id("ϵ"), "varepsilon": id("ε"), "zeta": id("ζ"), "eta": id("η"), + "theta": id("θ"), "vartheta": id("ϑ"), "iota": id("ι"), "kappa": id("κ"), + "lambda": id("λ"), "mu": id("μ"), "nu": id("ν"), "xi": id("ξ"), + "omicron": id("ο"), "pi": id("π"), "varpi": id("ϖ"), "rho": id("ρ"), + "varrho": id("ϱ"), "sigma": id("σ"), "varsigma": id("ς"), "tau": id("τ"), + "upsilon": id("υ"), "phi": id("ϕ"), "varphi": id("φ"), "chi": id("χ"), + "psi": id("ψ"), "omega": id("ω"), "digamma": id("ϝ"), + // Greek letters, uppercase; the ones that coincide with Latin capitals + // still stand as identifiers, and the var forms are the italic capitals. + "Gamma": id("Γ"), "Delta": id("Δ"), "Theta": id("Θ"), "Lambda": id("Λ"), + "Xi": id("Ξ"), "Pi": id("Π"), "Sigma": id("Σ"), "Upsilon": id("Υ"), + "Phi": id("Φ"), "Psi": id("Ψ"), "Omega": id("Ω"), + "Alpha": id("Α"), "Beta": id("Β"), "Epsilon": id("Ε"), "Zeta": id("Ζ"), + "Eta": id("Η"), "Iota": id("Ι"), "Kappa": id("Κ"), "Mu": id("Μ"), + "Nu": id("Ν"), "Omicron": id("Ο"), "Rho": id("Ρ"), "Tau": id("Τ"), + "Chi": id("Χ"), + "varGamma": id("𝛤"), "varDelta": id("𝛥"), "varTheta": id("𝛩"), + "varLambda": id("𝛬"), "varXi": id("𝛯"), "varPi": id("𝛱"), + "varSigma": id("𝛴"), "varUpsilon": id("𝛶"), "varPhi": id("𝛷"), + "varPsi": id("𝛹"), "varOmega": id("𝛺"), + "varkappa": id("ϰ"), "thetasym": id("ϑ"), + // Letterlike symbols. + "hbar": id("ℏ"), "ell": id("ℓ"), "imath": id("ı"), "jmath": id("ȷ"), + "wp": id("℘"), "Re": id("ℜ"), "Im": id("ℑ"), "aleph": id("ℵ"), + "beth": id("ℶ"), "gimel": id("ℷ"), "daleth": id("ℸ"), + "partial": id("∂"), "nabla": op("∇"), "mho": id("℧"), + "complement": op("∁"), "eth": id("ð"), "Finv": id("⅁"), "Game": id("⅂"), + "circledS": id("Ⓢ"), "Bbbk": id("𝕜"), "weierp": id("℘"), + // Big operators, with movable limits. + "sum": mov("∑"), "prod": mov("∏"), "coprod": mov("∐"), + "int": op("∫"), "iint": op("∬"), "iiint": op("∭"), "iiiint": op("⨌"), + "oint": op("∮"), "oiint": op("∯"), "oiiint": op("∰"), + "bigcap": mov("⋂"), "bigcup": mov("⋃"), "bigsqcup": mov("⨆"), + "bigvee": mov("⋁"), "bigwedge": mov("⋀"), "bigodot": mov("⨀"), + "bigotimes": mov("⨂"), "bigoplus": mov("⨁"), "biguplus": mov("⨄"), + "varointclockwise": op("∱"), "ointclockwise": op("∲"), + "ointctrclockwise": op("∳"), + // Binary operators. + "pm": op("±"), "mp": op("∓"), "times": op("×"), "div": op("÷"), + "ast": op("∗"), "star": op("⋆"), "circ": op("∘"), "bullet": op("∙"), + "cdot": op("⋅"), "cap": op("∩"), "cup": op("∪"), "uplus": op("⊎"), + "sqcap": op("⊓"), "sqcup": op("⊔"), "vee": op("∨"), "lor": op("∨"), + "wedge": op("∧"), "land": op("∧"), "setminus": op("∖"), + "wr": op("≀"), "diamond": op("⋄"), "bigtriangleup": op("△"), + "bigtriangledown": op("▽"), "triangleleft": op("◃"), "triangleright": op("▹"), + "lhd": op("⊲"), "rhd": op("⊳"), "unlhd": op("⊴"), "unrhd": op("⊵"), + "oplus": op("⊕"), "ominus": op("⊖"), "otimes": op("⊗"), "oslash": op("⊘"), + "odot": op("⊙"), "bigcirc": op("○"), "dagger": op("†"), "ddagger": op("‡"), + "amalg": op("⨿"), "dotplus": op("∔"), "smallsetminus": op("∖"), + "Cap": op("⋒"), "Cup": op("⋓"), "barwedge": op("⊼"), "veebar": op("⊻"), + "doublebarwedge": op("⩞"), "boxminus": op("⊟"), "boxplus": op("⊞"), + "boxtimes": op("⊠"), "boxdot": op("⊡"), "divideontimes": op("⋇"), + "intercal": op("⊺"), "circledcirc": op("⊚"), "circledast": op("⊛"), + "circleddash": op("⊝"), "curlywedge": op("⋏"), "curlyvee": op("⋎"), + "leftthreetimes": op("⋋"), "rightthreetimes": op("⋌"), + "looparrowleft": op("↫"), "looparrowright": op("↬"), + "curvearrowleft": op("↶"), "curvearrowright": op("↷"), + "circlearrowleft": op("↺"), "circlearrowright": op("↻"), + // Relations. + "leq": op("≤"), "le": op("≤"), "geq": op("≥"), "ge": op("≥"), + "neq": op("≠"), "ne": op("≠"), "sim": op("∼"), "simeq": op("≃"), + "approx": op("≈"), "cong": op("≅"), "equiv": op("≡"), + "prec": op("≺"), "preceq": op("⪯"), "precapprox": op("⪵"), + "precsim": op("≾"), "succ": op("≻"), "succeq": op("⪰"), + "succapprox": op("⪶"), "succsim": op("≿"), + "ll": op("≪"), "lll": op("⋘"), "gg": op("≫"), "ggg": op("⋙"), + "asymp": op("≍"), "doteq": op("≐"), "propto": op("∝"), + "mid": op("∣"), "nmid": op("∤"), "parallel": op("∥"), "shortparallel": op("∥"), + "perp": op("⊥"), "Subset": op("⋐"), "Supset": op("⋑"), + "sqsubset": op("⊏"), "sqsupset": op("⊐"), + "subset": op("⊂"), "supset": op("⊃"), "subseteq": op("⊆"), "supseteq": op("⊇"), + "subseteqq": op("⫅"), "supseteqq": op("⫆"), + "sqsubseteq": op("⊑"), "sqsupseteq": op("⊒"), + "in": op("∈"), "ni": op("∋"), "owns": op("∋"), "notin": op("∉"), + "vdash": op("⊢"), "dashv": op("⊣"), "Vdash": op("⊩"), "Vvdash": op("⊪"), + "models": op("⊨"), "smile": op("⌣"), "frown": op("⌢"), + "lesssim": op("≲"), "gtrsim": op("≳"), "lessapprox": op("⪅"), + "gtrapprox": op("⪆"), "lessgtr": op("≶"), "gtrless": op("≷"), + "lesseqgtr": op("⋚"), "gtreqless": op("⋛"), "leqq": op("≦"), + "geqq": op("≧"), "lneq": op("⪇"), "gneq": op("⪈"), "lvertneqq": op("≨"), + "gvertneqq": op("≩"), "lnsim": op("⪦"), "gnsim": op("⪧"), + "eqslantless": op("⪕"), "eqslantgtr": op("⪖"), "backsim": op("∽"), + "backsimeq": op("⋍"), "lesseqqgtr": op("⪋"), "gtreqqless": op("⪌"), + "nestedlessgreater": op("≺"), "nless": op("≮"), "ngtr": op("≯"), + "nleq": op("≰"), "nleqslant": op("≰"), "ngeq": op("≱"), "ngeqslant": op("≱"), + "nprec": op("⊀"), "nsucc": op("⊁"), "precnsim": op("⋨"), "succnsim": op("⋩"), + "nsubseteq": op("⊈"), "nsupseteq": op("⊉"), "subsetneq": op("⊊"), + "supsetneq": op("⊋"), "subsetneqq": op("⫋"), "supsetneqq": op("⫌"), + "vartriangleleft": op("⊲"), "vartriangleright": op("⊳"), + "trianglelefteq": op("⊴"), "trianglerighteq": op("⊵"), + "triangleq": op("≜"), "bumpeq": op("≏"), "Bumpeq": op("≎"), + "eqcirc": op("≖"), "circeq": op("≗"), "doteqdot": op("≑"), + "risingdotseq": op("≓"), "fallingdotseq": op("≒"), + "pitchfork": op("⋔"), "smallfrown": op("⌢"), "smallsmile": op("⌣"), + "therefore": op("∴"), "because": op("∵"), + "eqsim": op("≟"), + "bowtie": op("⋈"), "Join": op("⋈"), "backepsilon": op("϶"), + "thicksim": op("∼"), "thickapprox": op("≈"), + "preccurlyeq": op("≼"), "succcurlyeq": op("≽"), + "varpropto": op("∝"), "ratio": op("∶"), "vcentcolon": op(":"), + "curlyeqprec": op("⋞"), "curlyeqsucc": op("⋟"), + "between": op("≬"), + // The colon relations, mapped the way KaTeX's MathML branch maps + // them: the precomposed character where Unicode has one, the linear + // two-character operator where it has none. + "dblcolon": op("∷"), + "coloneqq": op("≔"), + "coloneq": op(":−"), + "Coloneqq": op("∷="), + "Coloneq": op("∷−"), + "eqqcolon": op("≕"), + "eqcolon": op("∹"), + "Eqqcolon": op("=∷"), + "Eqcolon": op("−∷"), + "colonapprox": op(":≈"), + "Colonapprox": op("∷≈"), + "colonsim": op(":∼"), + "Colonsim": op("∷∼"), + "approxcolon": op("≈:"), + "approxcoloncolon": op("≈∷"), + "simcolon": op("∼:"), + "simcoloncolon": op("∼∷"), + "origof": op("⊶"), + "imageof": op("⊷"), + // The last stragglers the full KaTeX symbol table carries. + "Doteq": op("≑"), + "Diamond": op("◆"), + "approxeq": op("≊"), + "doublecap": op("⋒"), + "doublecup": op("⋓"), + "geqslant": op("⩾"), + "leqslant": op("⩽"), + "gggtr": op("⋙"), + "llless": op("⋘"), + "gneqq": op("≩"), + "lneqq": op("≨"), + "gtrdot": op("⋗"), + "lessdot": op("⋖"), + "intop": op("∫"), + "smallint": op("∫"), + "ltimes": op("⋉"), + "rtimes": op("⋊"), + "nparallel": op("∦"), + "nsim": op("≁"), + "nvdash": op("⊬"), + "vDash": op("⊨"), + "shortmid": op("∣"), + "varvdots": op("⋮"), + // Negated relations. + "ncong": op("≇"), "npreceq": op("⋠"), + "nsucceq": op("⋡"), + "precnapprox": op("⪹"), "succnapprox": op("⪺"), + "precneqq": op("⪵"), "succneqq": op("⪶"), + "gnapprox": op("⪊"), "lnapprox": op("⪉"), + "nshortmid": op("∤"), "nshortparallel": op("∦"), + "nvDash": op("⊭"), "nVDash": op("⊯"), "nVdash": op("⊮"), + "ntriangleleft": op("⋪"), "ntriangleright": op("⋫"), + "ntrianglelefteq": op("⋬"), "ntrianglerighteq": op("⋭"), + "nleftrightarrow": op("↮"), "nLeftarrow": op("⇍"), + "nLeftrightarrow": op("⇎"), "nRightarrow": op("⇏"), + "nleftarrow": op("↰"), "nrightarrow": op("↱"), + // Arrows. + "leftarrow": op("←"), "gets": op("←"), "Leftarrow": op("⇐"), + "rightarrow": op("→"), "to": op("→"), "Rightarrow": op("⇒"), + "leftrightarrow": op("↔"), "Leftrightarrow": op("⇔"), "iff": op("⟺"), + "longleftarrow": op("⟵"), "Longleftarrow": op("⟸"), + "longrightarrow": op("⟶"), "Longrightarrow": op("⟹"), + "longleftrightarrow": op("⟷"), "Longleftrightarrow": op("⟺"), + "implies": op("⟹"), "impliedby": op("⟸"), + "mapsto": op("↦"), "longmapsto": op("⟼"), + "hookleftarrow": op("↩"), "hookrightarrow": op("↪"), + "leftharpoonup": op("↼"), "leftharpoondown": op("↽"), + "rightharpoonup": op("⇀"), "rightharpoondown": op("⇁"), + "rightleftharpoons": op("⇌"), "leadsto": op("↝"), + "nearrow": op("↗"), "searrow": op("↘"), "swarrow": op("↙"), "nwarrow": op("↖"), + "uparrow": op("↑"), "downarrow": op("↓"), "updownarrow": op("↕"), + "Uparrow": op("⇑"), "Downarrow": op("⇓"), "Updownarrow": op("⇕"), + "downdownarrows": op("⇊"), "upuparrows": op("⇈"), + "rightrightarrows": op("⇉"), "leftleftarrows": op("⇇"), + "rightleftarrows": op("⇄"), "leftrightarrows": op("⇆"), + "twoheadrightarrow": op("↠"), "twoheadleftarrow": op("↞"), + "leftarrowtail": op("↢"), "rightarrowtail": op("↣"), + "Lleftarrow": op("⤅"), "Rrightarrow": op("⤇"), + "upharpoonleft": op("↿"), "upharpoonright": op("↾"), + "downharpoonleft": op("⇃"), "downharpoonright": op("⇂"), + "restriction": op("↾"), "multimap": op("⊸"), + "harr": op("↔"), "hArr": op("⇔"), "Harr": op("⇔"), + "larr": op("←"), "lArr": op("⇐"), "Larr": op("⇐"), + "rarr": op("→"), "rArr": op("⇒"), "Rarr": op("⇒"), + "lrarr": op("↔"), "lrArr": op("⇔"), "Lrarr": op("⇔"), + "darr": op("↓"), "dArr": op("⇓"), "Darr": op("⇓"), + "uarr": op("↑"), "uArr": op("⇑"), "Uarr": op("⇑"), + "dashleftarrow": op("⇠"), "dashrightarrow": op("⇢"), + "rightsquigarrow": op("↝"), "leftrightsquigarrow": op("↭"), + "leftrightharpoons": op("⇋"), "mapsfrom": op("↤"), + "Lsh": op("↰"), "Rsh": op("↱"), + // Delimiters. + "langle": op("⟨"), "rangle": op("⟩"), "lfloor": op("⌊"), "rfloor": op("⌋"), + "lceil": op("⌈"), "rceil": op("⌉"), "vert": op("|"), "lvert": op("|"), + "rvert": op("|"), "Vert": op("‖"), "lVert": op("‖"), "rVert": op("‖"), + "lbrace": op("{"), "rbrace": op("}"), "lbrack": op("["), "rbrack": op("]"), + "lgroup": op("⟮"), "rgroup": op("⟯"), + "lmoustache": op("⌠"), "rmoustache": op("⌡"), "backslash": op("\\"), + "lparen": op("("), "rparen": op(")"), "lang": op("⟨"), "rang": op("⟩"), + "ulcorner": op("⌜"), "urcorner": op("⌝"), "llcorner": op("⌞"), + "lrcorner": op("⌟"), "llbracket": op("⟦"), "rrbracket": op("⟧"), + "lBrace": op("{"), "rBrace": op("}"), + // Miscellaneous symbols. + "infty": op("∞"), "forall": op("∀"), "exists": op("∃"), "nexists": op("∄"), + "emptyset": op("∅"), "varnothing": op("∅"), "top": op("⊤"), "bot": op("⊥"), + "vdots": op("⋮"), "cdots": op("⋯"), "ddots": op("⋱"), "iddots": op("⋰"), + "ldots": op("…"), "dots": op("…"), "dotsc": op("…"), "dotsb": op("⋯"), + "dotsm": op("⋯"), "dotsi": op("⋯"), "dotso": op("…"), + "prime": op("′"), "backprime": op("‵"), "degree": op("°"), + "angle": op("∠"), "measuredangle": op("∡"), "sphericalangle": op("∢"), + "triangle": op("△"), "square": op("□"), "blacksquare": op("■"), + "bigstar": op("★"), "blacktriangle": op("▲"), "blacktriangledown": op("▼"), + "blacktriangleleft": op("◀"), "blacktriangleright": op("▶"), + "diamondsuit": op("♦"), "heartsuit": op("♥"), "clubsuit": op("♣"), + "spadesuit": op("♠"), "flat": op("♭"), "natural": op("♮"), "sharp": op("♯"), + "checkmark": op("✓"), "maltese": op("✠"), "bull": op("∙"), + "ldotp": op("."), "cdotp": op("⋅"), "colon": op(":"), + "S": op("§"), "P": op("¶"), "copyright": op("©"), "circledR": op("®"), + "diagup": op("╱"), "diagdown": op("╲"), + "lozenge": op("◊"), "blacklozenge": op("◆"), "surd": op("√"), + "Box": op("□"), "triangledown": op("▽"), "vartriangle": op("△"), + "pounds": op("£"), "mathsterling": op("£"), "yen": op("¥"), + "dag": op("†"), "ddag": op("‡"), "Dagger": op("‡"), + "minuso": op("⦵"), "centerdot": op("·"), "plusmn": op("±"), + "And": op("&"), "lq": op("‘"), "rq": op("’"), + "sdot": op("⋅"), "mathellipsis": op("…"), + "neg": op("¬"), "lnot": op("¬"), "empty": op("∅"), + "isin": op("∈"), "exist": op("∃"), + "lt": op("<"), "gt": op(">"), + // Letter-like aliases KaTeX carries. + "alef": id("ℵ"), "alefsym": id("ℵ"), "hslash": id("ℏ"), + "image": id("ℑ"), "real": id("ℜ"), "reals": id("ℝ"), + "cnums": id("ℂ"), "Complex": id("ℂ"), "natnums": id("ℕ"), + "RR": id("ℝ"), "NN": id("ℕ"), "ZZ": id("ℤ"), "Q": id("ℚ"), + "infin": op("∞"), +} + +// functions are the names typeset upright as identifiers. +var functions = map[string]bool{ + "arccos": true, "arcsin": true, "arctan": true, "arg": true, + "cos": true, "cosh": true, "cot": true, "coth": true, "csc": true, + "deg": true, "det": true, "dim": true, "exp": true, "gcd": true, + "hom": true, "ker": true, "lg": true, "ln": true, "log": true, + "Pr": true, "sec": true, "sin": true, "sinh": true, "tan": true, + "tanh": true, "arcsinh": true, "arccosh": true, "arctanh": true, + "argmax": true, "argmin": true, + "mod": true, "bmod": true, + "min": true, "max": true, "sup": true, "inf": true, + "lim": true, "limsup": true, "liminf": true, + "injlim": true, "projlim": true, "varinjlim": true, "varprojlim": true, + "varliminf": true, "varlimsup": true, "plim": true, + "arctg": true, "arcctg": true, "ch": true, "cosec": true, "cotg": true, + "ctg": true, "cth": true, "sh": true, "tg": true, "th": true, +} + +// movableFunctions take their scripts under and over: the limit operators. +var movableFunctions = map[string]bool{ + "lim": true, "limsup": true, "liminf": true, "max": true, "min": true, + "sup": true, "inf": true, "gcd": true, "det": true, "Pr": true, + "injlim": true, "projlim": true, "varinjlim": true, "varprojlim": true, + "varliminf": true, "varlimsup": true, "plim": true, + "argmax": true, "argmin": true, +} + +// accents put a mark over or under their argument. +var accents = map[string]struct { + char string + under bool +}{ + "hat": {"\u005e", false}, "widehat": {"\u005e", false}, + "tilde": {"~", false}, "widetilde": {"~", false}, + "utilde": {"~", true}, + "bar": {"\u00af", false}, "overline": {"\u203e", false}, + "vec": {"\u2192", false}, "dot": {"\u02d9", false}, "ddot": {"\u00a8", false}, + "dddot": {"\u20db", false}, "ddddot": {"\u20dc", false}, + "mathring": {"\u02da", false}, "breve": {"\u02d8", false}, + "check": {"\u02c7", false}, "widecheck": {"\u02c7", false}, + "acute": {"\u00b4", false}, "grave": {"\u0060", false}, + "overbrace": {"\u23de", false}, "underbrace": {"\u23df", true}, + "overbracket": {"\u23b4", false}, "underbracket": {"\u23b5", true}, + "overleftarrow": {"\u2190", false}, "overrightarrow": {"\u2192", false}, + "Overrightarrow": {"\u21d2", false}, + "underleftarrow": {"\u2190", true}, "underrightarrow": {"\u2192", true}, + "overleftrightarrow": {"\u2194", false}, "underleftrightarrow": {"\u2194", true}, + "overleftharpoon": {"\u21bc", false}, "overrightharpoon": {"\u21c0", false}, + "overgroup": {"\u23e0", false}, "undergroup": {"\u23e1", true}, + "underline": {"_", true}, "underbar": {"\u02cd", true}, + "overlinesegment": {"\u23af", false}, "underlinesegment": {"\u23af", true}, +} + +// styles map the style commands to mathvariant values; the empty value +// marks a switch that renders its argument unchanged. +var styles = map[string]string{ + "mathrm": "normal", "mathnormal": "italic", "mathit": "italic", + "mathbf": "bold", "mathbfit": "bold-italic", "mathbb": "double-struck", + "mathcal": "script", "mathscr": "script", "mathfrak": "fraktur", + "mathsf": "sans-serif", "mathsfit": "sans-serif-italic", + "mathsfbf": "sans-serif-bold", "mathtt": "monospace", + "boldsymbol": "bold-italic", "bm": "bold-italic", + // The old TeX switches, applied to what follows in the group. + "rm": "normal", "bf": "bold", "it": "italic", "sf": "sans-serif", + "tt": "monospace", "cal": "script", "scr": "script", "frak": "fraktur", +} + +// spaces maps spacing commands to mspace widths; the empty width carries +// nothing. +var spaces = map[string]string{ + ",": "0.1667em", "thinspace": "0.1667em", + ":": "0.2222em", "medspace": "0.2222em", + ";": "0.2778em", "thickspace": "0.2778em", + "!": "-0.1667em", "negthinspace": "-0.1667em", + "negmedspace": "-0.2222em", "negthickspace": "-0.2778em", + " ": "0.25em", "quad": "1em", "qquad": "2em", "enspace": "0.5em", +} + +// delimiterChars are the single characters accepted after \left, \right +// and the big size commands. +var delimiterChars = map[string]bool{ + "(": true, ")": true, "[": true, "]": true, "|": true, "/": true, + "<": true, ">": true, +} diff --git a/internal/mathml/token.go b/internal/mathml/token.go new file mode 100644 index 0000000..54d2c9a --- /dev/null +++ b/internal/mathml/token.go @@ -0,0 +1,85 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package mathml + +import "unicode/utf8" + +type tokenKind uint8 + +const ( + tokEOF tokenKind = iota + tokCommand + tokChar + tokLBrace + tokRBrace + tokCaret + tokUnderscore + tokAmpersand + // tokDegraded stands for a span the parser gave up on: the parser + // renders it as its verbatim source. + tokDegraded +) + +type token struct { + kind tokenKind + text string + start int + end int +} + +// tokenise splits source into TeX tokens. Whitespace between tokens is +// dropped: math mode ignores it, and the text commands read their argument +// from the raw source instead. +func tokenise(src []byte) []token { + var toks []token + i := 0 + for i < len(src) { + c := src[i] + start := i + switch { + case c == '\\' && i+1 < len(src): + i++ + if isLetter(src[i]) { + for i < len(src) && isLetter(src[i]) { + i++ + } + text := string(src[start:i]) + // A control word eats the spaces behind it, without them + // becoming part of its name. + for i < len(src) && (src[i] == ' ' || src[i] == '\t' || src[i] == '\n') { + i++ + } + toks = append(toks, token{kind: tokCommand, text: text, start: start, end: i}) + } else { + i++ + toks = append(toks, token{kind: tokCommand, text: string(src[start:i]), start: start, end: i}) + } + case c == '{': + i++ + toks = append(toks, token{kind: tokLBrace, text: "{", start: start, end: i}) + case c == '}': + i++ + toks = append(toks, token{kind: tokRBrace, text: "}", start: start, end: i}) + case c == '^': + i++ + toks = append(toks, token{kind: tokCaret, text: "^", start: start, end: i}) + case c == '_': + i++ + toks = append(toks, token{kind: tokUnderscore, text: "_", start: start, end: i}) + case c == '&': + i++ + toks = append(toks, token{kind: tokAmpersand, text: "&", start: start, end: i}) + case c == ' ' || c == '\t' || c == '\n' || c == '\r': + i++ + default: + _, size := utf8.DecodeRune(src[i:]) + i += size + toks = append(toks, token{kind: tokChar, text: string(src[start:i]), start: start, end: i}) + } + } + toks = append(toks, token{kind: tokEOF, start: len(src), end: len(src)}) + return toks +} + +func isLetter(c byte) bool { return c >= 'a' && c <= 'z' || c >= 'A' && c <= 'Z' } diff --git a/justfile b/justfile new file mode 100644 index 0000000..2fbfaff --- /dev/null +++ b/justfile @@ -0,0 +1,75 @@ +# scriptorium + +# What the test and bench recipes sweep. Never name a directory the +# project does not have: a pattern that matches nothing is a setup +# failure, not an empty run. +packages := "./..." + +# The memory fence for the test recipes: a cgroup ceiling with swap off, so a +# runaway run dies as a failed run and never eats the machine. 4G is the +# default; raise it only with a reason recorded here. +memlimit := "4G" + +default: + @just --list + +# Compile every package. Zero errors, zero warnings. +build: + go build ./... + +# The test gate: the suite, no cache, the coverage floor, under the memory fence. +test: + #!/usr/bin/env perl + my @fence = (q{systemd-run}, q{--user}, q{--scope}, + q{-p}, q{MemoryMax={{memlimit}}}, q{-p}, q{MemorySwapMax=0}); + system(@fence, q{go}, q{test}, q{-count=1}, q{-timeout}, q{10m}, + q{-coverprofile}, q{coverage.out}, qw({{packages}})) == 0 + or die qq{the test suite failed\n}; + open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!}; + my $total; + while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} } + close($c); + die qq{no total line in coverage.out\n} unless defined $total; + printf qq{Total coverage: %s%%\n}, $total; + exit($total < 80 ? 1 : 0); + +# The same suite under the race detector. The expensive one, still fenced. +race: + systemd-run --user --scope -p MemoryMax={{memlimit}} -p MemorySwapMax=0 go test -race -count=1 -timeout 10m {{packages}} + +# Fast scoped run for iterating. This is the one that runs after every edit. +unit pkgs=packages run=".*": + systemd-run --user --scope -p MemoryMax={{memlimit}} -p MemorySwapMax=0 go test {{pkgs}} -run '{{run}}' + +# Time-boxed fuzz of one target in one package. The package is required; never a gate. +fuzz target pkg fuzztime="60s": + systemd-run --user --scope -p MemoryMax={{memlimit}} -p MemorySwapMax=0 go test -run '^$' -fuzz '{{target}}' -fuzztime={{fuzztime}} {{pkg}} + +# Benchmarks. On an idle machine only, unfenced: a ceiling would distort the measurement. +bench pkgs=packages: + go test -run '^$' -bench=. -benchmem -count=5 {{pkgs}} + +# Format in place. +fmt: + gofmt -w . + +# Zero diff. Prints nothing when everything is formatted. +fmt-check: + #!/usr/bin/env perl + open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!}; + my @bad = <$g>; + close($g); + print @bad; + exit(@bad ? 1 : 0); + +# Both static gates: go vet and go fix -diff. +vet: + go vet ./... + go fix -diff ./... + +# The definition of done, in one command. Once per task, never per edit. +gates: build fmt-check vet test race + +# Build artefacts only, not an installed binary. +clean: + rm -f coverage.out diff --git a/scriptorium.go b/scriptorium.go new file mode 100644 index 0000000..4be8b87 --- /dev/null +++ b/scriptorium.go @@ -0,0 +1,69 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +// Package scriptorium renders scientific documents server-side. It renders +// Markdown to HTML, TeX mathematics to MathML Core, and Mermaid flowchart +// and sequence diagrams to SVG. The output is deterministic, so the same +// input always produces byte-identical output. +// +// The library is built on the Go standard library alone: it carries no +// dependency of its own. It renders trusted input and applies no +// sanitisation; whether the produced HTML may reach a given audience is +// the consumer's policy, not the renderer's. +// +// A construct the renderer refuses to guess at is never dropped or +// mistranslated: it stays visible as its verbatim source in a marked +// element, so the author sees exactly what was not understood. +package scriptorium + +import ( + "sourcedock.dev/petrbalvin/scriptorium/internal/diagram" + "sourcedock.dev/petrbalvin/scriptorium/internal/markdown" + "sourcedock.dev/petrbalvin/scriptorium/internal/mathml" +) + +// Render renders the Markdown source to HTML. The grammar is the full +// CommonMark specification with the GitHub Flavored Markdown extensions, +// footnotes and definition lists. Rendering never fails and never +// sanitises: the input is trusted, and whether the output may reach an +// audience is the consumer's policy. +func Render(source []byte) []byte { + return markdown.RenderHTML(source) +} + +// RenderMath renders the TeX mathematics source to a MathML Core math +// element, in the inline form. A construct outside the mappable surface +// stays in the output as its verbatim source inside an merror element. +func RenderMath(source []byte) []byte { + return mathml.Render(source, false) +} + +// RenderMathDisplay renders the TeX mathematics source to a MathML Core +// math element in the display form, with display="block". +func RenderMathDisplay(source []byte) []byte { + return mathml.Render(source, true) +} + +// RenderDiagram renders the Mermaid diagram source to SVG. The diagram +// type must be one SupportedDiagram reports; every other type is refused +// with an error naming it. +func RenderDiagram(source []byte) ([]byte, error) { + return diagram.Render(source) +} + +// The Mermaid scope: every diagram type the library accepts. The choice +// is measured, not aesthetic: across the documentation of this forge the +// diagrams are 44 flowcharts (including the historical "graph" spelling) +// and 22 sequence diagrams, with a single outlying state diagram. Every +// other type is refused with a clear error rather than a guess. +var supportedDiagrams = map[string]bool{ + "flowchart": true, + "graph": true, // the historical alias of flowchart + "sequenceDiagram": true, +} + +// SupportedDiagram reports whether the library renders the named +// Mermaid diagram type. +func SupportedDiagram(kind string) bool { + return supportedDiagrams[kind] +} diff --git a/scriptorium_test.go b/scriptorium_test.go new file mode 100644 index 0000000..df5dbeb --- /dev/null +++ b/scriptorium_test.go @@ -0,0 +1,88 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package scriptorium + +import ( + "bytes" + "testing" +) + +func TestSupportedDiagram(t *testing.T) { + supported := []string{"flowchart", "graph", "sequenceDiagram"} + for _, kind := range supported { + if !SupportedDiagram(kind) { + t.Errorf("SupportedDiagram(%q) = false, want true", kind) + } + } + refused := []string{"", "stateDiagram-v2", "pie", "gantt", "classDiagram", "Flowchart"} + for _, kind := range refused { + if SupportedDiagram(kind) { + t.Errorf("SupportedDiagram(%q) = true, want false", kind) + } + } +} + +func TestRender(t *testing.T) { + got := string(Render([]byte("# Title\n\nHello *world*.\n"))) + want := "

    Title

    \n

    Hello world.

    \n" + if got != want { + t.Errorf("Render = %q, want %q", got, want) + } +} + +func TestRenderDeterministic(t *testing.T) { + source := []byte("- one\n- two\n\n```go\nfmt.Println(1)\n```\n") + first := Render(source) + for range 5 { + if next := Render(source); !bytes.Equal(first, next) { + t.Fatal("Render of the same source differs between calls") + } + } +} + +func TestRenderMath(t *testing.T) { + got := string(RenderMath([]byte(`\frac{1}{2}`))) + want := `12` + if got != want { + t.Errorf("RenderMath = %s, want %s", got, want) + } +} + +func TestRenderMathDisplay(t *testing.T) { + got := string(RenderMathDisplay([]byte("x"))) + want := `x` + if got != want { + t.Errorf("RenderMathDisplay = %s, want %s", got, want) + } +} + +func TestRenderDiagram(t *testing.T) { + out, err := RenderDiagram([]byte("flowchart TD\n A[Start] --> B[End]\n")) + if err != nil { + t.Fatalf("render: %v", err) + } + if !bytes.HasPrefix(out, []byte(`>B: hello\n B-->>A: bye\n") + first, err := RenderDiagram(source) + if err != nil { + t.Fatalf("render: %v", err) + } + for range 5 { + next, err := RenderDiagram(source) + if err != nil || !bytes.Equal(first, next) { + t.Fatal("RenderDiagram of the same source differs between calls") + } + } +}