diff --git a/AGENTS.md b/AGENTS.md index f29daa4..f2a449b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -298,10 +298,17 @@ someone spells it or pins it. has — for the fallback to refuse at zero rather than walk again; counting every list twice halved the list limit, counting the directive form once doubled the parser's frames per level (the maintainer, 2026-09-18). +- Nothing spreads an unbounded array into a call — a node's siblings, a code block's held lines, a + mark run's segments: the argument list caps near 125k and throws a `RangeError` where a `Result` + is owed. A walk pushes one at a time. A literal spread (`[...value]`) is not the same thing and + is fine (4c). - A reader takes the text and an index — a sticky regex whose `lastIndex` the caller sets on the line before it reads, `indexOf` — never a fresh slice per character, and a per-character walk hoists the scan that does not vary with the character. The pipeline persona feeds documents - nobody typed, and a megabyte through a quadratic walk is a minute rather than a millisecond. + nobody typed, and a megabyte through a quadratic walk is a minute rather than a millisecond. A + scan may keep what it read for a later walk of the same text, and the fallback where it kept + nothing must be the same reader over the same text at the same index, so the two cannot disagree + — which is what makes the kept value a memo rather than a second spelling (4c). - No casts: `as`, `as unknown as`, non-null `!`. A boundary owes a type guard validating the fields it claims (`isAdfDocument`); past it everything is typed. Make invalid states unrepresentable. diff --git a/src/markdown/commonmark-grammar.ts b/src/markdown/commonmark-grammar.ts index 9047d0f..53c9f1a 100644 --- a/src/markdown/commonmark-grammar.ts +++ b/src/markdown/commonmark-grammar.ts @@ -174,7 +174,6 @@ export function isThematicBreak(line: string): boolean { return holdsThematicBreak(thematicBreakTail(line), line) } -// A break runs to the line's end, so one scan of that run answers every level the list walk opens. export function thematicBreakTail(text: string): ThematicBreakTail | undefined { let marker: string | undefined let markers = 0 @@ -195,6 +194,8 @@ export function thematicBreakTail(text: string): ThematicBreakTail | undefined { return { longest: text.length - first, marker, shortest: text.length - third } } +// `text` is a suffix of what `tail` was read from, or opens with a space: a length alone names a +// suffix, and the marker check is what refuses one starting mid-run — `--- ---` holds ` ---`. export function holdsThematicBreak(tail: ThematicBreakTail | undefined, text: string): boolean { if (tail === undefined) return false return text.length >= tail.shortest && text.length <= tail.longest && text.charAt(0) === tail.marker diff --git a/src/markdown/directive-syntax.ts b/src/markdown/directive-syntax.ts index 41e7fcb..5ae76af 100644 --- a/src/markdown/directive-syntax.ts +++ b/src/markdown/directive-syntax.ts @@ -17,7 +17,7 @@ export type DirectiveLine = | { argument: string | undefined; attributes: DirectiveAttributes; kind: 'opener'; name: string } | { kind: 'closer'; name: string } -// spans: the directives the content holds, at their offset in it, so reading the slot back never scans them again. +// spans keys index content, so the two travel together: separate them and every offset is wrong. export type DirectiveSpan = { attributes: DirectiveAttributes; content: string | undefined; length: number; name: string; spans: NestedSpans } export type NestedSpans = ReadonlyMap @@ -254,10 +254,11 @@ function readDirectiveContent(text: string, start: number, depth: number): Read< cursor = span continue } - const nested = keepNestedSpan(text, cursor, depth, start, spans) + const nested = readNestedDirective(text, cursor, depth + 1) if (nested?.fault !== undefined) return { fault: nested.fault } if (nested !== undefined) { - cursor = nested.value + spans.set(cursor - start, nested.value) + cursor += nested.value.length continue } if (character === ']' && brackets === 0) return { value: { end: cursor, spans } } @@ -268,14 +269,6 @@ function readDirectiveContent(text: string, start: number, depth: number): Read< return { fault: malformedDirective(`an inline directive [content] is unclosed; ${directiveEscape}`) } } -function keepNestedSpan(text: string, cursor: number, depth: number, start: number, spans: Map): Read | undefined { - const nested = readNestedDirective(text, cursor, depth + 1) - if (nested === undefined) return undefined - if (nested.fault !== undefined) return { fault: nested.fault } - spans.set(cursor - start, nested.value) - return { value: cursor + nested.value.length } -} - // `undefined` where the span crosses the line ending an inline directive may not cross. function readCodeSpanEnd(text: string, index: number): number | undefined { const opener = backtickRun(text, index) diff --git a/src/markdown/emit/adf-to-markdown.test.ts b/src/markdown/emit/adf-to-markdown.test.ts index ca07e0a..e5a2969 100644 --- a/src/markdown/emit/adf-to-markdown.test.ts +++ b/src/markdown/emit/adf-to-markdown.test.ts @@ -736,3 +736,8 @@ test('carries whitespace CommonMark strips in the reserved text directive', () = // CommonMark strips spaces and tabs alone, so the whitespace beside them is plain text. assert.equal(emitted({ text: '\va\f', type: 'text' }), '\va\f\n') }) + +test("joins a mark run's segments as a walk rather than as one call's arguments", () => { + const run = Array.from({ length: 200000 }, (): AdfNode => ({ marks: [{ type: 'strong' }], text: 'a', type: 'text' })) + assert.equal(markdown(adfToMarkdown(document({ content: run, type: 'paragraph' }))), `**${'a'.repeat(200000)}**\n`) +}) diff --git a/src/markdown/emit/inline-line.ts b/src/markdown/emit/inline-line.ts index 77311f5..0cddd44 100644 --- a/src/markdown/emit/inline-line.ts +++ b/src/markdown/emit/inline-line.ts @@ -166,7 +166,7 @@ function emitRun(nodes: readonly AdfNode[], depth: number, firstIndex: number, c const emitted = run.kind === 'plain' ? emitLeaf(run.node, runContext, run.index) : emitMarkedRun(run.nodes, run.mark, depth, run.index, runContext) if (!emitted.ok) return emitted if (emitted.value.carry !== undefined) return emitted - segments.push(...emitted.value.segments) + for (const segment of emitted.value.segments) segments.push(segment) } return success({ segments }) } diff --git a/src/markdown/parse/blocks.test.ts b/src/markdown/parse/blocks.test.ts index f4061be..2866051 100644 --- a/src/markdown/parse/blocks.test.ts +++ b/src/markdown/parse/blocks.test.ts @@ -118,3 +118,7 @@ test('closes the containers a closer names past as unclosed, and crosses no list assert.deepEqual(kinds('!adf:rule {localId=a-1}\nPart.\n!adf:/rule\n'), ['fault', 'paragraph']) assert.deepEqual(kinds('!adf:rule {localId=a-1}\n!adf:/rule\n!adf:/rule\n'), ['fault', 'fault']) }) + +test("releases an indented code block's held blank lines as a walk rather than as one call's arguments", () => { + assert.deepEqual(kinds(` a\n${'\n'.repeat(200000)} b\n`), ['code']) +}) diff --git a/src/markdown/parse/blocks.ts b/src/markdown/parse/blocks.ts index 09edb33..b011452 100644 --- a/src/markdown/parse/blocks.ts +++ b/src/markdown/parse/blocks.ts @@ -365,7 +365,8 @@ function readIndentedCodeLine(leaf: Extract return true } if (leadingColumns(line) < indentedCodeColumns) return false - leaf.lines.push(...leaf.held, removeColumns(line, indentedCodeColumns).text) + for (const held of leaf.held) leaf.lines.push(held) + leaf.lines.push(removeColumns(line, indentedCodeColumns).text) leaf.held.length = 0 return true } diff --git a/todo-history.md b/todo-history.md index 65ed0be..07ab2b7 100644 --- a/todo-history.md +++ b/todo-history.md @@ -534,6 +534,11 @@ The done `todo.md` items in full, as they were written. `todo.md` keeps a one-li 38 ms where the tail scan alone would have left it quadratic. The sixth site the sweep found went to 18 rather than landing here (the maintainer, 2026-09-18). + **Widened** (the systems-architect, 2026-09-18): the guard's spread was a class rather than a + site, and two more threw out of the public API — `readIndentedCodeLine` releasing the blank + lines an indented code block held (200k of them at 200 kB), and `emitRun` joining a mark + run's segments (200k nodes under one mark). Both fixed here with the same loop and a test + each, and §11 gained the rule so the spelling cannot walk back in. - [x] **5a — Rename to `@larvit/adf-codec` (`0.1.0`).** Before the first publish, the name being the published identity: `package.json` `name` and `repository`, the Gitea repo and its remote, the README title, §6's published-as line, the checkout directory.