diff --git a/AGENTS.md b/AGENTS.md index 96dbcba..c132b13 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -313,6 +313,15 @@ someone spells it or pins it. scan may keep what it read for a later walk of the same text, and the fallback where it kept nothing must be the same reader over the same text at the same index, so the two cannot disagree — which is what makes the kept value a memo rather than a second spelling (4c). +- The parse keeps each node's readable spelling in a memo, so the `commonMarkSpelling` ask stops + spelling a node once per level above it (18). The node reference is the key, which holds + because the parse builds one object per position; `adfToMarkdown` passes no memo, where a + consumer's document may hold one node at two positions (4b). `text` and `spelling` carry no depth + and `headroom` is affine in it, so a read at or above the depth that filled the entry rebases; a + read below re-spells, because a hit skips the depth guards the walk it replaces runs and an + ordered list past the marker cap gives way, spending two emitter levels where the parser spent + one. A give-way is kept too and serves any depth, reading the node's shape alone. Only what + succeeded is kept, so no path minted at another position is ever read. - No casts: `as`, `as unknown as`, non-null `!`. A boundary owes a type guard validating the fields it claims (`isAdfDocument`); past it everything is typed. Make invalid states unrepresentable. diff --git a/src/markdown/emit/adf-to-markdown.ts b/src/markdown/emit/adf-to-markdown.ts index 00190ab..6eb0da3 100644 --- a/src/markdown/emit/adf-to-markdown.ts +++ b/src/markdown/emit/adf-to-markdown.ts @@ -19,7 +19,9 @@ import { tryPipeTable } from './pipe-table.ts' type BlockContainer = 'directive' | 'document' | 'list-item' type BlockSpelling = 'commonmark' | 'directive' | 'list' type EmittedBlock = { headroom: number; spelling: BlockSpelling; text: string } +type KeptSpelling = { block: EmittedBlock | undefined; depth: number } type PlacedBlock = EmittedBlock & { node: AdfNode } +export type SpellingMemo = Map type Walk = { blocks: readonly PlacedBlock[]; headroom: number } type WalkedItem = { node: AdfNode; walk: Walk } @@ -31,19 +33,19 @@ export function adfToMarkdown(document: AdfDocument): Result { const fault = adfDocumentFault(document) if (fault !== undefined) return faulted(fault, []) if (document.version !== 1) return failure('unsupported-document-version', `no markdown spelling carries ADF version ${document.version}`, []) - const walk = walkBlocks(nodeContent(document), [], 0) + const walk = walkBlocks(nodeContent(document), [], 0, undefined) if (!walk.ok) return walk const text = joinBlocks(walk.value.blocks, 'document') return success(text === '' ? '' : `${text}\n`) } // headroom: the least slack any depth guard below the walk has. -function walkBlocks(nodes: readonly AdfNode[], path: ConvertErrorPath, depth: number): Result { +function walkBlocks(nodes: readonly AdfNode[], path: ConvertErrorPath, depth: number, memo: SpellingMemo | undefined): Result { let headroom = largestNesting - depth if (headroom < 0) return tooDeep(path) const blocks: PlacedBlock[] = [] for (const [index, node] of nodes.entries()) { - const block = emitBlock(node, [...path, 'content', index], depth) + const block = emitBlock(node, [...path, 'content', index], depth, memo) if (!block.ok) return block headroom = Math.min(headroom, block.value.headroom) blocks.push({ ...block.value, node }) @@ -84,24 +86,37 @@ function interruptsParagraph(node: AdfNode): boolean { return markerInterruptsParagraph(listStart(node, items.length) ?? 0, empty) } -function emitBlock(node: AdfNode, path: ConvertErrorPath, depth: number): Result { +function emitBlock(node: AdfNode, path: ConvertErrorPath, depth: number, memo: SpellingMemo | undefined): Result { const directive = blockDirective(node.type) if (directive === undefined) return commonMarkLine(carriedBlock(node, path, depth)) - const readable = readableBlock(node, path, depth) + const readable = readableBlock(node, path, depth, memo) if (readable !== undefined) return readable - return emitDirectiveBlock(node, directive, path, depth, () => walkBlocks(nodeContent(node), path, depth + 1)) + return emitDirectiveBlock(node, directive, path, depth, () => walkBlocks(nodeContent(node), path, depth + 1, memo)) } -export function commonMarkSpelling(node: AdfNode, path: ConvertErrorPath, depth: number): Result | undefined { - const readable = readableBlock(node, path, depth) +export function commonMarkSpelling(node: AdfNode, path: ConvertErrorPath, depth: number, memo: SpellingMemo): Result | undefined { + const readable = readableBlock(node, path, depth, memo) if (readable === undefined) return undefined if (!readable.ok) return readable return readable.value.spelling === 'directive' ? undefined : success(null) } -function readableBlock(node: AdfNode, path: ConvertErrorPath, depth: number): Result | undefined { - if (node.type === 'blockquote') return emitBlockquote(node, path, depth) - if (node.type === 'bulletList' || node.type === 'orderedList') return emitList(node, path, depth) +function readableBlock(node: AdfNode, path: ConvertErrorPath, depth: number, memo: SpellingMemo | undefined): Result | undefined { + const kept = memo?.get(node) + if (kept !== undefined) { + if (kept.block === undefined) return undefined + // A read below the fill would skip the depth guards the walk it replaces runs (AGENTS.md §11). + if (depth <= kept.depth) return success({ ...kept.block, headroom: kept.block.headroom + kept.depth - depth }) + } + const spelled = spellReadableBlock(node, path, depth, memo) + if (spelled === undefined) memo?.set(node, { block: undefined, depth }) + else if (spelled.ok) memo?.set(node, { block: spelled.value, depth }) + return spelled +} + +function spellReadableBlock(node: AdfNode, path: ConvertErrorPath, depth: number, memo: SpellingMemo | undefined): Result | undefined { + if (node.type === 'blockquote') return emitBlockquote(node, path, depth, memo) + if (node.type === 'bulletList' || node.type === 'orderedList') return emitList(node, path, depth, memo) if (node.type === 'codeBlock') return emitCodeBlock(node, path) if (node.type === 'heading') return emitHeading(node, path) if (node.type === 'mediaSingle') return readableText(tryImage(node, path)) @@ -149,9 +164,9 @@ function emitDirectiveBody(node: AdfNode, directive: BlockDirective, opener: str return success(directivePair(node, opener, joinBlocks(walk.value.blocks, 'directive'), walk.value.headroom)) } -function emitBlockquote(node: AdfNode, path: ConvertErrorPath, depth: number): Result | undefined { +function emitBlockquote(node: AdfNode, path: ConvertErrorPath, depth: number, memo: SpellingMemo | undefined): Result | undefined { if (!carriesOnly(node, [])) return undefined - const inner = walkBlocks(nodeContent(node), path, depth + 1) + const inner = walkBlocks(nodeContent(node), path, depth + 1, memo) if (!inner.ok) return inner const text = joinBlocks(inner.value.blocks, 'document') .split('\n') @@ -211,7 +226,7 @@ function emitHeading(node: AdfNode, path: ConvertErrorPath): Result | undefined { +function emitList(node: AdfNode, path: ConvertErrorPath, depth: number, memo: SpellingMemo | undefined): Result | undefined { const ordered = node.type === 'orderedList' if (!carriesOnly(node, ordered ? ['order'] : [])) return undefined const items = nodeContent(node) @@ -221,7 +236,7 @@ function emitList(node: AdfNode, path: ConvertErrorPath, depth: number): Result< const walked: WalkedItem[] = [] let headroom = Number.POSITIVE_INFINITY for (const [offset, item] of items.entries()) { - const walk = walkBlocks(nodeContent(item), [...path, 'content', offset], depth + 1) + const walk = walkBlocks(nodeContent(item), [...path, 'content', offset], depth + 1, memo) if (!walk.ok) return walk headroom = Math.min(headroom, walk.value.headroom) walked.push({ node: item, walk: walk.value }) diff --git a/src/markdown/parse/markdown-to-adf.test.ts b/src/markdown/parse/markdown-to-adf.test.ts index 32da72d..283be1e 100644 --- a/src/markdown/parse/markdown-to-adf.test.ts +++ b/src/markdown/parse/markdown-to-adf.test.ts @@ -672,7 +672,7 @@ test('refuses input nested deeper than the parser carries', () => { assert.equal(code(markdownToAdf(marks(largestNesting + 1))), 'unsupported-nesting-depth') assert.deepEqual(content(markdownToAdf(marks(largestNesting))), [{ content: [marked('a', underline)], type: 'paragraph' }]) const nest = (names: readonly string[], body: string): string => [...names.map((name) => `!adf:${name}\n`), body, ...names.map((name) => `!adf:/${name}\n`).reverse()].join('') - const repeated = (name: string): string[] => Array.from({ length: largestNesting }, () => name) + const repeated = (name: string, levels: number = largestNesting): string[] => Array.from({ length: levels }, () => name) assert.ok(markdownToAdf(nest(repeated('panel'), '!adf:paragraph {localId=a-1}\nPart.\n!adf:/paragraph\n')).ok) assert.deepEqual(position(markdownToAdf(nest(['expand', ...repeated('panel'), 'expand'], 'Part.\n'))), { line: 501, offset: 5501 }) assert.equal(code(markdownToAdf(nest(['panel', ...repeated('expand'), 'panel'], 'Part.\n'))), 'unsupported-nesting-depth') @@ -680,6 +680,19 @@ test('refuses input nested deeper than the parser carries', () => { const directiveLists = largestNesting / 2 assert.ok(markdownToAdf(listed(directiveLists)).ok) assert.equal(code(markdownToAdf(listed(directiveLists + 1))), 'unsupported-nesting-depth') + const asking = (body: string): string => `!adf:bulletList\n!adf:listItem\n${body}!adf:/listItem\n!adf:/bulletList\n` + // Two items past the largest list marker leave the list no readable spelling, so the emitter + // walks it as list plus item where the parser counted one level. + const overflowing = (body: string): string => { + const marker = '999999999. ' + return `${marker}${body.replace(/^(?!$)/gm, ' '.repeat(marker.length)).slice(marker.length)}\n${marker}z\n` + } + const overflowed = (panels: number): string => asking(overflowing(overflowing(asking(`---\n${nest(repeated('panel', panels), 'Part.\n')}`)))) + assert.equal(code(markdownToAdf(overflowed(493))), 'unsupported-node-shape') + assert.equal(code(markdownToAdf(overflowed(494))), 'unsupported-nesting-depth') + const rebasing = (panels: number): string => asking(`---\n${overflowing(asking(`---\n${nest(repeated('panel', panels), 'Part.\n')}`))}`) + assert.ok(markdownToAdf(rebasing(494)).ok) + assert.equal(code(markdownToAdf(rebasing(495))), 'unsupported-nesting-depth') assert.ok(markdownToAdf(`${'- '.repeat(largestNesting)}a\n`).ok) assert.equal(code(markdownToAdf(`${'- '.repeat(largestNesting + 1)}a\n`)), 'unsupported-nesting-depth') }) diff --git a/src/markdown/parse/markdown-to-adf.ts b/src/markdown/parse/markdown-to-adf.ts index 2c2425b..20eb3a5 100644 --- a/src/markdown/parse/markdown-to-adf.ts +++ b/src/markdown/parse/markdown-to-adf.ts @@ -5,7 +5,7 @@ import type { ConvertFault } from '../../result.ts' import type { LineContainer } from '../emit/line-escaping.ts' import type { LinkDefinitions } from './inline-content.ts' import { carryName, readCarriedBlock } from '../opaque-carry.ts' -import { commonMarkSpelling } from '../emit/adf-to-markdown.ts' +import { commonMarkSpelling, type SpellingMemo } from '../emit/adf-to-markdown.ts' import { failure, faulted, positioned, success, type ConvertErrorPath, type ParseError, type Result, type SourcePosition } from '../../result.ts' import { languageSlot } from '../code-language.ts' import { largestNesting } from '../../nesting.ts' @@ -20,12 +20,12 @@ const documentStart: SourcePosition = { line: 1, offset: 0 } export function markdownToAdf(markdown: string): Result { const parsed = parseBlocks(markdown) - const content = positioned(blockNodes(parsed.blocks, parsed.definitions, [], 0), documentStart) + const content = positioned(blockNodes(parsed.blocks, parsed.definitions, [], 0, new Map()), documentStart) if (!content.ok) return content return success(content.value.length === 0 ? { type: 'doc', version: 1 } : { content: content.value, type: 'doc', version: 1 }) } -function blockNodes(blocks: readonly Block[], definitions: LinkDefinitions, path: ConvertErrorPath, depth: number): Result { +function blockNodes(blocks: readonly Block[], definitions: LinkDefinitions, path: ConvertErrorPath, depth: number, memo: SpellingMemo): Result { if (depth > largestNesting) return failure('unsupported-nesting-depth', `the input nests deeper than the ${largestNesting} levels the parser carries`, path) const content: AdfNode[] = [] for (const [index, block] of blocks.entries()) { @@ -35,7 +35,7 @@ function blockNodes(blocks: readonly Block[], definitions: LinkDefinitions, path if (fault !== undefined) return positioned(faulted(fault, nodePath), block.position) continue } - const node = positioned(blockNode(block, definitions, nodePath, depth), block.position) + const node = positioned(blockNode(block, definitions, nodePath, depth, memo), block.position) if (!node.ok) return node content.push(node.value) } @@ -55,16 +55,16 @@ function partsFault(): ConvertFault { return unsupportedNodeShape(`${listBreakName} parts two adjacent lists of one type: this one parts something else`) } -function blockNode(block: Block, definitions: LinkDefinitions, path: ConvertErrorPath, depth: number): Result { +function blockNode(block: Block, definitions: LinkDefinitions, path: ConvertErrorPath, depth: number, memo: SpellingMemo): Result { switch (block.kind) { case 'blockquote': - return containerNode({ type: 'blockquote' }, block.blocks, definitions, path, depth) + return containerNode({ type: 'blockquote' }, block.blocks, definitions, path, depth, memo) case 'bulletList': - return listNode({ type: 'bulletList' }, block.items, definitions, path, depth) + return listNode({ type: 'bulletList' }, block.items, definitions, path, depth, memo) case 'code': return codeBlockNode(block.language, block.text, path, depth) case 'directive': - return directiveNode(block, definitions, path, depth) + return directiveNode(block, definitions, path, depth, memo) case 'fault': return faulted(block.fault, path) case 'heading': @@ -72,7 +72,7 @@ function blockNode(block: Block, definitions: LinkDefinitions, path: ConvertErro case 'html': return failure('unmappable-html', `no raw HTML converts at this version: ${block.construct}`, path) case 'orderedList': - return listNode({ attrs: { order: block.start }, type: 'orderedList' }, block.items, definitions, path, depth) + return listNode({ attrs: { order: block.start }, type: 'orderedList' }, block.items, definitions, path, depth, memo) case 'paragraph': return paragraphNode(block.text, definitions, path) case 'rule': @@ -82,23 +82,23 @@ function blockNode(block: Block, definitions: LinkDefinitions, path: ConvertErro } } -function directiveNode(block: DirectiveBlock, definitions: LinkDefinitions, path: ConvertErrorPath, depth: number): Result { +function directiveNode(block: DirectiveBlock, definitions: LinkDefinitions, path: ConvertErrorPath, depth: number, memo: SpellingMemo): Result { const read = readBlockDirectiveNode(block.name, block.argument, block.attributes, path) if (!read.ok) return read - const built = directiveBody(read.value, block.blocks, definitions, path, depth) + const built = directiveBody(read.value, block.blocks, definitions, path, depth, memo) if (!built.ok) return built - const readable = commonMarkSpelling(built.value, path, depth) + const readable = commonMarkSpelling(built.value, path, depth, memo) if (readable === undefined) return built if (!readable.ok) return readable return failure('unsupported-node-shape', `${built.value.type} takes the CommonMark spelling, not the directive form`, path) } -function directiveBody(read: BlockDirectiveNode, blocks: Block[] | undefined, definitions: LinkDefinitions, path: ConvertErrorPath, depth: number): Result { +function directiveBody(read: BlockDirectiveNode, blocks: Block[] | undefined, definitions: LinkDefinitions, path: ConvertErrorPath, depth: number, memo: SpellingMemo): Result { const { contentModel, node } = read if (blocks === undefined) return success(node) if (contentModel === 'code') return codeDirectiveNode(node, blocks, path) if (contentModel === 'inline') return inlineBodyNode(node, blocks, definitions, path) - return containerNode(node, blocks, definitions, path, depth) + return containerNode(node, blocks, definitions, path, depth, memo) } function codeDirectiveNode(node: AdfNode, blocks: readonly Block[], path: ConvertErrorPath): Result { @@ -137,8 +137,8 @@ function inlineBodyNode(node: AdfNode, blocks: readonly Block[], definitions: Li return positioned(contentNode(node, only.text, definitions, path, 'paragraph'), only.position) } -function containerNode(node: AdfNode, blocks: readonly Block[], definitions: LinkDefinitions, path: ConvertErrorPath, depth: number): Result { - const content = blockNodes(blocks, definitions, path, depth + 1) +function containerNode(node: AdfNode, blocks: readonly Block[], definitions: LinkDefinitions, path: ConvertErrorPath, depth: number, memo: SpellingMemo): Result { + const content = blockNodes(blocks, definitions, path, depth + 1, memo) if (!content.ok) return content return success(withContent(node, content.value)) } @@ -147,10 +147,10 @@ function withContent(node: AdfNode, content: readonly AdfNode[]): AdfNode { return content.length === 0 ? node : { ...node, content: [...content] } } -function listNode(node: AdfNode, items: readonly Block[][], definitions: LinkDefinitions, path: ConvertErrorPath, depth: number): Result { +function listNode(node: AdfNode, items: readonly Block[][], definitions: LinkDefinitions, path: ConvertErrorPath, depth: number, memo: SpellingMemo): Result { const content: AdfNode[] = [] for (const [index, blocks] of items.entries()) { - const item = containerNode({ type: 'listItem' }, blocks, definitions, [...path, 'content', index], depth) + const item = containerNode({ type: 'listItem' }, blocks, definitions, [...path, 'content', index], depth, memo) if (!item.ok) return item content.push(item.value) } diff --git a/todo-history.md b/todo-history.md index bc45a24..d2bf5d0 100644 --- a/todo-history.md +++ b/todo-history.md @@ -820,6 +820,36 @@ The done `todo.md` items in full, as they were written. `todo.md` keeps a one-li dropping the outer link silently as `closeLink`'s `applyMark` does today, with a normalization fixture per shape (the stability-reviewer, 2026-09-16; the maintainer, 2026-09-17). +- [x] **18 — The subtree the directive spelling asks about (`0.2.0`).** The parser asks + `commonMarkSpelling` at every directive-spelled block and the answer emits the whole subtree + below, so a node at depth d is spelled d times: three nested rule-first directive lists cost + 18 asks over 10 nodes, and 250 levels parse in 1.2 s at 16.4 kB, 4.9 s at 261 kB with a + kilobyte of content per level. The depth guard bounds the levels at about 250, never the + content, so this is the pipeline persona's hang on an input nobody typed (§11). Keeping each + child's emitted result for its parent's ask is not a straight handover: the same node object + is asked at different depths — 4, 3 and 2 for the innermost list of three — because the + parser counts a list and its item as two levels where the emitter's readable list counts one + (4b), and `headroom` is that guard's slack. The parts that survive the measurement: the paths + agree, `text` and `spelling` carry no depth, `headroom` is affine in it, and the parser asks + first at the deepest of them, so a kept result rebases by the difference. Either rebase and + record that argument in `AGENTS.md`, or give both directions one list accounting so a node + has one depth and nothing needs rebasing — which reopens 4b. A single post-build walk was + rejected: it reports the outer offender where the build reports the inner one (the + maintainer, 2026-09-18). + **Measured** (2026-09-19): the parse keeps each node's readable spelling (AGENTS.md §11), and + 250 nested directive lists fall from 1.22 s to 0.04 s at 16.1 kB and from 5.20 s to 0.12 s at + 261 kB; a panel between every pair of lists 1.00 s to 0.05 s, an opaque carry in every item + 1.08 s to 0.03 s. No figure enters the gate (§14), as 4c settled for a behaviour-preserving + cost fix; what the gate holds is the refusal. An ordered list past the marker cap gives way, + so the emitter spends two levels where the parser spent one: one such list between an ask and + a kept entry cancels the credit the directive form starts with, and two put the read below the + depth that filled it — which a hit would answer without the depth guards the walk runs. That + read re-spells. Three cases hold the arithmetic between them: the two depth-boundary tests + already there pin the sign, the two-overflow case pins the guard, and the one-overflow case + pins the magnitude, a constant rebase accepting there a document the emitter then refuses + (the stability-reviewer, 2026-09-19). The one list accounting was not + taken: 4b settled that accounting the day this was filed, and reopening it is an ask rather + than a chunk. ## 5 — Ship `0.1.0` diff --git a/todo.md b/todo.md index 93188bb..98361fb 100644 --- a/todo.md +++ b/todo.md @@ -219,22 +219,7 @@ bundle size and the tagline. over four concerns in one loop — escape, code span, nested directive, bracket balance — which 4c left half-split and this item either passes or forces apart (the systems-architect and the maintainer, 2026-09-18). -- [ ] **18 — The subtree the directive spelling asks about (`0.2.0`).** The parser asks - `commonMarkSpelling` at every directive-spelled block and the answer emits the whole subtree - below, so a node at depth d is spelled d times: three nested rule-first directive lists cost - 18 asks over 10 nodes, and 250 levels parse in 1.2 s at 16.4 kB, 4.9 s at 261 kB with a - kilobyte of content per level. The depth guard bounds the levels at about 250, never the - content, so this is the pipeline persona's hang on an input nobody typed (§11). Keeping each - child's emitted result for its parent's ask is not a straight handover: the same node object - is asked at different depths — 4, 3 and 2 for the innermost list of three — because the - parser counts a list and its item as two levels where the emitter's readable list counts one - (4b), and `headroom` is that guard's slack. The parts that survive the measurement: the paths - agree, `text` and `spelling` carry no depth, `headroom` is affine in it, and the parser asks - first at the deepest of them, so a kept result rebases by the difference. Either rebase and - record that argument in `AGENTS.md`, or give both directions one list accounting so a node - has one depth and nothing needs rebasing — which reopens 4b. A single post-build walk was - rejected: it reports the outer offender where the build reports the inner one (the - maintainer, 2026-09-18). +- [x] **18 — The subtree the directive spelling asks about (`0.2.0`).** ## The ADF inventory to cover