From 53ef06f56579c327ea60d40b1885e129a31ec973 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Fri, 18 Sep 2026 00:13:55 +0200 Subject: [PATCH] 4b - the block walk spells a list from one walk, a list and its item counting two levels in every spelling --- AGENTS.md | 10 ++- src/markdown/emit/adf-to-markdown.test.ts | 20 ++++++ src/markdown/emit/adf-to-markdown.ts | 71 +++++++++++++++------- src/markdown/parse/markdown-to-adf.test.ts | 6 ++ src/markdown/parse/markdown-to-adf.ts | 2 +- src/nesting.ts | 2 +- todo.md | 12 +++- 7 files changed, 95 insertions(+), 28 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 4ea6928..df7fa21 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -284,10 +284,16 @@ someone spells it or pins it. where the general form fails on the same node. Refusing there refuses a document the general form spells, so a refusal the general form does not share belongs in the general form or nowhere — save the nested list a tight spelling would swallow, whose refusal the - tight-versus-blank answer owns (`todo.md` 2b). + tight-versus-blank answer owns (`todo.md` 2b). A readable spelling that must spell its subtree + before it can give way — the list, whose thematic-break first line and blank lines exist only + spelled — hands that one walk to the general form instead: giving way after the walk walks + again at every level, doubling per level (4b). - Nothing recurses unbounded: the guards walk iteratively, and blocks, marks and JSON values — an attribute's and a carried node's alike — are all held to 500 levels, so a deep document is a - `Result` rather than the stack overflow that waits near 2000. + `Result` rather than the stack overflow that waits near 2000. A block's level is its count of + block ancestors — a list item's children two below the list — in either direction and whichever + spelling holds them, so the guards agree at the boundary and no spelling recurses twice per + level it counts once (the maintainer, 2026-09-18). - A reader takes the text and an index — a sticky regex whose `lastIndex` the caller sets on the line before it reads, `indexOf` — never a fresh slice per character, and a per-character walk hoists the scan that does not vary with the character. The pipeline persona feeds documents diff --git a/src/markdown/emit/adf-to-markdown.test.ts b/src/markdown/emit/adf-to-markdown.test.ts index f2eea24..26fd572 100644 --- a/src/markdown/emit/adf-to-markdown.test.ts +++ b/src/markdown/emit/adf-to-markdown.test.ts @@ -440,6 +440,15 @@ test('spells a list item whose marker completes a thematic break as a directive' const nested: AdfNode = { content: [item({ content: [item()], type: 'bulletList' })], type: 'bulletList' } assert.equal(markdown(adfToMarkdown(document(nested))), '- -\n') assert.equal(markdown(adfToMarkdown(document({ content: [item(nested)], type: 'bulletList' }))), '!adf:bulletList\n!adf:listItem\n- -\n!adf:/listItem\n!adf:/bulletList\n') + assert.equal( + markdown(adfToMarkdown(document({ content: [item(paragraph({ text: 'a', type: 'text' }), nested), item({ type: 'rule' })], type: 'bulletList' }))), + '!adf:bulletList\n!adf:listItem\na\n\n- -\n!adf:/listItem\n!adf:listItem\n---\n!adf:/listItem\n!adf:/bulletList\n', + ) + const spaced: AdfNode = { content: [{ text: 'a\n \nb', type: 'text' }], type: 'codeBlock' } + assert.equal( + markdown(adfToMarkdown(document({ attrs: { order: 3 }, content: [item(spaced)], type: 'orderedList' }))), + '!adf:orderedList {order=3}\n!adf:listItem\n```\na\n \nb\n```\n!adf:/listItem\n!adf:/orderedList\n', + ) }) test('refuses the characters CommonMark rewrites', () => { @@ -500,6 +509,17 @@ test('refuses a document nested deeper than the emitter carries', () => { let carried: AdfNode = paragraph({ text: 'x', type: 'text' }) for (let depth = 0; depth < 500; depth += 1) carried = { content: [carried], type: 'blockquote' } assert.ok(adfToMarkdown(document(carried)).ok) + const listed = (levels: number, first: readonly AdfNode[]): AdfNode => { + let list: AdfNode = { content: [{ content: [...first], type: 'listItem' }], type: 'bulletList' } + for (let level = 1; level < levels; level += 1) list = { content: [{ content: [...first, list], type: 'listItem' }], type: 'bulletList' } + return list + } + const lists = largestNesting / 2 + assert.ok(adfToMarkdown(document(listed(lists, []))).ok) + assert.equal(code(adfToMarkdown(document(listed(lists + 1, [])))), 'unsupported-nesting-depth') + const deep = document(listed(lists, [{ type: 'rule' }])) + assert.deepEqual(markdownToAdf(markdown(adfToMarkdown(deep))), { ok: true, value: deep }) + assert.equal(code(adfToMarkdown(document(listed(lists + 1, [{ type: 'rule' }])))), 'unsupported-nesting-depth') }) test('emits an empty list item without trailing whitespace', () => { diff --git a/src/markdown/emit/adf-to-markdown.ts b/src/markdown/emit/adf-to-markdown.ts index 3fcabfa..8946689 100644 --- a/src/markdown/emit/adf-to-markdown.ts +++ b/src/markdown/emit/adf-to-markdown.ts @@ -1,7 +1,7 @@ import type { AdfDocument, AdfNode } from '../../adf/document.ts' import type { BlockDirective } from '../../adf/block-directives.ts' import { adfDocumentFault, carriesOnly, nodeAttrs, nodeContent, nodeMarks } from '../../adf/document.ts' -import { blockDirective } from '../../adf/block-directives.ts' +import { blockDirective, blockDirectives } from '../../adf/block-directives.ts' import { blockDirectiveForm } from '../block-directive-forms.ts' import { carriedBlock } from '../opaque-carry.ts' import { emitInlineLine } from './inline-line.ts' @@ -12,7 +12,7 @@ import { languageSlot } from '../code-language.ts' import { largestNesting } from '../../nesting.ts' import { listBreakSpelling } from '../list-break.ts' import { spellBlockDirectiveOpener } from './block-directive-spelling.ts' -import { spellDirectiveCloser } from '../directive-syntax.ts' +import { spellDirectiveCloser, spellDirectiveOpener } from '../directive-syntax.ts' import { tryImage } from './image.ts' import { tryPipeTable } from './pipe-table.ts' @@ -20,8 +20,10 @@ type BlockContainer = 'directive' | 'document' | 'list-item' type BlockSpelling = 'commonmark' | 'directive' | 'list' type EmittedBlock = { spelling: BlockSpelling; text: string } type PlacedBlock = EmittedBlock & { node: AdfNode } +type WalkedItem = { blocks: readonly PlacedBlock[]; node: AdfNode } const largestListMarker = 999999999 +const listItemOpener = spellDirectiveOpener('listItem', undefined, '') export function adfToMarkdown(document: AdfDocument): Result { const fault = adfDocumentFault(document) @@ -33,6 +35,12 @@ export function adfToMarkdown(document: AdfDocument): Result { } function emitBlocks(nodes: readonly AdfNode[], container: BlockContainer, path: ConvertErrorPath, depth: number): Result { + const blocks = walkBlocks(nodes, path, depth) + if (!blocks.ok) return blocks + return success(joinBlocks(blocks.value, container)) +} + +function walkBlocks(nodes: readonly AdfNode[], path: ConvertErrorPath, depth: number): Result { if (depth > largestNesting) return failure('unsupported-nesting-depth', `the document nests deeper than the ${largestNesting} levels the emitter carries`, path) const blocks: PlacedBlock[] = [] for (const [index, node] of nodes.entries()) { @@ -40,13 +48,17 @@ function emitBlocks(nodes: readonly AdfNode[], container: BlockContainer, path: if (!block.ok) return block blocks.push({ ...block.value, node }) } + return success(blocks) +} + +function joinBlocks(blocks: readonly PlacedBlock[], container: BlockContainer): string { let text = '' for (const [index, block] of blocks.entries()) { const previous = blocks[index - 1] if (previous !== undefined) text += separationBetween(previous, block, container) text += block.text } - return success(text) + return text } function separationBetween(previous: PlacedBlock, next: PlacedBlock, container: BlockContainer): string { @@ -73,13 +85,14 @@ function emitBlock(node: AdfNode, path: ConvertErrorPath, depth: number): Result if (directive === undefined) return commonMarkLine(carriedBlock(node, path, depth)) const readable = readableBlock(node, path, depth) if (readable !== undefined) return readable - return emitDirectiveBlock(node, directive, path, depth) + return emitDirectiveBlock(node, directive, path, depth, () => walkBlocks(nodeContent(node), path, depth + 1)) } export function commonMarkSpelling(node: AdfNode, path: ConvertErrorPath, depth: number): Result | undefined { const readable = readableBlock(node, path, depth) if (readable === undefined) return undefined - return readable.ok ? success(null) : readable + if (!readable.ok) return readable + return readable.value.spelling === 'directive' ? undefined : success(null) } function readableBlock(node: AdfNode, path: ConvertErrorPath, depth: number): Result | undefined { @@ -111,21 +124,25 @@ function directivePair(node: AdfNode, opener: string, body: string): EmittedBloc return { spelling: 'directive', text: `${opener}\n${body === '' ? '' : `${body}\n`}${spellDirectiveCloser(node.type)}` } } -function emitDirectiveBlock(node: AdfNode, directive: BlockDirective, path: ConvertErrorPath, depth: number): Result { +function emitDirectiveBlock(node: AdfNode, directive: BlockDirective, path: ConvertErrorPath, depth: number, walkBody: () => Result): Result { if (node.text !== undefined) return failure('unsupported-node-shape', `a ${node.type} carries no text: this one holds text`, path) if (blockDirectiveForm(node.type) === 'leaf' && nodeContent(node).length > 0) return failure('unsupported-node-shape', `a ${node.type} holds no content: this one holds some`, path) if (directive.contentModel === 'code') return emitCodeDirective(node, directive, path, depth) const opener = spellBlockDirectiveOpener(node, directive) if (opener === undefined) return commonMarkLine(carriedBlock(node, path, depth)) - return emitDirectiveBody(node, directive, opener, path, depth) + return emitDirectiveBody(node, directive, opener, path, walkBody) } -function emitDirectiveBody(node: AdfNode, directive: BlockDirective, opener: string, path: ConvertErrorPath, depth: number): Result { +function emitDirectiveBody(node: AdfNode, directive: BlockDirective, opener: string, path: ConvertErrorPath, walkBody: () => Result): Result { if (blockDirectiveForm(node.type) === 'leaf') return success({ spelling: 'directive', text: opener }) - const content = nodeContent(node) - const body = directive.contentModel === 'inline' ? emitInlineLine(content, 'paragraph', path) : emitBlocks(content, 'directive', path, depth + 1) - if (!body.ok) return body - return success(directivePair(node, opener, body.value)) + if (directive.contentModel === 'inline') { + const line = emitInlineLine(nodeContent(node), 'paragraph', path) + if (!line.ok) return line + return success(directivePair(node, opener, line.value)) + } + const blocks = walkBody() + if (!blocks.ok) return blocks + return success(directivePair(node, opener, joinBlocks(blocks.value, 'directive'))) } function emitBlockquote(node: AdfNode, path: ConvertErrorPath, depth: number): Result | undefined { @@ -197,16 +214,25 @@ function emitList(node: AdfNode, path: ConvertErrorPath, depth: number): Result< const start = listStart(node, items.length) if (start === undefined || items.length === 0) return undefined if (items.some((item) => item.type !== 'listItem' || !carriesOnly(item, []))) return undefined - const lines: string[] = [] + const walked: WalkedItem[] = [] for (const [offset, item] of items.entries()) { - const emitted = emitListItem(item, ordered ? `${start + offset}. ` : '- ', [...path, 'content', offset], depth) - if (emitted === undefined) return undefined - if (!emitted.ok) return emitted - lines.push(emitted.value) + const blocks = walkBlocks(nodeContent(item), [...path, 'content', offset], depth + 2) + if (!blocks.ok) return blocks + walked.push({ blocks: blocks.value, node: item }) + } + const lines: string[] = [] + for (const [offset, item] of walked.entries()) { + const line = listItemLines(item.blocks, ordered ? `${start + offset}. ` : '- ') + if (line === undefined) return emitDirectiveBlock(node, ordered ? blockDirectives.orderedList : blockDirectives.bulletList, path, depth, () => success(directiveItems(walked))) + lines.push(line) } return success({ spelling: 'list', text: lines.join('\n') }) } +function directiveItems(items: readonly WalkedItem[]): PlacedBlock[] { + return items.map((item) => ({ ...directivePair(item.node, listItemOpener, joinBlocks(item.blocks, 'directive')), node: item.node })) +} + function listStart(node: AdfNode, items: number): number | undefined { if (node.type !== 'orderedList') return 0 const start = nodeAttrs(node)['order'] @@ -214,16 +240,15 @@ function listStart(node: AdfNode, items: number): number | undefined { return start + items - 1 > largestListMarker ? undefined : start } -function emitListItem(item: AdfNode, marker: string, path: ConvertErrorPath, depth: number): Result | undefined { - const inner = emitBlocks(nodeContent(item), 'list-item', path, depth + 1) - if (!inner.ok) return inner - if (inner.value === '') return success(marker.trimEnd()) - const body = inner.value.split('\n') +function listItemLines(blocks: readonly PlacedBlock[], marker: string): string | undefined { + const inner = joinBlocks(blocks, 'list-item') + if (inner === '') return marker.trimEnd() + const body = inner.split('\n') if (body.some((line) => line !== '' && isBlankLine(line))) return undefined const indent = ' '.repeat(marker.length) const lines = body.map((line, index) => (index === 0 ? `${marker}${line}` : line === '' ? '' : `${indent}${line}`)) if (isThematicBreak(lines[0] ?? '')) return undefined - return success(lines.join('\n')) + return lines.join('\n') } function emitParagraph(node: AdfNode, path: ConvertErrorPath): Result | undefined { diff --git a/src/markdown/parse/markdown-to-adf.test.ts b/src/markdown/parse/markdown-to-adf.test.ts index fcceb47..e820109 100644 --- a/src/markdown/parse/markdown-to-adf.test.ts +++ b/src/markdown/parse/markdown-to-adf.test.ts @@ -675,6 +675,12 @@ test('refuses input nested deeper than the parser carries', () => { assert.ok(markdownToAdf(nest(repeated('panel'), '!adf:paragraph {localId=a-1}\nPart.\n!adf:/paragraph\n')).ok) assert.deepEqual(position(markdownToAdf(nest(['expand', ...repeated('panel'), 'expand'], 'Part.\n'))), { line: 501, offset: 5501 }) assert.equal(code(markdownToAdf(nest(['panel', ...repeated('expand'), 'panel'], 'Part.\n'))), 'unsupported-nesting-depth') + const listed = (levels: number): string => `${'!adf:bulletList\n!adf:listItem\n---\n'.repeat(levels)}${'!adf:/listItem\n!adf:/bulletList\n'.repeat(levels)}` + const lists = largestNesting / 2 + assert.ok(markdownToAdf(listed(lists)).ok) + assert.equal(code(markdownToAdf(listed(lists + 1))), 'unsupported-nesting-depth') + assert.ok(markdownToAdf(`${'- '.repeat(lists)}a\n`).ok) + assert.equal(code(markdownToAdf(`${'- '.repeat(lists + 1)}a\n`)), 'unsupported-nesting-depth') }) test('decodes the backslash escapes CommonMark spells, and keeps the rest literal', () => { diff --git a/src/markdown/parse/markdown-to-adf.ts b/src/markdown/parse/markdown-to-adf.ts index 2c2425b..b2d24c5 100644 --- a/src/markdown/parse/markdown-to-adf.ts +++ b/src/markdown/parse/markdown-to-adf.ts @@ -150,7 +150,7 @@ function withContent(node: AdfNode, content: readonly AdfNode[]): AdfNode { function listNode(node: AdfNode, items: readonly Block[][], definitions: LinkDefinitions, path: ConvertErrorPath, depth: number): Result { const content: AdfNode[] = [] for (const [index, blocks] of items.entries()) { - const item = containerNode({ type: 'listItem' }, blocks, definitions, [...path, 'content', index], depth) + const item = containerNode({ type: 'listItem' }, blocks, definitions, [...path, 'content', index], depth + 1) if (!item.ok) return item content.push(item.value) } diff --git a/src/nesting.ts b/src/nesting.ts index 89984d2..db16b17 100644 --- a/src/nesting.ts +++ b/src/nesting.ts @@ -1,2 +1,2 @@ -// One level per block-list recursion in either direction — a list and its items count once — or the two guards disagree. +// A block's depth is its count of block ancestors, a list item's children two below the list, in either direction and whichever spelling holds them — or the two guards disagree. export const largestNesting = 500 diff --git a/todo.md b/todo.md index 65a5002..668bfa3 100644 --- a/todo.md +++ b/todo.md @@ -64,6 +64,12 @@ proves 12, 13 spells 11's gaps in 12's grammar, and 12 rewrites code 4b and 4c c export persona runs in bulk walks the document twice. Both walks are linear, so this is a constant factor rather than 4b's class change, and the parting is what gives depth its own code (§8) — measure before joining them back. + **Settled** (the maintainer, 2026-09-18): the one walk cannot know the spelling it will take, + so a list and its item count two levels in every spelling and both directions — the readable + list counted one and the directive form two, and letting the directive form count one put + twice the frames on the stack at 500 — leaving 250 nested lists the limit. + **Measured** (2026-09-18): `adfDocumentFault` walks a 9 MB document in 52 ms against 314 ms + for the emit, so its two walks stay parted. - [ ] **4c — The scanning rule's remaining sites (`0.2.0`).** A trailing-anchored regex re-walks its run from every start position, so an interior whitespace run costs quadratic time rather than linear — 3h measured 80k spaces inside an ATX heading at 11.3s, and 3ms once the walk @@ -81,7 +87,11 @@ proves 12, 13 spells 11's gaps in 12's grammar, and 12 rewrites code 4b and 4c c (the stability-reviewer, 2026-09-16). §11's scanning rule is the whole argument; the pipeline persona feeds documents nobody typed. `readDirectiveContent`'s scan splits into named steps with that fix rather than keeping its - complexity (the maintainer, 2026-09-16). + complexity (the maintainer, 2026-09-16). A sixth 4b leaves behind: the parser asks + `commonMarkSpelling` at every directive-spelled list it reads, and the answer spells the whole + subtree below, so nested directive lists cost O(depth × subtree) — 250 rule-first levels parse + in 1.3 s at 16.5 kB, 1.5 MB of the same shape in 0.6 s — bounded by the depth guard like + `readNestedDirective` (the maintainer, 2026-09-18). - [ ] **4d — What the gate says while it runs (`0.2.1`).** `ci.sh` runs nine legs and announces none of them, so five minutes of a Gitea run read as silence and a hang cannot be told from a slow pull — the maintainer hit exactly this on the `0.1.0` release. Three causes, each its