From 98ff478fe9627968335f2a7401ca89280e05b31c Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Fri, 18 Sep 2026 00:34:52 +0200 Subject: [PATCH] 4b - the walk reports its headroom, so a list falling back to the directive form refuses at zero and the list limit stays 500 --- AGENTS.md | 11 ++-- src/json-value.ts | 13 ++-- src/markdown/emit/adf-to-markdown.test.ts | 18 +++-- src/markdown/emit/adf-to-markdown.ts | 76 +++++++++++++--------- src/markdown/opaque-carry.ts | 15 +++-- src/markdown/parse/markdown-to-adf.test.ts | 10 +-- src/markdown/parse/markdown-to-adf.ts | 2 +- src/nesting.ts | 2 +- todo.md | 11 ++-- 9 files changed, 94 insertions(+), 64 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index df7fa21..dde33ac 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -290,10 +290,13 @@ someone spells it or pins it. again at every level, doubling per level (4b). - Nothing recurses unbounded: the guards walk iteratively, and blocks, marks and JSON values — an attribute's and a carried node's alike — are all held to 500 levels, so a deep document is a - `Result` rather than the stack overflow that waits near 2000. A block's level is its count of - block ancestors — a list item's children two below the list — in either direction and whichever - spelling holds them, so the guards agree at the boundary and no spelling recurses twice per - level it counts once (the maintainer, 2026-09-18). + `Result` rather than the stack overflow that waits near 2000. A level is one block-list + recursion in either direction: a readable list's items sit one below it, its directive + spelling's two. So a list giving way after its walk owes the directive form a level the walk + did not count, and the walk reports its headroom — the least slack any depth guard below it + has — for the fallback to refuse at zero rather than walk again; counting every list twice + halved the list limit, counting the directive form once doubled the parser's frames per level + (the maintainer, 2026-09-18). - A reader takes the text and an index — a sticky regex whose `lastIndex` the caller sets on the line before it reads, `indexOf` — never a fresh slice per character, and a per-character walk hoists the scan that does not vary with the character. The pipeline persona feeds documents diff --git a/src/json-value.ts b/src/json-value.ts index 5c7da45..6448b4d 100644 --- a/src/json-value.ts +++ b/src/json-value.ts @@ -19,15 +19,20 @@ export function isJsonValue(value: unknown): value is JsonValue { return true } -export function overNested(value: JsonValue, levels: number = largestNesting): boolean { - const pending: { depth: number; item: JsonValue }[] = [{ depth: 0, item: value }] +export function nestingDepth(value: unknown): number { + const pending: { depth: number; item: unknown }[] = [{ depth: 0, item: value }] + let deepest = 0 while (pending.length > 0) { const entry = pending.pop() if (entry === undefined) continue const { depth, item } = entry - if (depth > levels) return true + deepest = Math.max(deepest, depth) if (Array.isArray(item)) for (const child of item) pending.push({ depth: depth + 1, item: child }) else if (item !== null && typeof item === 'object') for (const child of Object.values(item)) pending.push({ depth: depth + 1, item: child }) } - return false + return deepest +} + +export function overNested(value: unknown, levels: number = largestNesting): boolean { + return nestingDepth(value) > levels } diff --git a/src/markdown/emit/adf-to-markdown.test.ts b/src/markdown/emit/adf-to-markdown.test.ts index 26fd572..e3ecae1 100644 --- a/src/markdown/emit/adf-to-markdown.test.ts +++ b/src/markdown/emit/adf-to-markdown.test.ts @@ -514,12 +514,20 @@ test('refuses a document nested deeper than the emitter carries', () => { for (let level = 1; level < levels; level += 1) list = { content: [{ content: [...first, list], type: 'listItem' }], type: 'bulletList' } return list } - const lists = largestNesting / 2 - assert.ok(adfToMarkdown(document(listed(lists, []))).ok) - assert.equal(code(adfToMarkdown(document(listed(lists + 1, [])))), 'unsupported-nesting-depth') - const deep = document(listed(lists, [{ type: 'rule' }])) + assert.ok(adfToMarkdown(document(listed(largestNesting, [paragraph({ text: 'x', type: 'text' })]))).ok) + assert.equal(code(adfToMarkdown(document(listed(largestNesting + 1, [paragraph({ text: 'x', type: 'text' })])))), 'unsupported-nesting-depth') + const directiveLists = largestNesting / 2 + const deep = document(listed(directiveLists, [{ type: 'rule' }])) assert.deepEqual(markdownToAdf(markdown(adfToMarkdown(deep))), { ok: true, value: deep }) - assert.equal(code(adfToMarkdown(document(listed(lists + 1, [{ type: 'rule' }])))), 'unsupported-nesting-depth') + assert.equal(code(adfToMarkdown(document(listed(directiveLists + 1, [{ type: 'rule' }])))), 'unsupported-nesting-depth') + const chain = (levels: number): AdfNode => { + let card: AdfNode = { type: 'blockCard' } + for (let level = 0; level < levels; level += 1) card = { content: [card], type: 'blockCard' } + return card + } + const carriedInItem = (levels: number): AdfDocument => document(listed(1, [{ type: 'rule' }, chain(levels)])) + assert.deepEqual(markdownToAdf(markdown(adfToMarkdown(carriedInItem(248)))), { ok: true, value: carriedInItem(248) }) + assert.equal(code(adfToMarkdown(carriedInItem(249))), 'unsupported-nesting-depth') }) test('emits an empty list item without trailing whitespace', () => { diff --git a/src/markdown/emit/adf-to-markdown.ts b/src/markdown/emit/adf-to-markdown.ts index 8946689..7e2027c 100644 --- a/src/markdown/emit/adf-to-markdown.ts +++ b/src/markdown/emit/adf-to-markdown.ts @@ -18,11 +18,13 @@ import { tryPipeTable } from './pipe-table.ts' type BlockContainer = 'directive' | 'document' | 'list-item' type BlockSpelling = 'commonmark' | 'directive' | 'list' -type EmittedBlock = { spelling: BlockSpelling; text: string } +type EmittedBlock = { headroom: number; spelling: BlockSpelling; text: string } type PlacedBlock = EmittedBlock & { node: AdfNode } -type WalkedItem = { blocks: readonly PlacedBlock[]; node: AdfNode } +type Walk = { blocks: readonly PlacedBlock[]; headroom: number } +type WalkedItem = { node: AdfNode; walk: Walk } const largestListMarker = 999999999 +// Bare because emitList admits no item carrying attributes, marks or text. const listItemOpener = spellDirectiveOpener('listItem', undefined, '') export function adfToMarkdown(document: AdfDocument): Result { @@ -35,20 +37,27 @@ export function adfToMarkdown(document: AdfDocument): Result { } function emitBlocks(nodes: readonly AdfNode[], container: BlockContainer, path: ConvertErrorPath, depth: number): Result { - const blocks = walkBlocks(nodes, path, depth) - if (!blocks.ok) return blocks - return success(joinBlocks(blocks.value, container)) + const walk = walkBlocks(nodes, path, depth) + if (!walk.ok) return walk + return success(joinBlocks(walk.value.blocks, container)) } -function walkBlocks(nodes: readonly AdfNode[], path: ConvertErrorPath, depth: number): Result { - if (depth > largestNesting) return failure('unsupported-nesting-depth', `the document nests deeper than the ${largestNesting} levels the emitter carries`, path) +// The walk's headroom is the least slack any depth guard below it has, so a spelling that sinks the walked blocks a level can refuse rather than walk again. +function walkBlocks(nodes: readonly AdfNode[], path: ConvertErrorPath, depth: number): Result { + let headroom = largestNesting - depth + if (headroom < 0) return tooDeep(path) const blocks: PlacedBlock[] = [] for (const [index, node] of nodes.entries()) { const block = emitBlock(node, [...path, 'content', index], depth) if (!block.ok) return block + headroom = Math.min(headroom, block.value.headroom) blocks.push({ ...block.value, node }) } - return success(blocks) + return success({ blocks, headroom }) +} + +function tooDeep(path: ConvertErrorPath): Result { + return failure('unsupported-nesting-depth', `the document nests deeper than the ${largestNesting} levels the emitter carries`, path) } function joinBlocks(blocks: readonly PlacedBlock[], container: BlockContainer): string { @@ -111,20 +120,20 @@ function readableText(text: string | undefined): Result | undefine return text === undefined ? undefined : success(commonMarkText(text)) } -function commonMarkLine(text: Result): Result { - if (!text.ok) return text - return success(commonMarkText(text.value)) +function commonMarkLine(carried: Result<{ headroom: number; text: string }>): Result { + if (!carried.ok) return carried + return success({ ...carried.value, spelling: 'commonmark' }) } -function commonMarkText(text: string): EmittedBlock { - return { spelling: 'commonmark', text } +function commonMarkText(text: string, headroom: number = Number.POSITIVE_INFINITY): EmittedBlock { + return { headroom, spelling: 'commonmark', text } } -function directivePair(node: AdfNode, opener: string, body: string): EmittedBlock { - return { spelling: 'directive', text: `${opener}\n${body === '' ? '' : `${body}\n`}${spellDirectiveCloser(node.type)}` } +function directivePair(node: AdfNode, opener: string, body: string, headroom: number = Number.POSITIVE_INFINITY): EmittedBlock { + return { headroom, spelling: 'directive', text: `${opener}\n${body === '' ? '' : `${body}\n`}${spellDirectiveCloser(node.type)}` } } -function emitDirectiveBlock(node: AdfNode, directive: BlockDirective, path: ConvertErrorPath, depth: number, walkBody: () => Result): Result { +function emitDirectiveBlock(node: AdfNode, directive: BlockDirective, path: ConvertErrorPath, depth: number, walkBody: () => Result): Result { if (node.text !== undefined) return failure('unsupported-node-shape', `a ${node.type} carries no text: this one holds text`, path) if (blockDirectiveForm(node.type) === 'leaf' && nodeContent(node).length > 0) return failure('unsupported-node-shape', `a ${node.type} holds no content: this one holds some`, path) if (directive.contentModel === 'code') return emitCodeDirective(node, directive, path, depth) @@ -133,27 +142,27 @@ function emitDirectiveBlock(node: AdfNode, directive: BlockDirective, path: Conv return emitDirectiveBody(node, directive, opener, path, walkBody) } -function emitDirectiveBody(node: AdfNode, directive: BlockDirective, opener: string, path: ConvertErrorPath, walkBody: () => Result): Result { - if (blockDirectiveForm(node.type) === 'leaf') return success({ spelling: 'directive', text: opener }) +function emitDirectiveBody(node: AdfNode, directive: BlockDirective, opener: string, path: ConvertErrorPath, walkBody: () => Result): Result { + if (blockDirectiveForm(node.type) === 'leaf') return success({ headroom: Number.POSITIVE_INFINITY, spelling: 'directive', text: opener }) if (directive.contentModel === 'inline') { const line = emitInlineLine(nodeContent(node), 'paragraph', path) if (!line.ok) return line return success(directivePair(node, opener, line.value)) } - const blocks = walkBody() - if (!blocks.ok) return blocks - return success(directivePair(node, opener, joinBlocks(blocks.value, 'directive'))) + const walk = walkBody() + if (!walk.ok) return walk + return success(directivePair(node, opener, joinBlocks(walk.value.blocks, 'directive'), walk.value.headroom)) } function emitBlockquote(node: AdfNode, path: ConvertErrorPath, depth: number): Result | undefined { if (!carriesOnly(node, [])) return undefined - const inner = emitBlocks(nodeContent(node), 'document', path, depth + 1) + const inner = walkBlocks(nodeContent(node), path, depth + 1) if (!inner.ok) return inner - const text = inner.value + const text = joinBlocks(inner.value.blocks, 'document') .split('\n') .map((line) => (line === '' ? '>' : `> ${line}`)) .join('\n') - return success(commonMarkText(text)) + return success(commonMarkText(text, inner.value.headroom)) } function emitCodeBlock(node: AdfNode, path: ConvertErrorPath): Result | undefined { @@ -216,21 +225,24 @@ function emitList(node: AdfNode, path: ConvertErrorPath, depth: number): Result< if (items.some((item) => item.type !== 'listItem' || !carriesOnly(item, []))) return undefined const walked: WalkedItem[] = [] for (const [offset, item] of items.entries()) { - const blocks = walkBlocks(nodeContent(item), [...path, 'content', offset], depth + 2) - if (!blocks.ok) return blocks - walked.push({ blocks: blocks.value, node: item }) + const walk = walkBlocks(nodeContent(item), [...path, 'content', offset], depth + 1) + if (!walk.ok) return walk + walked.push({ node: item, walk: walk.value }) } + const headroom = Math.min(...walked.map((item) => item.walk.headroom)) const lines: string[] = [] for (const [offset, item] of walked.entries()) { - const line = listItemLines(item.blocks, ordered ? `${start + offset}. ` : '- ') - if (line === undefined) return emitDirectiveBlock(node, ordered ? blockDirectives.orderedList : blockDirectives.bulletList, path, depth, () => success(directiveItems(walked))) + const line = listItemLines(item.walk.blocks, ordered ? `${start + offset}. ` : '- ') + if (line === undefined) return headroom < 1 ? tooDeep(path) : emitDirectiveBlock(node, ordered ? blockDirectives.orderedList : blockDirectives.bulletList, path, depth, () => success(directiveItems(walked))) lines.push(line) } - return success({ spelling: 'list', text: lines.join('\n') }) + return success({ headroom, spelling: 'list', text: lines.join('\n') }) } -function directiveItems(items: readonly WalkedItem[]): PlacedBlock[] { - return items.map((item) => ({ ...directivePair(item.node, listItemOpener, joinBlocks(item.blocks, 'directive')), node: item.node })) +// The directive form sinks each item's blocks a level below where the walk read them. +function directiveItems(items: readonly WalkedItem[]): Walk { + const blocks = items.map((item) => ({ ...directivePair(item.node, listItemOpener, joinBlocks(item.walk.blocks, 'directive'), item.walk.headroom - 1), node: item.node })) + return { blocks, headroom: Math.min(...blocks.map((block) => block.headroom)) } } function listStart(node: AdfNode, items: number): number | undefined { diff --git a/src/markdown/opaque-carry.ts b/src/markdown/opaque-carry.ts index f637bd9..63d0435 100644 --- a/src/markdown/opaque-carry.ts +++ b/src/markdown/opaque-carry.ts @@ -3,7 +3,7 @@ import type { DirectiveSpan, Read } from './directive-syntax.ts' import type { JsonSpelling } from '../canonical-json.ts' import { failure, success, type ConvertErrorPath, type Result } from '../result.ts' import { isAdfNode } from '../adf/document.ts' -import { isJsonValue, overNested } from '../json-value.ts' +import { isJsonValue, nestingDepth, overNested } from '../json-value.ts' import { fencedCodeBlock } from './backtick-runs.ts' import { largestNesting } from '../nesting.ts' import { malformedDirective, readSoleStringAttribute, spellAttributes, spellInlineLeafDirective, spellStringAttribute, unsupportedNodeShape } from './directive-syntax.ts' @@ -13,16 +13,16 @@ export const carryName = 'carry' const jsonAttribute = 'json' -export function carriedBlock(node: AdfNode, path: ConvertErrorPath, depth: number): Result { +export function carriedBlock(node: AdfNode, path: ConvertErrorPath, depth: number): Result<{ headroom: number; text: string }> { const json = carriedJson(node, 'two-space', path, largestNesting - depth) if (!json.ok) return json - return success(fencedCodeBlock(carryName, json.value)) + return success({ headroom: json.value.headroom, text: fencedCodeBlock(carryName, json.value.json) }) } export function carriedInline(node: AdfNode, path: ConvertErrorPath): Result { const json = carriedJson(node, 'compact', path, largestNesting) if (!json.ok) return json - return success(spellInlineLeafDirective(carryName, spellAttributes([[jsonAttribute, spellStringAttribute(json.value)]]))) + return success(spellInlineLeafDirective(carryName, spellAttributes([[jsonAttribute, spellStringAttribute(json.value.json)]]))) } export function readCarriedBlock(body: string, depth: number): Read { @@ -36,11 +36,12 @@ export function readCarriedInline(span: DirectiveSpan): Read | undefine return readCarriedJson(spelled.value, 'compact', largestNesting) } -function carriedJson(node: AdfNode, spelling: JsonSpelling, path: ConvertErrorPath, levels: number): Result { - if (!isJsonValue(node) || overNested(node, levels)) { +function carriedJson(node: AdfNode, spelling: JsonSpelling, path: ConvertErrorPath, levels: number): Result<{ headroom: number; json: string }> { + const headroom = levels - nestingDepth(node) + if (!isJsonValue(node) || headroom < 0) { return failure('unsupported-nesting-depth', `a carried node's JSON nests deeper than the ${levels} levels its position leaves`, path) } - return success(serializeCanonicalJson(node, spelling)) + return success({ headroom, json: serializeCanonicalJson(node, spelling) }) } function readCarriedJson(raw: string, spelling: JsonSpelling, levels: number): Read { diff --git a/src/markdown/parse/markdown-to-adf.test.ts b/src/markdown/parse/markdown-to-adf.test.ts index e820109..d0f3f87 100644 --- a/src/markdown/parse/markdown-to-adf.test.ts +++ b/src/markdown/parse/markdown-to-adf.test.ts @@ -676,11 +676,11 @@ test('refuses input nested deeper than the parser carries', () => { assert.deepEqual(position(markdownToAdf(nest(['expand', ...repeated('panel'), 'expand'], 'Part.\n'))), { line: 501, offset: 5501 }) assert.equal(code(markdownToAdf(nest(['panel', ...repeated('expand'), 'panel'], 'Part.\n'))), 'unsupported-nesting-depth') const listed = (levels: number): string => `${'!adf:bulletList\n!adf:listItem\n---\n'.repeat(levels)}${'!adf:/listItem\n!adf:/bulletList\n'.repeat(levels)}` - const lists = largestNesting / 2 - assert.ok(markdownToAdf(listed(lists)).ok) - assert.equal(code(markdownToAdf(listed(lists + 1))), 'unsupported-nesting-depth') - assert.ok(markdownToAdf(`${'- '.repeat(lists)}a\n`).ok) - assert.equal(code(markdownToAdf(`${'- '.repeat(lists + 1)}a\n`)), 'unsupported-nesting-depth') + const directiveLists = largestNesting / 2 + assert.ok(markdownToAdf(listed(directiveLists)).ok) + assert.equal(code(markdownToAdf(listed(directiveLists + 1))), 'unsupported-nesting-depth') + assert.ok(markdownToAdf(`${'- '.repeat(largestNesting)}a\n`).ok) + assert.equal(code(markdownToAdf(`${'- '.repeat(largestNesting + 1)}a\n`)), 'unsupported-nesting-depth') }) test('decodes the backslash escapes CommonMark spells, and keeps the rest literal', () => { diff --git a/src/markdown/parse/markdown-to-adf.ts b/src/markdown/parse/markdown-to-adf.ts index b2d24c5..2c2425b 100644 --- a/src/markdown/parse/markdown-to-adf.ts +++ b/src/markdown/parse/markdown-to-adf.ts @@ -150,7 +150,7 @@ function withContent(node: AdfNode, content: readonly AdfNode[]): AdfNode { function listNode(node: AdfNode, items: readonly Block[][], definitions: LinkDefinitions, path: ConvertErrorPath, depth: number): Result { const content: AdfNode[] = [] for (const [index, blocks] of items.entries()) { - const item = containerNode({ type: 'listItem' }, blocks, definitions, [...path, 'content', index], depth + 1) + const item = containerNode({ type: 'listItem' }, blocks, definitions, [...path, 'content', index], depth) if (!item.ok) return item content.push(item.value) } diff --git a/src/nesting.ts b/src/nesting.ts index db16b17..19588ac 100644 --- a/src/nesting.ts +++ b/src/nesting.ts @@ -1,2 +1,2 @@ -// A block's depth is its count of block ancestors, a list item's children two below the list, in either direction and whichever spelling holds them — or the two guards disagree. +// Levels count per AGENTS.md §11. export const largestNesting = 500 diff --git a/todo.md b/todo.md index 668bfa3..67b8383 100644 --- a/todo.md +++ b/todo.md @@ -64,10 +64,11 @@ proves 12, 13 spells 11's gaps in 12's grammar, and 12 rewrites code 4b and 4c c export persona runs in bulk walks the document twice. Both walks are linear, so this is a constant factor rather than 4b's class change, and the parting is what gives depth its own code (§8) — measure before joining them back. - **Settled** (the maintainer, 2026-09-18): the one walk cannot know the spelling it will take, - so a list and its item count two levels in every spelling and both directions — the readable - list counted one and the directive form two, and letting the directive form count one put - twice the frames on the stack at 500 — leaving 250 nested lists the limit. + **Settled** (the maintainer, 2026-09-18): the limit stays 500 readable lists. The one walk + counts items one below the list, as the readable form does, and reports its headroom; a + list falling back to the directive form refuses when that form's extra level no longer + fits. Counting every list twice was rejected for halving the limit, counting the directive + form once for doubling the parser's frames per level. **Measured** (2026-09-18): `adfDocumentFault` walks a 9 MB document in 52 ms against 314 ms for the emit, so its two walks stay parted. - [ ] **4c — The scanning rule's remaining sites (`0.2.0`).** A trailing-anchored regex re-walks @@ -90,7 +91,7 @@ proves 12, 13 spells 11's gaps in 12's grammar, and 12 rewrites code 4b and 4c c complexity (the maintainer, 2026-09-16). A sixth 4b leaves behind: the parser asks `commonMarkSpelling` at every directive-spelled list it reads, and the answer spells the whole subtree below, so nested directive lists cost O(depth × subtree) — 250 rule-first levels parse - in 1.3 s at 16.5 kB, 1.5 MB of the same shape in 0.6 s — bounded by the depth guard like + in 1.3 s at 16.5 kB, 1.5 MB of that shape at 250 levels in 0.6 s — bounded by the depth guard like `readNestedDirective` (the maintainer, 2026-09-18). - [ ] **4d — What the gate says while it runs (`0.2.1`).** `ci.sh` runs nine legs and announces none of them, so five minutes of a Gitea run read as silence and a hang cannot be told from