4c - the guard's spread was a class: the code block's held lines and a mark run's segments too
This commit is contained in:
@@ -297,10 +297,17 @@ someone spells it or pins it.
|
|||||||
has — for the fallback to refuse at zero rather than walk again; counting every list twice
|
has — for the fallback to refuse at zero rather than walk again; counting every list twice
|
||||||
halved the list limit, counting the directive form once doubled the parser's frames per level
|
halved the list limit, counting the directive form once doubled the parser's frames per level
|
||||||
(the maintainer, 2026-09-18).
|
(the maintainer, 2026-09-18).
|
||||||
|
- Nothing spreads an unbounded array into a call — a node's siblings, a code block's held lines, a
|
||||||
|
mark run's segments: the argument list caps near 125k and throws a `RangeError` where a `Result`
|
||||||
|
is owed. A walk pushes one at a time. A literal spread (`[...value]`) is not the same thing and
|
||||||
|
is fine (4c).
|
||||||
- A reader takes the text and an index — a sticky regex whose `lastIndex` the caller sets on the
|
- A reader takes the text and an index — a sticky regex whose `lastIndex` the caller sets on the
|
||||||
line before it reads, `indexOf` — never a fresh slice per character, and a per-character walk
|
line before it reads, `indexOf` — never a fresh slice per character, and a per-character walk
|
||||||
hoists the scan that does not vary with the character. The pipeline persona feeds documents
|
hoists the scan that does not vary with the character. The pipeline persona feeds documents
|
||||||
nobody typed, and a megabyte through a quadratic walk is a minute rather than a millisecond.
|
nobody typed, and a megabyte through a quadratic walk is a minute rather than a millisecond. A
|
||||||
|
scan may keep what it read for a later walk of the same text, and the fallback where it kept
|
||||||
|
nothing must be the same reader over the same text at the same index, so the two cannot disagree
|
||||||
|
— which is what makes the kept value a memo rather than a second spelling (4c).
|
||||||
- No casts: `as`, `as unknown as`, non-null `!`. A boundary owes a type guard validating the
|
- No casts: `as`, `as unknown as`, non-null `!`. A boundary owes a type guard validating the
|
||||||
fields it claims (`isAdfDocument`); past it everything is typed. Make invalid states
|
fields it claims (`isAdfDocument`); past it everything is typed. Make invalid states
|
||||||
unrepresentable.
|
unrepresentable.
|
||||||
|
|||||||
@@ -174,7 +174,6 @@ export function isThematicBreak(line: string): boolean {
|
|||||||
return holdsThematicBreak(thematicBreakTail(line), line)
|
return holdsThematicBreak(thematicBreakTail(line), line)
|
||||||
}
|
}
|
||||||
|
|
||||||
// A break runs to the line's end, so one scan of that run answers every level the list walk opens.
|
|
||||||
export function thematicBreakTail(text: string): ThematicBreakTail | undefined {
|
export function thematicBreakTail(text: string): ThematicBreakTail | undefined {
|
||||||
let marker: string | undefined
|
let marker: string | undefined
|
||||||
let markers = 0
|
let markers = 0
|
||||||
@@ -195,6 +194,8 @@ export function thematicBreakTail(text: string): ThematicBreakTail | undefined {
|
|||||||
return { longest: text.length - first, marker, shortest: text.length - third }
|
return { longest: text.length - first, marker, shortest: text.length - third }
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// `text` is a suffix of what `tail` was read from, or opens with a space: a length alone names a
|
||||||
|
// suffix, and the marker check is what refuses one starting mid-run — `--- ---` holds ` ---`.
|
||||||
export function holdsThematicBreak(tail: ThematicBreakTail | undefined, text: string): boolean {
|
export function holdsThematicBreak(tail: ThematicBreakTail | undefined, text: string): boolean {
|
||||||
if (tail === undefined) return false
|
if (tail === undefined) return false
|
||||||
return text.length >= tail.shortest && text.length <= tail.longest && text.charAt(0) === tail.marker
|
return text.length >= tail.shortest && text.length <= tail.longest && text.charAt(0) === tail.marker
|
||||||
|
|||||||
@@ -17,7 +17,7 @@ export type DirectiveLine =
|
|||||||
| { argument: string | undefined; attributes: DirectiveAttributes; kind: 'opener'; name: string }
|
| { argument: string | undefined; attributes: DirectiveAttributes; kind: 'opener'; name: string }
|
||||||
| { kind: 'closer'; name: string }
|
| { kind: 'closer'; name: string }
|
||||||
|
|
||||||
// spans: the directives the content holds, at their offset in it, so reading the slot back never scans them again.
|
// spans keys index content, so the two travel together: separate them and every offset is wrong.
|
||||||
export type DirectiveSpan = { attributes: DirectiveAttributes; content: string | undefined; length: number; name: string; spans: NestedSpans }
|
export type DirectiveSpan = { attributes: DirectiveAttributes; content: string | undefined; length: number; name: string; spans: NestedSpans }
|
||||||
|
|
||||||
export type NestedSpans = ReadonlyMap<number, DirectiveSpan>
|
export type NestedSpans = ReadonlyMap<number, DirectiveSpan>
|
||||||
@@ -254,10 +254,11 @@ function readDirectiveContent(text: string, start: number, depth: number): Read<
|
|||||||
cursor = span
|
cursor = span
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
const nested = keepNestedSpan(text, cursor, depth, start, spans)
|
const nested = readNestedDirective(text, cursor, depth + 1)
|
||||||
if (nested?.fault !== undefined) return { fault: nested.fault }
|
if (nested?.fault !== undefined) return { fault: nested.fault }
|
||||||
if (nested !== undefined) {
|
if (nested !== undefined) {
|
||||||
cursor = nested.value
|
spans.set(cursor - start, nested.value)
|
||||||
|
cursor += nested.value.length
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if (character === ']' && brackets === 0) return { value: { end: cursor, spans } }
|
if (character === ']' && brackets === 0) return { value: { end: cursor, spans } }
|
||||||
@@ -268,14 +269,6 @@ function readDirectiveContent(text: string, start: number, depth: number): Read<
|
|||||||
return { fault: malformedDirective(`an inline directive [content] is unclosed; ${directiveEscape}`) }
|
return { fault: malformedDirective(`an inline directive [content] is unclosed; ${directiveEscape}`) }
|
||||||
}
|
}
|
||||||
|
|
||||||
function keepNestedSpan(text: string, cursor: number, depth: number, start: number, spans: Map<number, DirectiveSpan>): Read<number> | undefined {
|
|
||||||
const nested = readNestedDirective(text, cursor, depth + 1)
|
|
||||||
if (nested === undefined) return undefined
|
|
||||||
if (nested.fault !== undefined) return { fault: nested.fault }
|
|
||||||
spans.set(cursor - start, nested.value)
|
|
||||||
return { value: cursor + nested.value.length }
|
|
||||||
}
|
|
||||||
|
|
||||||
// `undefined` where the span crosses the line ending an inline directive may not cross.
|
// `undefined` where the span crosses the line ending an inline directive may not cross.
|
||||||
function readCodeSpanEnd(text: string, index: number): number | undefined {
|
function readCodeSpanEnd(text: string, index: number): number | undefined {
|
||||||
const opener = backtickRun(text, index)
|
const opener = backtickRun(text, index)
|
||||||
|
|||||||
@@ -736,3 +736,8 @@ test('carries whitespace CommonMark strips in the reserved text directive', () =
|
|||||||
// CommonMark strips spaces and tabs alone, so the whitespace beside them is plain text.
|
// CommonMark strips spaces and tabs alone, so the whitespace beside them is plain text.
|
||||||
assert.equal(emitted({ text: '\va\f', type: 'text' }), '\va\f\n')
|
assert.equal(emitted({ text: '\va\f', type: 'text' }), '\va\f\n')
|
||||||
})
|
})
|
||||||
|
|
||||||
|
test("joins a mark run's segments as a walk rather than as one call's arguments", () => {
|
||||||
|
const run = Array.from({ length: 200000 }, (): AdfNode => ({ marks: [{ type: 'strong' }], text: 'a', type: 'text' }))
|
||||||
|
assert.equal(markdown(adfToMarkdown(document({ content: run, type: 'paragraph' }))), `**${'a'.repeat(200000)}**\n`)
|
||||||
|
})
|
||||||
|
|||||||
@@ -166,7 +166,7 @@ function emitRun(nodes: readonly AdfNode[], depth: number, firstIndex: number, c
|
|||||||
const emitted = run.kind === 'plain' ? emitLeaf(run.node, runContext, run.index) : emitMarkedRun(run.nodes, run.mark, depth, run.index, runContext)
|
const emitted = run.kind === 'plain' ? emitLeaf(run.node, runContext, run.index) : emitMarkedRun(run.nodes, run.mark, depth, run.index, runContext)
|
||||||
if (!emitted.ok) return emitted
|
if (!emitted.ok) return emitted
|
||||||
if (emitted.value.carry !== undefined) return emitted
|
if (emitted.value.carry !== undefined) return emitted
|
||||||
segments.push(...emitted.value.segments)
|
for (const segment of emitted.value.segments) segments.push(segment)
|
||||||
}
|
}
|
||||||
return success({ segments })
|
return success({ segments })
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -118,3 +118,7 @@ test('closes the containers a closer names past as unclosed, and crosses no list
|
|||||||
assert.deepEqual(kinds('!adf:rule {localId=a-1}\nPart.\n!adf:/rule\n'), ['fault', 'paragraph'])
|
assert.deepEqual(kinds('!adf:rule {localId=a-1}\nPart.\n!adf:/rule\n'), ['fault', 'paragraph'])
|
||||||
assert.deepEqual(kinds('!adf:rule {localId=a-1}\n!adf:/rule\n!adf:/rule\n'), ['fault', 'fault'])
|
assert.deepEqual(kinds('!adf:rule {localId=a-1}\n!adf:/rule\n!adf:/rule\n'), ['fault', 'fault'])
|
||||||
})
|
})
|
||||||
|
|
||||||
|
test("releases an indented code block's held blank lines as a walk rather than as one call's arguments", () => {
|
||||||
|
assert.deepEqual(kinds(` a\n${'\n'.repeat(200000)} b\n`), ['code'])
|
||||||
|
})
|
||||||
|
|||||||
@@ -365,7 +365,8 @@ function readIndentedCodeLine(leaf: Extract<OpenLeaf, { kind: 'indented-code' }>
|
|||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
if (leadingColumns(line) < indentedCodeColumns) return false
|
if (leadingColumns(line) < indentedCodeColumns) return false
|
||||||
leaf.lines.push(...leaf.held, removeColumns(line, indentedCodeColumns).text)
|
for (const held of leaf.held) leaf.lines.push(held)
|
||||||
|
leaf.lines.push(removeColumns(line, indentedCodeColumns).text)
|
||||||
leaf.held.length = 0
|
leaf.held.length = 0
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -534,6 +534,11 @@ The done `todo.md` items in full, as they were written. `todo.md` keeps a one-li
|
|||||||
38 ms where the tail scan alone would have left it quadratic.
|
38 ms where the tail scan alone would have left it quadratic.
|
||||||
The sixth site the sweep found went to 18 rather than landing here (the maintainer,
|
The sixth site the sweep found went to 18 rather than landing here (the maintainer,
|
||||||
2026-09-18).
|
2026-09-18).
|
||||||
|
**Widened** (the systems-architect, 2026-09-18): the guard's spread was a class rather than a
|
||||||
|
site, and two more threw out of the public API — `readIndentedCodeLine` releasing the blank
|
||||||
|
lines an indented code block held (200k of them at 200 kB), and `emitRun` joining a mark
|
||||||
|
run's segments (200k nodes under one mark). Both fixed here with the same loop and a test
|
||||||
|
each, and §11 gained the rule so the spelling cannot walk back in.
|
||||||
- [x] **5a — Rename to `@larvit/adf-codec` (`0.1.0`).** Before the first publish, the name being
|
- [x] **5a — Rename to `@larvit/adf-codec` (`0.1.0`).** Before the first publish, the name being
|
||||||
the published identity: `package.json` `name` and `repository`, the Gitea repo and its
|
the published identity: `package.json` `name` and `repository`, the Gitea repo and its
|
||||||
remote, the README title, §6's published-as line, the checkout directory.
|
remote, the README title, §6's published-as line, the checkout directory.
|
||||||
|
|||||||
Reference in New Issue
Block a user