diff --git a/AGENTS.md b/AGENTS.md index f3017b0..aaaa56a 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -67,7 +67,8 @@ deliberate re-pin, exceptions re-derived by hand beside it. Atlassian's ADF JSON vendored the same way, at `spec/adf-schema/`, rather than as the `@atlaskit/adf-schema` dev dependency — CommonJS-only, some fifty packages with React among them, and a release most days for Renovate to automerge — re-pinned by hand when a payload or a report shows the need. -`devDependencies`: few, each earning its keep; they never reach a consumer. +`devDependencies`: few, each earning its keep; they never reach a consumer. `fast-check` earns its +place shrinking a failing generated document to the nodes that break it. ## 6. The package contract @@ -225,6 +226,10 @@ compared against `undefined` — have a half no valid document reaches. The corpus, all checked in: hand-built fixtures per node and combination; real sanitized ADF from live Atlassian APIs; the CommonMark spec suite against `markdownToAdf` and `markdownToHtml`. +Beside the corpus, properties run over documents generated from the node tables, on a fixed seed in +the gate; `PROPERTY_RUNS=` raises the runs and randomizes the seed for local digging, and a +counterexample found becomes a round-trip fixture. + `spec/flavour.md` is read as a source too, so the node tables cannot drift from the prose they copy: each `- ` bullet in `## Block nodes`, `## Inline nodes` and `## Marks` declares the nodes named before its first em dash, with the attributes following `Attributes: ` — a parenthesized diff --git a/ci.sh b/ci.sh index 5ab9650..dd3d712 100755 --- a/ci.sh +++ b/ci.sh @@ -16,7 +16,7 @@ if printf '%s' "$test_output" | grep -q 'ℹ tests 0'; then exit 1 fi -in_image "$deno_image" deno test --allow-read --no-check src/ +in_image "$deno_image" deno test --allow-env=PROPERTY_RUNS --allow-read --no-check src/ in_image "$bun_image" bun test src/ in_image "$node_image" npm run build diff --git a/corpus/round-trip/combinations/directive-autolink-backtick.json b/corpus/round-trip/combinations/directive-autolink-backtick.json new file mode 100644 index 0000000..2ddbd7f --- /dev/null +++ b/corpus/round-trip/combinations/directive-autolink-backtick.json @@ -0,0 +1,50 @@ +{ + "content": [ + { + "content": [ + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": "http://a`b" + }, + "type": "link" + } + ], + "text": "http://a`b", + "type": "text" + }, + { + "text": "`c", + "type": "text" + } + ], + "type": "paragraph" + }, + { + "content": [ + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": "http://d`e" + }, + "type": "link" + } + ], + "text": "http://d`e", + "type": "text" + } + ], + "type": "paragraph" + } + ], + "type": "doc", + "version": 1 +} diff --git a/corpus/round-trip/combinations/directive-autolink-backtick.md b/corpus/round-trip/combinations/directive-autolink-backtick.md new file mode 100644 index 0000000..163f313 --- /dev/null +++ b/corpus/round-trip/combinations/directive-autolink-backtick.md @@ -0,0 +1,3 @@ +:underline[[http://a\`b](http://a\`b)]`c + +:underline[[http://d\`e](http://d`e)] diff --git a/corpus/round-trip/combinations/directive-link-backticks.json b/corpus/round-trip/combinations/directive-link-backticks.json new file mode 100644 index 0000000..00c8b53 --- /dev/null +++ b/corpus/round-trip/combinations/directive-link-backticks.json @@ -0,0 +1,83 @@ +{ + "content": [ + { + "content": [ + { + "marks": [ + { + "attrs": { + "color": "" + }, + "type": "textColor" + }, + { + "attrs": { + "href": "`" + }, + "type": "link" + } + ], + "text": " ", + "type": "text" + }, + { + "text": "`a", + "type": "text" + } + ], + "type": "paragraph" + }, + { + "content": [ + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": "c", + "title": "`" + }, + "type": "link" + } + ], + "text": "b", + "type": "text" + }, + { + "marks": [ + { + "type": "code" + } + ], + "text": "c", + "type": "text" + } + ], + "type": "paragraph" + }, + { + "content": [ + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": "e`f" + }, + "type": "link" + } + ], + "text": "d", + "type": "text" + } + ], + "type": "paragraph" + } + ], + "type": "doc", + "version": 1 +} diff --git a/corpus/round-trip/combinations/directive-link-backticks.md b/corpus/round-trip/combinations/directive-link-backticks.md new file mode 100644 index 0000000..85d19f6 --- /dev/null +++ b/corpus/round-trip/combinations/directive-link-backticks.md @@ -0,0 +1,5 @@ +:textColor[[ ](\`)]{color=""}`a + +:underline[[b](c "\`")]`c` + +:underline[[d](e`f)] diff --git a/corpus/round-trip/combinations/directive-link-brackets.json b/corpus/round-trip/combinations/directive-link-brackets.json new file mode 100644 index 0000000..850d741 --- /dev/null +++ b/corpus/round-trip/combinations/directive-link-brackets.json @@ -0,0 +1,111 @@ +{ + "content": [ + { + "content": [ + { + "marks": [ + { + "type": "subsup" + }, + { + "attrs": { + "href": "[" + }, + "type": "link" + } + ], + "text": "a", + "type": "text" + } + ], + "type": "paragraph" + }, + { + "content": [ + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": "]" + }, + "type": "link" + } + ], + "text": "b", + "type": "text" + }, + { + "marks": [ + { + "type": "underline" + } + ], + "text": " ", + "type": "text" + }, + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": "x", + "title": "[" + }, + "type": "link" + } + ], + "text": "c", + "type": "text" + } + ], + "type": "paragraph" + }, + { + "content": [ + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": "e[f]" + }, + "type": "link" + } + ], + "text": "d", + "type": "text" + } + ], + "type": "paragraph" + }, + { + "content": [ + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": "http://a[b" + }, + "type": "link" + } + ], + "text": "http://a[b", + "type": "text" + } + ], + "type": "paragraph" + } + ], + "type": "doc", + "version": 1 +} diff --git a/corpus/round-trip/combinations/directive-link-brackets.md b/corpus/round-trip/combinations/directive-link-brackets.md new file mode 100644 index 0000000..e677943 --- /dev/null +++ b/corpus/round-trip/combinations/directive-link-brackets.md @@ -0,0 +1,7 @@ +:subsup[[a](\[)] + +:underline[[b](\]) [c](x "\[")] + +:underline[[d](e[f])] + +:underline[[http://a\[b](http://a\[b)] diff --git a/corpus/round-trip/combinations/directive-link-colon.json b/corpus/round-trip/combinations/directive-link-colon.json new file mode 100644 index 0000000..2cc9c86 --- /dev/null +++ b/corpus/round-trip/combinations/directive-link-colon.json @@ -0,0 +1,107 @@ +{ + "content": [ + { + "content": [ + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": ":a{" + }, + "type": "link" + } + ], + "text": "x", + "type": "text" + } + ], + "type": "paragraph" + }, + { + "content": [ + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": "y", + "title": ":a{" + }, + "type": "link" + } + ], + "text": "x", + "type": "text" + } + ], + "type": "paragraph" + }, + { + "content": [ + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": ":a[]{" + }, + "type": "link" + } + ], + "text": "x", + "type": "text" + } + ], + "type": "paragraph" + }, + { + "content": [ + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": "https://example.com/?q=:emoji{" + }, + "type": "link" + } + ], + "text": "x", + "type": "text" + } + ], + "type": "paragraph" + }, + { + "content": [ + { + "marks": [ + { + "type": "underline" + }, + { + "attrs": { + "href": "ab:c{" + }, + "type": "link" + } + ], + "text": "ab:c{", + "type": "text" + } + ], + "type": "paragraph" + } + ], + "type": "doc", + "version": 1 +} diff --git a/corpus/round-trip/combinations/directive-link-colon.md b/corpus/round-trip/combinations/directive-link-colon.md new file mode 100644 index 0000000..ecf5427 --- /dev/null +++ b/corpus/round-trip/combinations/directive-link-colon.md @@ -0,0 +1,9 @@ +:underline[[x](\:a{)] + +:underline[[x](y "\:a{")] + +:underline[[x](\:a[]{)] + +:underline[[x](https://example.com/?q=\:emoji{)] + +:underline[[ab\:c{](ab\:c{)] diff --git a/corpus/round-trip/combinations/list-item-whitespace-line.json b/corpus/round-trip/combinations/list-item-whitespace-line.json new file mode 100644 index 0000000..e1ffdb8 --- /dev/null +++ b/corpus/round-trip/combinations/list-item-whitespace-line.json @@ -0,0 +1,57 @@ +{ + "content": [ + { + "content": [ + { + "content": [ + { + "content": [ + { + "text": " \na", + "type": "text" + } + ], + "type": "codeBlock" + } + ], + "type": "listItem" + } + ], + "type": "bulletList" + }, + { + "attrs": { + "order": 1 + }, + "content": [ + { + "content": [ + { + "content": [ + { + "content": [ + { + "content": [ + { + "text": "\t", + "type": "text" + } + ], + "type": "codeBlock" + } + ], + "type": "listItem" + } + ], + "type": "bulletList" + } + ], + "type": "listItem" + } + ], + "type": "orderedList" + } + ], + "type": "doc", + "version": 1 +} diff --git a/corpus/round-trip/combinations/list-item-whitespace-line.md b/corpus/round-trip/combinations/list-item-whitespace-line.md new file mode 100644 index 0000000..f4b0d02 --- /dev/null +++ b/corpus/round-trip/combinations/list-item-whitespace-line.md @@ -0,0 +1,20 @@ +::::bulletList +:::listItem +``` + +a +``` +::: +:::: + +::::::orderedList {order=1} +:::::listItem +::::bulletList +:::listItem +``` + +``` +::: +:::: +::::: +:::::: diff --git a/corpus/round-trip/commonmark-subset/link-empty-destination.json b/corpus/round-trip/commonmark-subset/link-empty-destination.json new file mode 100644 index 0000000..9cad883 --- /dev/null +++ b/corpus/round-trip/commonmark-subset/link-empty-destination.json @@ -0,0 +1,24 @@ +{ + "content": [ + { + "content": [ + { + "marks": [ + { + "attrs": { + "href": "", + "title": "" + }, + "type": "link" + } + ], + "text": "a", + "type": "text" + } + ], + "type": "paragraph" + } + ], + "type": "doc", + "version": 1 +} diff --git a/corpus/round-trip/commonmark-subset/link-empty-destination.md b/corpus/round-trip/commonmark-subset/link-empty-destination.md new file mode 100644 index 0000000..95ed36f --- /dev/null +++ b/corpus/round-trip/commonmark-subset/link-empty-destination.md @@ -0,0 +1 @@ +[a](<> "") diff --git a/docker-runner.sh b/docker-runner.sh index 504109f..8602e71 100644 --- a/docker-runner.sh +++ b/docker-runner.sh @@ -7,7 +7,7 @@ node_image=node:24.19.0-alpine3.24 in_image() { local image=$1 entrypoint=$2 shift 2 - docker run --rm -u "$(id -u):$(id -g)" -e HOME=/tmp ${in_image_network:+--network "$in_image_network"} -v "$PWD:/app" -w /app --entrypoint "$entrypoint" "$image" "$@" + docker run --rm -u "$(id -u):$(id -g)" -e HOME=/tmp ${PROPERTY_RUNS+-e PROPERTY_RUNS} ${in_image_network:+--network "$in_image_network"} -v "$PWD:/app" -w /app --entrypoint "$entrypoint" "$image" "$@" } with_firefox() { diff --git a/package-lock.json b/package-lock.json index 54054ed..d7eeb19 100644 --- a/package-lock.json +++ b/package-lock.json @@ -10,6 +10,7 @@ "license": "MIT", "devDependencies": { "@types/node": "24.13.3", + "fast-check": "4.10.0", "typescript": "7.0.2" }, "engines": { @@ -366,6 +367,46 @@ "node": ">=16.20.0" } }, + "node_modules/fast-check": { + "version": "4.10.0", + "resolved": "https://registry.npmjs.org/fast-check/-/fast-check-4.10.0.tgz", + "integrity": "sha512-hhqQL+IJllZi3aM4TKvmCj3bywLEcycNTTLZeLhA9ttMxBrCqM07q7Di4kl+j9EWSTXvJH1+EpIgsDbF/+8H5Q==", + "dev": true, + "funding": [ + { + "type": "individual", + "url": "https://github.com/sponsors/dubzzz" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/fast-check" + } + ], + "license": "MIT", + "dependencies": { + "pure-rand": "^8.0.0" + }, + "engines": { + "node": ">=12.17.0" + } + }, + "node_modules/pure-rand": { + "version": "8.4.2", + "resolved": "https://registry.npmjs.org/pure-rand/-/pure-rand-8.4.2.tgz", + "integrity": "sha512-vvuOGgcuPJAirlHvuQw1TrOiw7ptaIXXmIbNuiNOY6lNGJJH49PQ1Kj4nd783nPdQhQdicgOjVI2yI/9BD6/Ng==", + "dev": true, + "funding": [ + { + "type": "individual", + "url": "https://github.com/sponsors/dubzzz" + }, + { + "type": "opencollective", + "url": "https://opencollective.com/fast-check" + } + ], + "license": "MIT" + }, "node_modules/typescript": { "version": "7.0.2", "resolved": "https://registry.npmjs.org/typescript/-/typescript-7.0.2.tgz", diff --git a/package.json b/package.json index 7a6670c..a188be2 100644 --- a/package.json +++ b/package.json @@ -28,6 +28,7 @@ }, "devDependencies": { "@types/node": "24.13.3", + "fast-check": "4.10.0", "typescript": "7.0.2" } } diff --git a/spec/flavour.md b/spec/flavour.md index 853f94b..c0f802b 100644 --- a/spec/flavour.md +++ b/spec/flavour.md @@ -28,7 +28,8 @@ normalizes to it through the round-trip. adjacent lists of a kind back as one. The leaf `::listBreak` parts them, taking the separation any directive block takes where it sits. It builds no node, and it reads only between two adjacent lists of one type: elsewhere, or carrying an argument, `{attrs}` or a body, it is a - named error. + named error. A list whose item holds a line of spaces or tabs alone, which a list item reads + back empty, takes the directive form. - Blockquotes prefix lines with `> `; a blank line inside a blockquote is a bare `>`. - ATX headings (`#` … `######`); setext input normalizes to ATX. - Code fences ``` with the node's language as info string, the fence lengthened past any backtick @@ -43,10 +44,12 @@ normalizes to it through the round-trip. CommonMark admits no spelling — the end of a block, inside an ATX heading — or where the node carries an attribute, it is the inline directive. - An empty paragraph — real payloads carry them — is `::paragraph`. -- Links `[text](url)`; `<…>` around a destination containing spaces; title in double quotes. A - backslash escapes a parenthesis the destination leaves unbalanced, and a quote inside the title; - a balanced pair stays bare. `` autolink form only when the text equals the destination and - the destination is a valid CommonMark autolink (absolute URI). +- Links `[text](url)`; `<…>` around a destination containing spaces, `<>` an empty one beside a + title; title in double quotes. A backslash escapes a parenthesis the destination leaves + unbalanced, and a quote inside the title; a balanced pair stays bare. `` autolink form only + when the text equals the destination and the destination is a valid CommonMark autolink + (absolute URI) — inside an inline directive's `[content]`, one holding no backtick, no + unbalanced bracket and no inline directive opener. - Paragraphs on one line — no soft wrapping; a soft line break in input becomes a single space. - Entity references in input decode to their characters; output backslash-escapes only where text would otherwise parse as syntax, scanning the assembled line rather than each text node: escape @@ -128,9 +131,11 @@ quoted where bare carries it, an escape longer than it need be, an empty `{attrs each a named error naming the spelling to write instead. **Escaping**: the emitter backslash-escapes whatever literal text would otherwise parse as -directive syntax — the leading `:` of a would-be directive, `]` inside content, a `{` right -after a directive's closing `]`, which would otherwise be read as the attributes it has none -of; outside code spans and code blocks, a backslash before `:` in input yields a literal colon. +directive syntax — the leading `:` of a would-be directive, `]` inside content, a bracket a link's +destination and title inside content leave unbalanced, a backtick there that would open a code span +and a `:` there that would open an inline directive, a `{` right after a directive's closing `]`, +which would otherwise be read as the attributes it has none of; outside code spans and code blocks, +a backslash before `:` in input yields a literal colon. **Malformed directives are error results**, named: an unclosed container at end of input, a body fence line of the container's length or longer, a bare colon-run line outside any container or diff --git a/src/adf-property.test.ts b/src/adf-property.test.ts new file mode 100644 index 0000000..4343cb1 --- /dev/null +++ b/src/adf-property.test.ts @@ -0,0 +1,183 @@ +import fc from 'fast-check' +import assert from 'node:assert/strict' +import { env } from 'node:process' +import test from 'node:test' + +import type { AdfAttributes, AdfDocument, AdfMark, AdfNode } from './adf/document.ts' +import type { Arbitrary } from 'fast-check' +import type { AttributeKind, AttributeVocabulary } from './adf/attribute-vocabulary.ts' +import type { JsonValue } from './json-value.ts' +import { adfToMarkdown } from './markdown/emit/adf-to-markdown.ts' +import { blockArgument } from './markdown/block-directive-arguments.ts' +import { blockDirectives } from './adf/block-directives.ts' +import { inlineDirectives } from './adf/inline-directives.ts' +import { markAttributes } from './adf/mark-attributes.ts' +import { markdownToAdf } from './markdown/parse/markdown-to-adf.ts' +import { toEditorNormal } from './adf/editor-normal.ts' + +type Positions = { block: AdfNode; inline: AdfNode } + +const deepRunsVariable = 'PROPERTY_RUNS' +const gateRuns = 1600 +const gateSeed = 20260914 +// Bun's test runner stops a test after five seconds unless the test sets its own timeout. +const propertyTimeout = 600000 + +const depthIdentifier = fc.createDepthIdentifier() +const emptyCell: AdfNode = { content: [{ type: 'paragraph' }], type: 'tableCell' } +const flatCommonMarkShapeWeight = 4 +const markdownPieces = fc.constantFrom(...'aZ09 \t\n!"#$%&\'()*+,-./:;<=>?@[\\]^_`{|}~é\xa0🎉', ':a[', ':a{', 'ab:', 'http://') +const nestingCommonMarkShapeWeight = 21 +const spelledTypes = new Set(['text', ...Object.keys(blockDirectives), ...Object.keys(inlineDirectives), ...Object.keys(markAttributes)]) + +function textOf(minLength: number): Arbitrary { + return fc.oneof( + { arbitrary: fc.string({ maxLength: 12, minLength, unit: markdownPieces }), weight: 4 }, + { arbitrary: fc.string({ maxLength: 6, minLength, unit: 'grapheme' }), weight: 1 }, + ) +} + +const text = textOf(1) +const unknownType = fc.oneof(fc.stringMatching(/^[a-z][A-Za-z0-9]{0,7}$/), text).filter((type) => !spelledTypes.has(type)) +const numberValue = fc.oneof({ arbitrary: fc.integer({ max: 10, min: -1 }), weight: 3 }, { arbitrary: fc.double({ noDefaultInfinity: true, noNaN: true }), weight: 1 }) + +// V8's JSON.parse returns a wrong key after parsing a key holding an escaped backslash (https://issues.chromium.org/issues/521080746); Bun is unaffected. +const keyPiece = fc + .oneof({ arbitrary: markdownPieces, weight: 4 }, { arbitrary: fc.string({ maxLength: 1, minLength: 1, unit: 'grapheme' }), weight: 1 }) + .filter((piece) => !/[\\"\x00-\x1f]/.test(piece)) +const jsonKey = fc.string({ maxLength: 8, unit: keyPiece }) + +const { jsonValue } = fc.letrec<{ jsonValue: JsonValue }>((tie) => ({ + jsonValue: fc.oneof( + { depthSize: 'small', maxDepth: 2 }, + fc.oneof(fc.constant(null), fc.boolean(), numberValue, textOf(0)), + fc.array(tie('jsonValue'), { maxLength: 3 }), + fc.dictionary(jsonKey, tie('jsonValue'), { maxKeys: 3, noNullPrototype: true }), + ), +})) + +const valueByKind: Readonly>> = { + boolean: fc.boolean(), + json: jsonValue, + number: numberValue, + string: textOf(0), +} + +function attributes(vocabulary: AttributeVocabulary): Arbitrary { + const model = Object.fromEntries( + Object.entries(vocabulary).map(([key, kind]) => [key, fc.oneof({ arbitrary: fc.constant(undefined), weight: 2 }, { arbitrary: valueByKind[kind], weight: 1 })]), + ) + return fc.record(model).map(heldAttributes) +} + +function heldAttributes(held: Readonly>): AdfAttributes { + const attrs: AdfAttributes = {} + for (const [key, value] of Object.entries(held)) if (value !== undefined) attrs[key] = value + return attrs +} + +function pipeTable({ body, header }: { body: AdfNode[][]; header: AdfNode[] }): AdfNode { + const rows = [header, ...body.map((cells) => header.map((_, index) => cells[index] ?? emptyCell))] + return { content: rows.map((content): AdfNode => ({ content, type: 'tableRow' })), type: 'table' } +} + +const mark: Arbitrary = fc.oneof( + { arbitrary: fc.oneof(...Object.entries(markAttributes).map(([type, vocabulary]) => attributes(vocabulary).map((attrs) => ({ attrs, type })))), weight: 9 }, + { arbitrary: fc.record({ attrs: fc.dictionary(jsonKey, jsonValue, { maxKeys: 2, noNullPrototype: true }), type: unknownType }), weight: 1 }, +) +const marks = fc.uniqueArray(mark, { maxLength: 3, selector: (held) => held.type }) + +const textNode = fc.record({ marks, text }).map((held): AdfNode => ({ ...held, type: 'text' })) + +const autolinkTextNode = fc + .record({ href: fc.tuple(fc.constantFrom('ab:', 'http://'), textOf(0)).map(([scheme, rest]) => `${scheme}${rest}`), marks }) + .map(({ href, marks: held }): AdfNode => ({ marks: [...held.filter((outer) => outer.type !== 'link'), { attrs: { href }, type: 'link' }], text: href, type: 'text' })) + +const inlineNodes = Object.entries(inlineDirectives).map(([type, directive]) => + fc.record({ attrs: attributes(directive.attributes), marks }).map((held): AdfNode => ({ ...held, type })), +) + +function weighted(arbitraries: readonly Arbitrary[], weight: number): { arbitrary: Arbitrary; weight: number }[] { + return arbitraries.map((arbitrary) => ({ arbitrary, weight })) +} + +const positions = fc.letrec((tie) => { + const blockContent = fc.array(tie('block'), { depthIdentifier, maxLength: 3 }) + const inlineContent = fc.array(tie('inline'), { depthIdentifier, maxLength: 4 }) + const contentByModel = { + block: blockContent, + code: fc.array(text.map((held): AdfNode => ({ text: held, type: 'text' })), { maxLength: 2 }), + inline: inlineContent, + none: fc.constant([]), + } + const blockMarks = fc.oneof({ arbitrary: fc.constant([]), weight: 4 }, { arbitrary: marks, weight: 1 }) + const blockNodes = Object.entries(blockDirectives).map(([type, directive]) => { + const argument = blockArgument(type) + const vocabulary: AttributeVocabulary = argument === undefined ? directive.attributes : { ...directive.attributes, [argument]: 'string' } + const node = fc.record({ attrs: attributes(vocabulary), content: contentByModel[directive.contentModel], marks: blockMarks }).map((held): AdfNode => ({ ...held, type })) + return { leaf: directive.contentModel === 'code' || directive.contentModel === 'none', node } + }) + const unknownNode = fc + .record({ attrs: fc.dictionary(jsonKey, jsonValue, { maxKeys: 2, noNullPrototype: true }), content: fc.array(tie('inline'), { depthIdentifier, maxLength: 2 }), marks, type: unknownType }) + .map((held): AdfNode => held) + const leafBlocks = blockNodes.filter((entry) => entry.leaf).map((entry) => entry.node) + const containerBlocks = blockNodes.filter((entry) => !entry.leaf).map((entry) => entry.node) + const misplacedWeight = 7 + const paragraph = inlineContent.map((content): AdfNode => ({ content, type: 'paragraph' })) + const cell = (type: string) => paragraph.map((held): AdfNode => ({ content: [held], type })) + const listItems = fc.array( + blockContent.map((content): AdfNode => ({ content, type: 'listItem' })), + { depthIdentifier, maxLength: 3, minLength: 1 }, + ) + const flatCommonMarkShapes = [ + fc.record({ content: inlineContent, level: fc.integer({ max: 6, min: 1 }) }).map(({ content, level }): AdfNode => ({ attrs: { level }, content, type: 'heading' })), + paragraph, + fc.record({ body: fc.array(fc.array(cell('tableCell'), { maxLength: 3 }), { maxLength: 2 }), header: fc.array(cell('tableHeader'), { maxLength: 3, minLength: 1 }) }).map(pipeTable), + ] + const nestingCommonMarkShapes = [ + blockContent.map((content): AdfNode => ({ content, type: 'blockquote' })), + listItems.map((content): AdfNode => ({ content, type: 'bulletList' })), + fc + .record({ content: listItems, order: fc.oneof({ arbitrary: fc.integer({ max: 3, min: 0 }), weight: 4 }, { arbitrary: fc.integer({ max: 999999999, min: 0 }), weight: 1 }) }) + .map(({ content, order }): AdfNode => ({ attrs: { order }, content, type: 'orderedList' })), + ] + const flatBlocks = [...weighted(leafBlocks, 2), ...weighted(flatCommonMarkShapes, flatCommonMarkShapeWeight)] + return { + block: fc.oneof( + { depthIdentifier, depthSize: 'small', maxDepth: 4 }, + { arbitrary: fc.oneof(...flatBlocks), weight: flatBlocks.reduce((sum, entry) => sum + entry.weight, 0) }, + { arbitrary: fc.oneof(...containerBlocks), weight: containerBlocks.length * 2 }, + { arbitrary: fc.oneof(textNode, ...inlineNodes, unknownNode), weight: misplacedWeight }, + { arbitrary: fc.oneof(...nestingCommonMarkShapes), weight: nestingCommonMarkShapes.length * nestingCommonMarkShapeWeight }, + ), + inline: fc.oneof( + { depthIdentifier, depthSize: 'small', maxDepth: 4 }, + { arbitrary: textNode, weight: 12 }, + { arbitrary: autolinkTextNode, weight: 2 }, + { arbitrary: fc.oneof(...inlineNodes), weight: 7 }, + { arbitrary: fc.oneof(...blockNodes.map((entry) => entry.node), unknownNode), weight: 2 }, + ), + } +}) + +const adfDocument = fc.array(positions.block, { depthIdentifier, maxLength: 4, minLength: 1 }).map((content): AdfDocument => toEditorNormal({ content, type: 'doc', version: 1 })) + +function runParameters(): { numRuns: number; seed?: number } { + const deepRuns = env[deepRunsVariable] + if (deepRuns === undefined) return { numRuns: gateRuns, seed: gateSeed } + assert.ok(/^[1-9]\d*$/.test(deepRuns), `${deepRunsVariable} is a run count in digits, such as ${deepRunsVariable}=10000: found ${JSON.stringify(deepRuns)}`) + return { numRuns: Number(deepRuns) } +} + +test('a generated document refuses to emit, or its markdown reads back to it', { timeout: propertyTimeout }, () => { + fc.assert( + fc.property(adfDocument, (document) => { + const emitted = adfToMarkdown(document) + if (!emitted.ok) return + const read = markdownToAdf(emitted.value) + assert.ok(read.ok, read.ok ? '' : `${read.error.code}: ${read.error.message} — reading ${JSON.stringify(emitted.value)}`) + assert.deepEqual(toEditorNormal(read.value), document, `reading ${JSON.stringify(emitted.value)}`) + }), + runParameters(), + ) +}) diff --git a/src/corpus.test.ts b/src/corpus.test.ts index 31a7f1d..33d752d 100644 --- a/src/corpus.test.ts +++ b/src/corpus.test.ts @@ -107,18 +107,6 @@ function roundTripFixtures(): { name: string; path: string }[] { ) } -test('no two round-trip documents share one markdown spelling', () => { - const spellings = new Map() - for (const fixture of roundTripFixtures()) { - const parsed: unknown = JSON.parse(readFileSync(fixture.path, 'utf8')) - assert.ok(isAdfDocument(parsed), `${fixture.name} is not an ADF document`) - const result = adfToMarkdown(parsed) - assert.ok(result.ok, result.ok ? '' : `${result.error.code}: ${result.error.message}`) - assert.equal(spellings.get(result.value), undefined, `${fixture.name} and ${spellings.get(result.value)} share one markdown spelling`) - spellings.set(result.value, fixture.name) - } -}) - test('no round-trip fixture repeats the document another holds', () => { const documents = new Map() for (const fixture of roundTripFixtures()) { diff --git a/src/markdown/commonmark-grammar.ts b/src/markdown/commonmark-grammar.ts index 0fbefe1..c36fbb0 100644 --- a/src/markdown/commonmark-grammar.ts +++ b/src/markdown/commonmark-grammar.ts @@ -51,6 +51,7 @@ const htmlBlockConditions: HtmlBlockCondition[] = [ ] const asciiPunctuation = /[!"#$%&'()*+,\-./:;<=>?@[\\\]^_`{|}~]/ const atxHeadingOpener = /^(#{1,6})(?:[ \t]|$)/ +const blankLine = /^[ \t]*$/ const codeFenceOpener = /^(`{3,}|~{3,})/ const directiveClaim = /^:{2,}(?:[A-Za-z0-9]|[ \t]*$)/ const pipeClaim = /^\|/ @@ -167,6 +168,10 @@ export function isAutolink(text: string): boolean { return autolink.test(text) } +export function isBlankLine(line: string): boolean { + return blankLine.test(line) +} + export function isThematicBreak(line: string): boolean { return thematicBreak.test(line) } diff --git a/src/markdown/directive-syntax.ts b/src/markdown/directive-syntax.ts index 87c4976..3c0ec9e 100644 --- a/src/markdown/directive-syntax.ts +++ b/src/markdown/directive-syntax.ts @@ -56,6 +56,11 @@ export function attributeValue(text: string, kind: AttributeKind): AttributeRead return overNested(parsed) ? { refusal: 'nesting' } : { value: { kind, value: parsed } } } +export function holdsInlineDirectiveOpener(text: string): boolean { + for (let index = text.indexOf(':'); index !== -1; index = text.indexOf(':', index + 1)) if (opensInlineDirective(text, index)) return true + return false +} + export function isBareToken(text: string): boolean { return bareToken.test(text) } diff --git a/src/markdown/emit/adf-to-markdown.ts b/src/markdown/emit/adf-to-markdown.ts index 7f861b8..dfd8827 100644 --- a/src/markdown/emit/adf-to-markdown.ts +++ b/src/markdown/emit/adf-to-markdown.ts @@ -6,7 +6,7 @@ import { carriedBlock } from '../opaque-carry.ts' import { emitInlineLine } from './inline-line.ts' import { failure, faulted, success, type ConvertErrorPath, type Result } from '../../result.ts' import { fencedCodeBlock } from '../backtick-runs.ts' -import { holdsNullCharacter, isThematicBreak, markerInterruptsParagraph } from '../commonmark-grammar.ts' +import { holdsNullCharacter, isBlankLine, isThematicBreak, markerInterruptsParagraph } from '../commonmark-grammar.ts' import { languageSlot } from '../code-language.ts' import { largestNesting } from '../../nesting.ts' import { listBreakSpelling } from '../list-break.ts' @@ -225,8 +225,10 @@ function emitListItem(item: AdfNode, marker: string, path: ConvertErrorPath, dep const inner = emitBlocks(nodeContent(item), 'list-item', path, depth + 1) if (!inner.ok) return inner if (inner.value.text === '') return success({ fenceColons: 0, text: marker.trimEnd() }) + const body = inner.value.text.split('\n') + if (body.some((line) => line !== '' && isBlankLine(line))) return undefined const indent = ' '.repeat(marker.length) - const lines = inner.value.text.split('\n').map((line, index) => (index === 0 ? `${marker}${line}` : line === '' ? '' : `${indent}${line}`)) + const lines = body.map((line, index) => (index === 0 ? `${marker}${line}` : line === '' ? '' : `${indent}${line}`)) if (isThematicBreak(lines[0] ?? '')) return undefined return success({ fenceColons: inner.value.fenceColons, text: lines.join('\n') }) } diff --git a/src/markdown/emit/inline-line.ts b/src/markdown/emit/inline-line.ts index 71aaaa1..6c0105b 100644 --- a/src/markdown/emit/inline-line.ts +++ b/src/markdown/emit/inline-line.ts @@ -1,18 +1,18 @@ import type { AdfMark, AdfNode } from '../../adf/document.ts' import type { InlineDirective } from '../../adf/inline-directives.ts' -import { assembleInlineLine, type InlineEscaping, type InlineSegment, type LineContainer, type NodeRange } from './line-escaping.ts' +import { assembleInlineLine, isSyntax, type InlineEscaping, type InlineSegment, type LineContainer, type NodeRange } from './line-escaping.ts' import { carriedInline } from '../opaque-carry.ts' import { claimsLine, holdsNullCharacter, isAutolink } from '../commonmark-grammar.ts' +import { escapeUnbalanced, spellDestination, spellLinkTarget } from '../link-syntax.ts' import { failure, faulted, success, type ConvertErrorPath, type Result } from '../../result.ts' import { holdsEntityReference } from '../entity-references.ts' +import { holdsInlineDirectiveOpener, slotLineEndingFault, spellLeafDirective } from '../directive-syntax.ts' import { inlineDirective } from '../../adf/inline-directives.ts' import { largestNesting } from '../../nesting.ts' import { longestBacktickRun } from '../backtick-runs.ts' import { markSpelling, spellMarkAttributes } from '../mark-spellings.ts' import { nodeAttrs, nodeContent, nodeMarks } from '../../adf/document.ts' import { sameMark } from '../../adf/editor-normal.ts' -import { slotLineEndingFault, spellLeafDirective } from '../directive-syntax.ts' -import { spellDestination, spellTitle } from '../link-syntax.ts' import { spellInlineNodeAttributes } from './inline-directive-spelling.ts' import { spellTextDirective } from '../text-directive.ts' @@ -41,7 +41,7 @@ export function emitInlineLine(nodes: readonly AdfNode[], container: LineContain export function tryPipeCell(nodes: readonly AdfNode[], path: ConvertErrorPath): string | undefined { const emitted = emitLine(nodes, 'table-cell', path) if (!emitted.ok) return undefined - if (emitted.value.segments.some((segment) => segment.escaping === 'none' && segment.text.includes('|'))) return undefined + if (emitted.value.segments.some((segment) => isSyntax(segment.escaping) && segment.text.includes('|'))) return undefined return emitted.value.line } @@ -284,14 +284,14 @@ function emitLink(nodes: readonly AdfNode[], mark: AdfMark, depth: number, range if (typeof href !== 'string') return success({ carry: range }) const node = nodes[0] const bare = nodes.length === 1 && node !== undefined && node.type === 'text' && node.text === href && nodeMarks(node).length === depth + 1 - if (bare && title === undefined && isAutolink(href) && !holdsEntityReference(href)) return success({ segments: [syntax(`<${href}>`)] }) - const destination = spellDestination(href, path) - if (!destination.ok) return destination - const spelledTitle = typeof title === 'string' ? spellTitle(title, path) : success('') - if (!spelledTitle.ok) return spelledTitle + const autolinkHolds = !context.bracketed || (!href.includes('`') && !holdsInlineDirectiveOpener(href) && escapeUnbalanced(href, '[', ']') === href) + if (bare && autolinkHolds && title === undefined && isAutolink(href) && !holdsEntityReference(href)) return success({ segments: [syntax(`<${href}>`)] }) + const target = spellLinkTarget(href, typeof title === 'string' ? title : undefined, path) + if (!target.ok) return target const inner = emitRun(nodes, depth + 1, range.first, { ...context, bracketed: true }) if (!inner.ok) return inner if (inner.value.carry !== undefined) return inner - return success({ segments: [syntax('['), ...inner.value.segments, syntax(`](${destination.value}${spelledTitle.value})`)] }) + const spelledTarget: InlineSegment = context.bracketed ? { escaping: 'bracketed-link-target', text: escapeUnbalanced(target.value, '[', ']') } : syntax(target.value) + return success({ segments: [syntax('['), ...inner.value.segments, syntax(']('), spelledTarget, syntax(')')] }) } diff --git a/src/markdown/emit/line-escaping.ts b/src/markdown/emit/line-escaping.ts index 4861771..eeaa7ec 100644 --- a/src/markdown/emit/line-escaping.ts +++ b/src/markdown/emit/line-escaping.ts @@ -7,7 +7,7 @@ import { readEntityReference } from '../entity-references.ts' export type EmphasisRole = 'close' | 'open' -export type InlineEscaping = 'backslash' | 'bracketed' | 'none' +export type InlineEscaping = 'backslash' | 'bracketed' | 'bracketed-link-target' | 'none' export type NodeRange = { first: number; last: number } @@ -77,10 +77,12 @@ function escape(segments: readonly InlineSegment[], container: LineContainer): A const escaping = escapings[index] const escapable = escaping === 'backslash' || escaping === 'bracketed' if ( - escapable && - (claimsLineStart(line, index, container) || - mergesWithSyntax(scan, escapings, index) || - opensConstruct(scan, linkClose, index, escaping === 'bracketed', container, escaped)) + (escapable && + (claimsLineStart(line, index, container) || + mergesWithSyntax(scan, escapings, index) || + opensConstruct(scan, linkClose, index, escaping === 'bracketed', container, escaped))) || + (escaping === 'bracketed-link-target' && + ((scan.charAt(index) === '`' && opensCodeSpan(scan, index, escaped)) || (scan.charAt(index) === ':' && opensInlineDirective(scan, index)))) ) { output += '\\' escaped.add(index) @@ -181,8 +183,8 @@ function touchesSyntax(scan: string, escapings: readonly (InlineEscaping | undef return scan.charAt(cursor) === character && isSyntax(escapings[cursor]) } -function isSyntax(escaping: InlineEscaping | undefined): boolean { - return escaping === 'none' +export function isSyntax(escaping: InlineEscaping | undefined): boolean { + return escaping === 'none' || escaping === 'bracketed-link-target' } function opensConstruct( diff --git a/src/markdown/link-syntax.ts b/src/markdown/link-syntax.ts index 488011e..38aff4d 100644 --- a/src/markdown/link-syntax.ts +++ b/src/markdown/link-syntax.ts @@ -103,27 +103,35 @@ export function spellDestination(href: string, path: ConvertErrorPath): Result`) } if (href.startsWith('<')) return failure('unspellable-link', 'a bare link destination cannot begin with an angle bracket', path) - return success(escapeUnbalanced(href)) + return success(escapeUnbalanced(href, '(', ')')) } -export function spellTitle(title: string, path: ConvertErrorPath): Result { +export function spellLinkTarget(href: string, title: string | undefined, path: ConvertErrorPath): Result { + const destination = spellDestination(href, path) + if (!destination.ok || title === undefined) return destination + const spelledTitle = spellTitle(title, path) + if (!spelledTitle.ok) return spelledTitle + return success(`${destination.value === '' ? '<>' : destination.value}${spelledTitle.value}`) +} + +export function escapeUnbalanced(spelling: string, opener: string, closer: string): string { + const open: number[] = [] + const unbalanced = new Set() + for (let index = 0; index < spelling.length; index += backslashEscape(spelling, index) === undefined ? 1 : 2) { + const character = spelling.charAt(index) + if (character === opener) open.push(index) + if (character === closer && open.pop() === undefined) unbalanced.add(index) + } + for (const index of open) unbalanced.add(index) + let spelled = '' + for (let index = 0; index < spelling.length; index += 1) spelled += (unbalanced.has(index) ? '\\' : '') + spelling.charAt(index) + return spelled +} + +function spellTitle(title: string, path: ConvertErrorPath): Result { if (/[\n\r\\]/.test(title)) { return failure('unspellable-link', 'no canonical escape spells a backslash or newline in a link title', path) } if (holdsEntityReference(title)) return failure('unspellable-link', 'a link title holds an entity reference that decodes on the way back', path) return success(` "${title.replaceAll('"', '\\"')}"`) } - -function escapeUnbalanced(href: string): string { - const open: number[] = [] - const unbalanced = new Set() - for (let index = 0; index < href.length; index += 1) { - const character = href.charAt(index) - if (character === '(') open.push(index) - if (character === ')' && open.pop() === undefined) unbalanced.add(index) - } - for (const index of open) unbalanced.add(index) - let spelled = '' - for (let index = 0; index < href.length; index += 1) spelled += (unbalanced.has(index) ? '\\' : '') + href.charAt(index) - return spelled -} diff --git a/src/markdown/parse/blocks.ts b/src/markdown/parse/blocks.ts index 866ff31..7527d38 100644 --- a/src/markdown/parse/blocks.ts +++ b/src/markdown/parse/blocks.ts @@ -7,6 +7,7 @@ import { claimsPipeLine, closingCodeFence, decodeTextEscapes, + isBlankLine, isThematicBreak, listMarker, markerInterruptsParagraph, @@ -61,7 +62,6 @@ type Line = { column: number; text: string } type Walk = ParsedBlocks & { leaf: OpenLeaf | undefined; position: SourcePosition; stack: OpenContainer[] } -const blankLine = /^[ \t]*$/ const indentedCodeColumns = 4 const largestOpenerIndentation = 3 const leafColons = 2 @@ -117,7 +117,7 @@ function continuesContainer(walk: Walk, container: OpenContainer, line: Line): L // A directive container has no continuation marker: only its own fence closes it. if (container.kind === 'directive') return line // A list item begins with at most one blank line: an empty one gives the second up. - if (blankLine.test(line.text)) { + if (isBlankLine(line.text)) { return container.blocks.length === 0 && walk.leaf === undefined ? undefined : { column: line.column, text: '' } } return leadingColumns(line) < container.indentation ? undefined : removeColumns(line, container.indentation) @@ -155,7 +155,7 @@ function itemStart(line: Line, opener: Line, paragraphOpen: boolean, enclosing: const marker = listMarker(opener.text) if (marker === undefined) return undefined const after: Line = { column: opener.column + marker.width, text: opener.text.slice(marker.width) } - const blank = blankLine.test(after.text) + const blank = isBlankLine(after.text) if (paragraphOpen && !markerInterruptsParagraph(marker.start, blank)) return undefined const spaces = leadingColumns(after) const padding = blank || spaces > indentedCodeColumns ? 1 : spaces @@ -268,7 +268,7 @@ function pushFault(walk: Walk, fault: ConvertFault): void { // A claimed line ends the lazy continuation CommonMark would fold it into (spec/flavour.md). function continuesLazily(walk: Walk, line: Line): boolean { - if (walk.leaf?.kind !== 'paragraph' || blankLine.test(line.text)) return false + if (walk.leaf?.kind !== 'paragraph' || isBlankLine(line.text)) return false if (leadingColumns(line) >= indentedCodeColumns) return true const opener = removeColumns(line, largestOpenerIndentation).text if (claimsDirectiveLine(opener) || claimsPipeLine(opener) || isThematicBreak(opener)) return false @@ -283,7 +283,7 @@ function readBlockLine(walk: Walk, line: Line): void { return } if (leaf?.kind === 'html') { - if (leaf.closer === undefined ? blankLine.test(line.text) : leaf.closer.test(line.text)) closeLeaf(walk) + if (leaf.closer === undefined ? isBlankLine(line.text) : leaf.closer.test(line.text)) closeLeaf(walk) return } if (leaf?.kind === 'pipe-table') { @@ -298,7 +298,7 @@ function readBlockLine(walk: Walk, line: Line): void { if (readIndentedCodeLine(leaf, line)) return closeLeaf(walk) } - if (blankLine.test(line.text)) { + if (isBlankLine(line.text)) { closeLeaf(walk) return } @@ -310,7 +310,7 @@ function readBlockLine(walk: Walk, line: Line): void { } function readIndentedCodeLine(leaf: Extract, line: Line): boolean { - if (blankLine.test(line.text)) { + if (isBlankLine(line.text)) { leaf.held.push(removeColumns(line, indentedCodeColumns).text) return true }