diff --git a/README.md b/README.md index a70ce34..ba7e164 100644 --- a/README.md +++ b/README.md @@ -3,8 +3,9 @@ Lossless conversion between **Atlassian Document Format** (ADF), an extended markdown flavour, and an HTML dialect. -**Status: scaffold only, no conversion code yet.** Plan: `todo.md`. Decisions: `AGENTS.md`. The -flavour's grammar: [`spec/flavour.md`](spec/flavour.md). +**Status: pre-release — `adfToMarkdown` emits the CommonMark subset, nothing else is built.** +Plan: `todo.md`. Decisions: `AGENTS.md`. The flavour's grammar: +[`spec/flavour.md`](spec/flavour.md). ## What it is for diff --git a/package.json b/package.json index b0e722a..29f0456 100644 --- a/package.json +++ b/package.json @@ -13,7 +13,7 @@ "node": ">=24" }, "scripts": { - "test": "node --test \"src/**/*.test.ts\"", + "test": "node --test --experimental-test-coverage --test-coverage-exclude=\"src/**/*.test.ts\" --test-coverage-branches=89 --test-coverage-functions=100 --test-coverage-lines=100 \"src/**/*.test.ts\"", "typecheck": "tsc --noEmit" }, "devDependencies": { diff --git a/src/adf-document.test.ts b/src/adf-document.test.ts new file mode 100644 index 0000000..1c5eaba --- /dev/null +++ b/src/adf-document.test.ts @@ -0,0 +1,34 @@ +import assert from 'node:assert/strict' +import test from 'node:test' + +import { isAdfDocument } from './adf-document.ts' + +test('accepts an editor-normal document', () => { + assert.equal(isAdfDocument({ content: [{ content: [{ text: 'x', type: 'text' }], type: 'paragraph' }], type: 'doc', version: 1 }), true) + assert.equal(isAdfDocument({ type: 'doc', version: 1 }), true) +}) + +test('rejects anything that is not a doc node', () => { + assert.equal(isAdfDocument(null), false) + assert.equal(isAdfDocument([]), false) + assert.equal(isAdfDocument('doc'), false) + assert.equal(isAdfDocument({ type: 'paragraph', version: 1 }), false) + assert.equal(isAdfDocument({ type: 'doc' }), false) + assert.equal(isAdfDocument({ type: 'doc', version: Number.NaN }), false) + assert.equal(isAdfDocument({ extra: 1, type: 'doc', version: 1 }), false) +}) + +test('rejects a node whose shape ProseMirror JSON cannot hold', () => { + assert.equal(isAdfDocument({ content: [{ type: 1 }], type: 'doc', version: 1 }), false) + assert.equal(isAdfDocument({ content: [{ text: 1, type: 'text' }], type: 'doc', version: 1 }), false) + assert.equal(isAdfDocument({ content: [{ node: 'x', type: 'paragraph' }], type: 'doc', version: 1 }), false) + assert.equal(isAdfDocument({ content: [{ content: {}, type: 'paragraph' }], type: 'doc', version: 1 }), false) + assert.equal(isAdfDocument({ content: [{ marks: [{ type: 1 }], text: 'x', type: 'text' }], type: 'doc', version: 1 }), false) + assert.equal(isAdfDocument({ content: [{ attrs: { a: Number.POSITIVE_INFINITY }, type: 'paragraph' }], type: 'doc', version: 1 }), false) + assert.equal(isAdfDocument({ content: [{ attrs: [], type: 'paragraph' }], type: 'doc', version: 1 }), false) +}) + +test('accepts the JSON values an attribute may hold', () => { + assert.equal(isAdfDocument({ content: [{ attrs: { a: [1, 'x', null, true, { b: 2 }] }, type: 'paragraph' }], type: 'doc', version: 1 }), true) + assert.equal(isAdfDocument({ content: [{ attrs: { a: [() => 1] }, type: 'paragraph' }], type: 'doc', version: 1 }), false) +}) diff --git a/src/adf-document.ts b/src/adf-document.ts new file mode 100644 index 0000000..bbe333f --- /dev/null +++ b/src/adf-document.ts @@ -0,0 +1,64 @@ +import { isJsonValue, type JsonValue } from './canonical-json.ts' + +export type AdfAttributes = { [key: string]: JsonValue } + +export type AdfMark = { + attrs?: AdfAttributes + type: string +} + +export type AdfNode = { + attrs?: AdfAttributes + content?: AdfNode[] + marks?: AdfMark[] + text?: string + type: string +} + +export type AdfDocument = { + content?: AdfNode[] + type: 'doc' + version: number +} + +const documentKeys = ['content', 'type', 'version'] +const markKeys = ['attrs', 'type'] +const nodeKeys = ['attrs', 'content', 'marks', 'text', 'type'] + +export function isAdfDocument(value: unknown): value is AdfDocument { + if (!isRecord(value) || !holdsOnly(value, documentKeys)) return false + if (value['type'] !== 'doc') return false + if (typeof value['version'] !== 'number' || !Number.isFinite(value['version'])) return false + return !('content' in value) || isArrayOf(value['content'], isAdfNode) +} + +function isAdfMark(value: unknown): value is AdfMark { + if (!isRecord(value) || !holdsOnly(value, markKeys)) return false + if (typeof value['type'] !== 'string') return false + return !('attrs' in value) || isAttributes(value['attrs']) +} + +function isAdfNode(value: unknown): value is AdfNode { + if (!isRecord(value) || !holdsOnly(value, nodeKeys)) return false + if (typeof value['type'] !== 'string') return false + if ('attrs' in value && !isAttributes(value['attrs'])) return false + if ('content' in value && !isArrayOf(value['content'], isAdfNode)) return false + if ('marks' in value && !isArrayOf(value['marks'], isAdfMark)) return false + return !('text' in value) || typeof value['text'] === 'string' +} + +function isArrayOf(value: unknown, guard: (item: unknown) => item is T): value is T[] { + return Array.isArray(value) && value.every(guard) +} + +function isAttributes(value: unknown): value is AdfAttributes { + return isRecord(value) && isJsonValue(value) +} + +function isRecord(value: unknown): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value) +} + +function holdsOnly(value: Record, keys: readonly string[]): boolean { + return Object.keys(value).every((key) => keys.includes(key)) +} diff --git a/src/adf-to-markdown.test.ts b/src/adf-to-markdown.test.ts new file mode 100644 index 0000000..ec6ad2b --- /dev/null +++ b/src/adf-to-markdown.test.ts @@ -0,0 +1,155 @@ +import assert from 'node:assert/strict' +import test from 'node:test' + +import type { AdfDocument, AdfNode } from './adf-document.ts' +import type { Result } from './result.ts' +import { adfToMarkdown } from './index.ts' + +function document(...content: AdfNode[]): AdfDocument { + return { content, type: 'doc', version: 1 } +} + +function paragraph(...content: AdfNode[]): AdfNode { + return { content, type: 'paragraph' } +} + +function code(result: Result): string { + return result.ok ? `emitted ${JSON.stringify(result.value)}` : result.error.code +} + +function markdown(result: Result): string { + return result.ok ? result.value : `${result.error.code}: ${result.error.message}` +} + +test('refuses a value that is not an ADF document', () => { + assert.equal(code(adfToMarkdown({ type: 'doc', version: Number.NaN })), 'not-an-adf-document') +}) + +test('refuses a document version the markdown cannot carry', () => { + assert.equal(code(adfToMarkdown({ type: 'doc', version: 2 })), 'unsupported-document-version') +}) + +test('refuses an attribute the canonical form does not spell', () => { + assert.equal(code(adfToMarkdown(document({ attrs: { localId: 'a' }, type: 'paragraph' }))), 'unspelled-node-attribute') + assert.equal(code(adfToMarkdown(document({ attrs: { wrap: true }, type: 'codeBlock' }))), 'unspelled-node-attribute') + assert.equal(code(adfToMarkdown(document(paragraph({ attrs: { localId: 'a' }, type: 'hardBreak' }, { text: 'x', type: 'text' })))), 'unspelled-node-attribute') +}) + +test('refuses marks on a block node', () => { + assert.equal(code(adfToMarkdown(document({ marks: [{ type: 'border' }], type: 'blockquote' }))), 'unspelled-block-marks') +}) + +test('refuses an ordered list whose markdown start is ambiguous', () => { + const items: AdfNode[] = [{ content: [paragraph({ text: 'x', type: 'text' })], type: 'listItem' }] + assert.equal(code(adfToMarkdown(document({ content: items, type: 'orderedList' }))), 'ambiguous-ordered-list-start') + assert.equal(code(adfToMarkdown(document({ attrs: { order: 1 }, content: items, type: 'orderedList' }))), 'ambiguous-ordered-list-start') + assert.equal(markdown(adfToMarkdown(document({ attrs: { order: 2 }, content: items, type: 'orderedList' }))), '2. x\n') +}) + +test('refuses the code block info strings the fence cannot hold', () => { + assert.equal(code(adfToMarkdown(document({ attrs: { language: '' }, type: 'codeBlock' }))), 'ambiguous-empty-code-block-language') + assert.equal(code(adfToMarkdown(document({ attrs: { language: 'adf' }, type: 'codeBlock' }))), 'reserved-adf-language') + assert.equal(code(adfToMarkdown(document({ attrs: { language: 'a`b' }, type: 'codeBlock' }))), 'unspellable-code-block-language') + assert.equal(code(adfToMarkdown(document({ attrs: { language: ' sql' }, type: 'codeBlock' }))), 'unspellable-code-block-language') +}) + +test('refuses a link destination CommonMark cannot spell', () => { + const link = (href: string): AdfDocument => document(paragraph({ marks: [{ attrs: { href }, type: 'link' }], text: 't', type: 'text' })) + assert.equal(code(adfToMarkdown(link('https://example.com/a)b'))), 'unspellable-link-destination') + assert.equal(code(adfToMarkdown(link('https://example.com/a b>c'))), 'unspellable-link-destination') + assert.equal(code(adfToMarkdown(link(' { + const titled = (title: string): AdfDocument => + document(paragraph({ marks: [{ attrs: { href: 'https://example.com/', title }, type: 'link' }], text: 't', type: 'text' })) + assert.equal(code(adfToMarkdown(titled('He said "hi"'))), 'unspellable-link-title') + assert.equal(code(adfToMarkdown(titled('a\nb'))), 'unspellable-link-title') +}) + +test('refuses a link attribute no markdown spelling holds', () => { + assert.equal( + code(adfToMarkdown(document(paragraph({ marks: [{ attrs: { href: 'https://example.com/', id: 'x' }, type: 'link' }], text: 't', type: 'text' })))), + 'unspellable-mark', + ) +}) + +test('refuses a mark the canonical spellings cannot nest', () => { + assert.equal(code(adfToMarkdown(document(paragraph({ marks: [{ type: 'underline' }], text: 'x', type: 'text' })))), 'unspellable-mark') + assert.equal(code(adfToMarkdown(document(paragraph({ marks: [{ type: 'code' }, { type: 'strong' }], text: 'x', type: 'text' })))), 'unspellable-mark') + assert.equal(code(adfToMarkdown(document(paragraph({ marks: [{ attrs: { colour: 'red' }, type: 'em' }], text: 'x', type: 'text' })))), 'unspellable-mark') +}) + +test('refuses whitespace CommonMark cannot hold', () => { + assert.equal(code(adfToMarkdown(document(paragraph({ text: ' lead', type: 'text' })))), 'unspellable-whitespace') + assert.equal(code(adfToMarkdown(document(paragraph({ text: 'trail ', type: 'text' })))), 'unspellable-whitespace') + assert.equal(code(adfToMarkdown(document(paragraph({ text: 'a\nb', type: 'text' })))), 'unspellable-whitespace') + assert.equal(code(adfToMarkdown(document(paragraph({ marks: [{ type: 'em' }], text: 'x ', type: 'text' }, { text: 'y', type: 'text' })))), 'unspellable-whitespace') +}) + +test('refuses a line whose start block parsing would claim', () => { + assert.equal(code(adfToMarkdown(document(paragraph({ marks: [{ type: 'code' }], text: '```', type: 'text' })))), 'unspellable-line-start') +}) + +test('refuses two adjacent lists of the same kind', () => { + const list: AdfNode = { content: [{ content: [paragraph({ text: 'x', type: 'text' })], type: 'listItem' }], type: 'bulletList' } + assert.equal(code(adfToMarkdown(document(list, list))), 'unspellable-adjacent-lists') +}) + +test('refuses a node type the canonical form does not cover', () => { + assert.equal(code(adfToMarkdown(document({ type: 'panel' }))), 'unsupported-node-type') + assert.equal(code(adfToMarkdown(document(paragraph({ type: 'mention' })))), 'unsupported-node-type') +}) + +test('refuses a node whose content model the canonical form cannot emit', () => { + assert.equal(code(adfToMarkdown(document({ type: 'listItem' }))), 'unsupported-node-shape') + assert.equal(code(adfToMarkdown(document({ content: [paragraph()], type: 'codeBlock' }))), 'unsupported-node-shape') + assert.equal(code(adfToMarkdown(document({ content: [paragraph()], type: 'bulletList' }))), 'unsupported-node-shape') +}) + +test('refuses an ordered list start no marker spells', () => { + const items: AdfNode[] = [{ content: [paragraph({ text: 'x', type: 'text' })], type: 'listItem' }] + assert.equal(code(adfToMarkdown(document({ attrs: { order: 1.5 }, content: items, type: 'orderedList' }))), 'unsupported-node-shape') +}) + +test('refuses a code span over anything but one text node', () => { + assert.equal(code(adfToMarkdown(document(paragraph({ marks: [{ type: 'code' }], type: 'hardBreak' })))), 'unspellable-mark') +}) + +test('refuses a heading level outside the ATX range', () => { + assert.equal(code(adfToMarkdown(document({ attrs: { level: 7 }, content: [{ text: 'x', type: 'text' }], type: 'heading' }))), 'unsupported-heading-level') + assert.equal(code(adfToMarkdown(document({ content: [{ text: 'x', type: 'text' }], type: 'heading' }))), 'unsupported-heading-level') +}) + +test('escapes only text that would otherwise open a construct', () => { + const emitted = (text: string): string => markdown(adfToMarkdown(document(paragraph({ text, type: 'text' })))) + assert.equal(emitted('
'), '\\
\n') + assert.equal(emitted('a < b'), 'a < b\n') + assert.equal(emitted('& & x'), '\\& & x\n') + assert.equal(emitted('| a | b |'), '\\| a | b |\n') + assert.equal(emitted(':mention[@x]{id=1}'), '\\:mention[@x]{id=1}\n') + assert.equal(emitted(':::panel info'), '\\:::panel info\n') + assert.equal(emitted('10:30 tomorrow'), '10:30 tomorrow\n') + assert.equal(emitted('[a](b)'), '\\[a](b)\n') + assert.equal(emitted('**bold**'), '\\*\\*bold**\n') + assert.equal(emitted('a `x` b'), 'a \\`x` b\n') + assert.equal(emitted('~~struck~~'), '\\~~struck~~\n') + assert.equal(emitted('a \\* b'), 'a \\\\* b\n') + assert.equal(emitted('1. not a list'), '1\\. not a list\n') + assert.equal(emitted('*"quoted"*'), '\\*"quoted"*\n') + assert.equal(emitted('x"_y"'), 'x"\\_y"\n') +}) + +test('escapes a heading closing sequence', () => { + const heading = (text: string): string => markdown(adfToMarkdown(document({ attrs: { level: 2 }, content: [{ text, type: 'text' }], type: 'heading' }))) + assert.equal(heading('done #'), '## done \\#\n') + assert.equal(heading('#tag first'), '## #tag first\n') +}) + +test('emits an empty list item without trailing whitespace', () => { + assert.equal(markdown(adfToMarkdown(document({ content: [{ type: 'listItem' }], type: 'bulletList' }))), '-\n') +}) diff --git a/src/adf-to-markdown.ts b/src/adf-to-markdown.ts new file mode 100644 index 0000000..eeaa908 --- /dev/null +++ b/src/adf-to-markdown.ts @@ -0,0 +1,160 @@ +import type { AdfDocument, AdfNode } from './adf-document.ts' +import { emitInlineLine } from './markdown-inline.ts' +import { failure, success, type Result } from './result.ts' +import { isAdfDocument } from './adf-document.ts' +import { longestBacktickRun } from './backtick-runs.ts' + +const listTypes = ['bulletList', 'orderedList'] + +export function adfToMarkdown(document: AdfDocument): Result { + if (!isAdfDocument(document)) return failure('not-an-adf-document', 'the value is not an ADF document') + if (document.version !== 1) return failure('unsupported-document-version', `no markdown spelling carries ADF version ${document.version}`) + const blocks = emitBlocks(document.content ?? [], false) + if (!blocks.ok) return blocks + return success(blocks.value === '' ? '' : `${blocks.value}\n`) +} + +function emitBlocks(nodes: readonly AdfNode[], inListItem: boolean): Result { + let output = '' + for (let index = 0; index < nodes.length; index += 1) { + const node = nodes[index] + if (node === undefined) return failure('unsupported-node-shape', 'the block content holds a hole') + const previous = nodes[index - 1] + if (previous !== undefined) { + if (listTypes.includes(node.type) && previous.type === node.type) { + return failure('unspellable-adjacent-lists', `two adjacent ${node.type} nodes read back as one list`) + } + output += inListItem && listTypes.includes(node.type) ? '\n' : '\n\n' + } + const block = emitBlock(node) + if (!block.ok) return block + output += block.value + } + return success(output) +} + +function emitBlock(node: AdfNode): Result { + if (node.type === 'blockquote') return emitBlockquote(node) + if (node.type === 'bulletList' || node.type === 'orderedList') return emitList(node) + if (node.type === 'codeBlock') return emitCodeBlock(node) + if (node.type === 'heading') return emitHeading(node) + if (node.type === 'paragraph') return emitParagraph(node) + if (node.type === 'rule') return emitRule(node) + if (node.type === 'hardBreak' || node.type === 'listItem' || node.type === 'text') { + return failure('unsupported-node-shape', `a ${node.type} node cannot stand where a block belongs`) + } + return failure('unsupported-node-type', `the canonical form spells no block node of type ${node.type}`) +} + +function emitBlockquote(node: AdfNode): Result { + const invalid = validateBlockNode(node, []) + if (!invalid.ok) return invalid + const inner = emitBlocks(node.content ?? [], false) + if (!inner.ok) return inner + return success( + inner.value + .split('\n') + .map((line) => (line === '' ? '>' : `> ${line}`)) + .join('\n'), + ) +} + +function emitCodeBlock(node: AdfNode): Result { + const invalid = validateBlockNode(node, ['language']) + if (!invalid.ok) return invalid + const language = node.attrs?.['language'] + if (language !== undefined) { + if (typeof language !== 'string') return failure('unsupported-node-shape', 'a codeBlock language is no string') + if (language === '') return failure('ambiguous-empty-code-block-language', 'an empty codeBlock language and an absent one share one markdown spelling') + if (language === 'adf') return failure('reserved-adf-language', 'the adf info string is reserved for the opaque carry') + if (/[`\n\r]/.test(language) || language !== language.trim()) { + return failure('unspellable-code-block-language', 'a fence info string holds no backtick and no edge whitespace') + } + } + let text = '' + for (const child of node.content ?? []) { + if (child.type !== 'text' || typeof child.text !== 'string' || (child.marks ?? []).length > 0 || Object.keys(child.attrs ?? {}).length > 0) { + return failure('unsupported-node-shape', 'a codeBlock holds plain text nodes only') + } + text += child.text + } + const fence = '`'.repeat(Math.max(3, longestBacktickRun(text) + 1)) + const opening = `${fence}${language ?? ''}` + return success(text === '' ? `${opening}\n${fence}` : `${opening}\n${text}\n${fence}`) +} + +function emitHeading(node: AdfNode): Result { + const invalid = validateBlockNode(node, ['level']) + if (!invalid.ok) return invalid + const level = node.attrs?.['level'] + if (typeof level !== 'number' || !Number.isInteger(level) || level < 1 || level > 6) { + return failure('unsupported-heading-level', `no ATX heading spells level ${JSON.stringify(level ?? null)}`) + } + const hashes = '#'.repeat(level) + const content = node.content ?? [] + if (content.length === 0) return success(hashes) + const line = emitInlineLine(content, 'heading') + if (!line.ok) return line + return success(`${hashes} ${line.value}`) +} + +function emitList(node: AdfNode): Result { + const ordered = node.type === 'orderedList' + const invalid = validateBlockNode(node, ordered ? ['order'] : []) + if (!invalid.ok) return invalid + const start = ordered ? node.attrs?.['order'] : 0 + if (ordered && (start === undefined || start === 1)) { + return failure('ambiguous-ordered-list-start', 'an orderedList starting at 1 and one with no order share one markdown spelling') + } + if (typeof start !== 'number' || !Number.isInteger(start) || start < 0 || start > 999999999) { + return failure('unsupported-node-shape', `no list marker spells the order ${JSON.stringify(start ?? null)}`) + } + const items: string[] = [] + for (const [offset, item] of (node.content ?? []).entries()) { + if (item.type !== 'listItem') return failure('unsupported-node-shape', `a ${node.type} holds listItem nodes only`) + const invalidItem = validateBlockNode(item, []) + if (!invalidItem.ok) return invalidItem + const inner = emitBlocks(item.content ?? [], true) + if (!inner.ok) return inner + const marker = ordered ? `${start + offset}. ` : '- ' + if (inner.value === '') { + items.push(marker.trimEnd()) + continue + } + const indent = ' '.repeat(marker.length) + items.push( + inner.value + .split('\n') + .map((line, lineIndex) => (lineIndex === 0 ? `${marker}${line}` : line === '' ? '' : `${indent}${line}`)) + .join('\n'), + ) + } + return success(items.join('\n')) +} + +function emitParagraph(node: AdfNode): Result { + const invalid = validateBlockNode(node, []) + if (!invalid.ok) return invalid + const content = node.content ?? [] + if (content.length === 0) return success('::paragraph') + return emitInlineLine(content, 'paragraph') +} + +function emitRule(node: AdfNode): Result { + const invalid = validateBlockNode(node, []) + if (!invalid.ok) return invalid + if ((node.content ?? []).length > 0) return failure('unsupported-node-shape', 'a rule holds no content') + return success('---') +} + +function validateBlockNode(node: AdfNode, spelled: readonly string[]): Result { + if ((node.marks ?? []).length > 0) { + return failure('unspelled-block-marks', `the canonical form has no place for the marks a ${node.type} carries`) + } + if (node.text !== undefined) return failure('unsupported-node-shape', `a ${node.type} carries no text`) + const unspelled = Object.keys(node.attrs ?? {}).find((key) => !spelled.includes(key)) + if (unspelled !== undefined) { + return failure('unspelled-node-attribute', `the ${node.type} attribute ${unspelled} has no canonical markdown spelling`) + } + return success(null) +} diff --git a/src/backtick-runs.ts b/src/backtick-runs.ts new file mode 100644 index 0000000..e5580db --- /dev/null +++ b/src/backtick-runs.ts @@ -0,0 +1,9 @@ +export function longestBacktickRun(text: string): number { + let longest = 0 + let current = 0 + for (const character of text) { + current = character === '`' ? current + 1 : 0 + longest = Math.max(longest, current) + } + return longest +} diff --git a/src/canonical-json.test.ts b/src/canonical-json.test.ts new file mode 100644 index 0000000..b6fdaf4 --- /dev/null +++ b/src/canonical-json.test.ts @@ -0,0 +1,41 @@ +import assert from 'node:assert/strict' +import test from 'node:test' + +import { serializeCanonicalJson } from './canonical-json.ts' + +test('sorts object keys recursively', () => { + const value = { b: 1, a: { d: 2, c: 3 } } + assert.equal(serializeCanonicalJson(value, 'compact'), '{"a":{"c":3,"d":2},"b":1}') +}) + +test('spells two-space indentation the way the corpus holds it', () => { + const value = { content: [{ text: 'x', type: 'text' }], type: 'doc', version: 1 } + assert.equal( + serializeCanonicalJson(value, 'two-space'), + [ + '{', + ' "content": [', + ' {', + ' "text": "x",', + ' "type": "text"', + ' }', + ' ],', + ' "type": "doc",', + ' "version": 1', + '}', + ].join('\n'), + ) +}) + +test('keeps empty objects and arrays on one line in both spellings', () => { + assert.equal(serializeCanonicalJson({ a: {}, b: [] }, 'two-space'), '{\n "a": {},\n "b": []\n}') + assert.equal(serializeCanonicalJson({ a: {}, b: [] }, 'compact'), '{"a":{},"b":[]}') +}) + +test('leaves non-ASCII raw', () => { + assert.equal(serializeCanonicalJson({ text: '🎉 räksmörgås' }, 'compact'), '{"text":"🎉 räksmörgås"}') +}) + +test('spells scalars in canonical JSON', () => { + assert.equal(serializeCanonicalJson([null, true, false, 0, -1.5, 'a"b'], 'compact'), '[null,true,false,0,-1.5,"a\\"b"]') +}) diff --git a/src/canonical-json.ts b/src/canonical-json.ts new file mode 100644 index 0000000..e19b8c3 --- /dev/null +++ b/src/canonical-json.ts @@ -0,0 +1,38 @@ +export type JsonValue = JsonValue[] | boolean | null | number | string | { [key: string]: JsonValue } + +export type JsonSpelling = 'compact' | 'two-space' + +export function isJsonValue(value: unknown): value is JsonValue { + if (value === null) return true + if (typeof value === 'boolean' || typeof value === 'string') return true + if (typeof value === 'number') return Number.isFinite(value) + if (Array.isArray(value)) return value.every(isJsonValue) + if (typeof value === 'object') return Object.values(value).every(isJsonValue) + return false +} + +export function serializeCanonicalJson(value: JsonValue, spelling: JsonSpelling): string { + return serialize(value, spelling === 'compact' ? '' : ' ', 0) +} + +function serialize(value: JsonValue, indent: string, depth: number): string { + if (Array.isArray(value)) { + if (value.length === 0) return '[]' + const items = value.map((item) => serialize(item, indent, depth + 1)) + return `[${join(items, indent, depth)}]` + } + if (value !== null && typeof value === 'object') { + const keys = Object.keys(value).sort() + if (keys.length === 0) return '{}' + const separator = indent === '' ? ':' : ': ' + const entries = keys.map((key) => `${JSON.stringify(key)}${separator}${serialize(value[key] ?? null, indent, depth + 1)}`) + return `{${join(entries, indent, depth)}}` + } + return JSON.stringify(value) +} + +function join(parts: readonly string[], indent: string, depth: number): string { + if (indent === '') return parts.join(',') + const inner = `\n${indent.repeat(depth + 1)}` + return `${inner}${parts.join(`,${inner}`)}\n${indent.repeat(depth)}` +} diff --git a/src/corpus.test.ts b/src/corpus.test.ts new file mode 100644 index 0000000..d22e715 --- /dev/null +++ b/src/corpus.test.ts @@ -0,0 +1,80 @@ +import assert from 'node:assert/strict' +import { readFileSync, readdirSync } from 'node:fs' +import { dirname, join } from 'node:path' +import test from 'node:test' +import { fileURLToPath } from 'node:url' + +import { adfToMarkdown } from './adf-to-markdown.ts' +import { isAdfDocument } from './adf-document.ts' +import { isJsonValue, serializeCanonicalJson } from './canonical-json.ts' + +const corpusRoot = join(dirname(fileURLToPath(import.meta.url)), '..', 'corpus') +const roundTripRoot = join(corpusRoot, 'round-trip') + +const emittingDirectories = ['commonmark-subset'] +const pendingDirectories = ['block-nodes', 'inline-nodes'] + +function directoryNames(root: string): string[] { + return readdirSync(root, { withFileTypes: true }) + .filter((entry) => entry.isDirectory()) + .map((entry) => entry.name) + .sort() +} + +function fixtureNames(directory: string, extension: string): string[] { + return readdirSync(join(roundTripRoot, directory)) + .filter((name) => name.endsWith(extension)) + .map((name) => name.slice(0, -extension.length)) + .sort() +} + +function corpusJsonPaths(): string[] { + return readdirSync(corpusRoot, { encoding: 'utf8', recursive: true }) + .filter((name) => name.endsWith('.json')) + .map((name) => join(corpusRoot, name)) + .sort() +} + +test('every round-trip directory is either emitting or explicitly pending', () => { + assert.deepEqual(directoryNames(roundTripRoot), [...emittingDirectories, ...pendingDirectories].sort()) +}) + +for (const directory of [...emittingDirectories, ...pendingDirectories].sort()) { + test(`${directory} pairs every .json with a .md`, () => { + assert.deepEqual(fixtureNames(directory, '.json'), fixtureNames(directory, '.md')) + }) +} + +for (const directory of emittingDirectories) { + const names = [...new Set([...fixtureNames(directory, '.json'), ...fixtureNames(directory, '.md')])].sort() + + test(`${directory} holds fixtures`, () => { + assert.ok(names.length > 0, `${directory} is expected to emit but holds no fixture pairs`) + }) + + for (const name of names) { + test(`${directory}/${name} emits its markdown byte for byte`, () => { + const parsed: unknown = JSON.parse(readFileSync(join(roundTripRoot, directory, `${name}.json`), 'utf8')) + assert.ok(isAdfDocument(parsed), `${name}.json is not an ADF document`) + const result = adfToMarkdown(parsed) + assert.ok(result.ok, result.ok ? '' : `${result.error.code}: ${result.error.message}`) + const expected = readFileSync(join(roundTripRoot, directory, `${name}.md`)) + const emitted = Buffer.from(result.value, 'utf8') + if (!emitted.equals(expected)) assert.equal(result.value, expected.toString('utf8')) + assert.ok(emitted.equals(expected)) + }) + } +} + +test('the corpus holds JSON to gate', () => { + assert.ok(corpusJsonPaths().length > 0) +}) + +for (const path of corpusJsonPaths()) { + test(`${path.slice(corpusRoot.length + 1)} re-serializes to itself`, () => { + const raw = readFileSync(path, 'utf8') + const parsed: unknown = JSON.parse(raw) + assert.ok(isJsonValue(parsed), `${path} does not hold a JSON value`) + assert.equal(`${serializeCanonicalJson(parsed, 'two-space')}\n`, raw) + }) +} diff --git a/src/index.test.ts b/src/index.test.ts deleted file mode 100644 index fc6fe0c..0000000 --- a/src/index.test.ts +++ /dev/null @@ -1,7 +0,0 @@ -import assert from 'node:assert/strict' -import test from 'node:test' - -test('the test harness runs TypeScript', () => { - const answer: number = 42 - assert.equal(answer, 42) -}) diff --git a/src/index.ts b/src/index.ts index 336ce12..8384354 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1 +1,4 @@ -export {} +export type { AdfAttributes, AdfDocument, AdfMark, AdfNode } from './adf-document.ts' +export type { ConvertError, ConvertErrorCode, Result } from './result.ts' +export { adfToMarkdown } from './adf-to-markdown.ts' +export { isAdfDocument } from './adf-document.ts' diff --git a/src/markdown-escaping.ts b/src/markdown-escaping.ts new file mode 100644 index 0000000..4ec7805 --- /dev/null +++ b/src/markdown-escaping.ts @@ -0,0 +1,160 @@ +export type InlineSegment = { + kind: 'emphasis-close' | 'emphasis-open' | 'link-text' | 'literal' | 'syntax' + text: string +} + +export type LineContainer = 'heading' | 'paragraph' + +const asciiPunctuation = /[!"#$%&'()*+,\-./:;<=>?@[\\\]^_`{|}~]/ +const entityReference = /^&(?:[A-Za-z][A-Za-z0-9]{1,31}|#\d{1,7}|#[Xx][A-Fa-f0-9]{1,6});/ +const htmlConstructs = [/^<[!?]/, /^<\/?[A-Za-z][A-Za-z0-9-]*(?:[\s/>]|$)/, /^<[A-Za-z][A-Za-z0-9+.-]{1,31}:[^\s<>]*>/, /^<[^\s<>@]+@[^\s<>@]+>/] +const inlineDirective = /^:[a-z][A-Za-z0-9]*[[{]/ +const linkOpener = /\](?=[([:])/ +const orderedListMarker = /^\d{1,9}$/ +const setextUnderline = /^=+$/ +const thematicBreak = /^(?:(?:\*[ \t]*){3,}|(?:-[ \t]*){3,}|(?:_[ \t]*){3,})$/ +const unicodePunctuation = /[\p{P}\p{S}]/u +const unicodeWhitespace = /[\t\n\f\r \p{Zs}]/u + +const escapableOpeners = [/^#{1,6}(?:[ \t]|$)/, /^>/, /^[*+-](?:[ \t]|$)/, /^`{3,}/, /^~{3,}/, /^:{2,}/, /^\|/] +const blockOpeners = [...escapableOpeners, /^\d{1,9}[.)](?:[ \t]|$)/] + +export function assembleInlineLine(segments: readonly InlineSegment[], container: LineContainer): string { + return escape(resolveEmphasis(segments), container) +} + +export function lineOpensBlock(line: string): boolean { + return blockOpeners.some((opener) => opener.test(line)) || thematicBreak.test(line) +} + +export function isWordCharacter(character: string): boolean { + return character !== '' && !unicodeWhitespace.test(character) && !unicodePunctuation.test(character) +} + +function resolveEmphasis(segments: readonly InlineSegment[]): InlineSegment[] { + const resolved = segments.map((segment) => ({ ...segment })) + const scan = resolved.map((segment) => segment.text).join('') + const offsets: number[] = [] + let offset = 0 + for (const segment of resolved) { + offsets.push(offset) + offset += segment.text.length + } + const open: number[] = [] + for (let index = 0; index < resolved.length; index += 1) { + const segment = resolved[index] + if (segment === undefined) continue + if (segment.kind === 'emphasis-open') open.push(index) + if (segment.kind !== 'emphasis-close') continue + const openerIndex = open.pop() + const opener = openerIndex === undefined ? undefined : resolved[openerIndex] + if (openerIndex === undefined || opener === undefined) continue + const openOffset = offsets[openerIndex] ?? 0 + const closeOffset = offsets[index] ?? 0 + if (!isWordCharacter(charAt(scan, openOffset - 1)) && !isWordCharacter(charAt(scan, closeOffset + 1))) continue + opener.text = '*' + segment.text = '*' + } + return resolved +} + +function escape(segments: readonly InlineSegment[], container: LineContainer): string { + const scan = segments.map((segment) => segment.text).join('') + const kinds: InlineSegment['kind'][] = [] + for (const segment of segments) for (let index = 0; index < segment.text.length; index += 1) kinds.push(segment.kind) + const escaped = new Set() + let output = '' + for (let index = 0; index < scan.length; index += 1) { + const kind = kinds[index] + const escapable = kind === 'literal' || kind === 'link-text' + if (escapable && opensConstruct(scan, index, kind === 'link-text', container, escaped)) { + output += '\\' + escaped.add(index) + } + output += scan.charAt(index) + } + return output +} + +function opensConstruct(scan: string, index: number, inLinkText: boolean, container: LineContainer, escaped: ReadonlySet): boolean { + const character = scan.charAt(index) + const rest = scan.slice(index) + const line = rest.split('\n')[0] ?? '' + if (container === 'paragraph' && (index === 0 || scan.charAt(index - 1) === '\n')) { + if (escapableOpeners.some((opener) => opener.test(rest))) return true + if (thematicBreak.test(line)) return true + if (index > 0 && setextUnderline.test(line)) return true + } + if (container === 'paragraph' && (character === '.' || character === ')') && closesOrderedListMarker(scan, index)) return true + if (container === 'heading' && character === '#' && /^#+$/.test(rest) && (index === 0 || /[ \t]/.test(scan.charAt(index - 1)))) return true + if (inLinkText && (character === '[' || character === ']')) return true + if (character === '\\') return asciiPunctuation.test(scan.charAt(index + 1)) + if (character === '&') return entityReference.test(rest) + if (character === '<') return htmlConstructs.some((construct) => construct.test(rest)) + if (character === ':') return inlineDirective.test(rest) + if (character === '[') return linkOpener.test(rest) + if (character === '`') return opensCodeSpan(scan, index, escaped) + if (character === '*' || character === '_' || character === '~') return opensEmphasis(scan, index, escaped) + return false +} + +function closesOrderedListMarker(scan: string, index: number): boolean { + const lineStart = scan.lastIndexOf('\n', index - 1) + 1 + if (!orderedListMarker.test(scan.slice(lineStart, index))) return false + const following = scan.charAt(index + 1) + return following === '' || following === ' ' || following === '\t' || following === '\n' +} + +function opensCodeSpan(scan: string, index: number, escaped: ReadonlySet): boolean { + if (!startsRun(scan, index, escaped)) return false + const length = runLength(scan, index) + return new RegExp('(?): boolean { + if (!startsRun(scan, index, escaped)) return false + const character = scan.charAt(index) + const length = runLength(scan, index) + const before = index === 0 ? '' : scan.charAt(index - 1) + const after = scan.charAt(index + length) + if (character === '~') return length === 2 && isLeftFlanking(before, after) + if (!isLeftFlanking(before, after)) return false + if (character === '*') return true + return !isRightFlanking(before, after) || isPunctuation(before) +} + +function startsRun(scan: string, index: number, escaped: ReadonlySet): boolean { + if (index === 0 || escaped.has(index - 1)) return true + return scan.charAt(index - 1) !== scan.charAt(index) +} + +function runLength(scan: string, index: number): number { + const character = scan.charAt(index) + let length = 0 + while (scan.charAt(index + length) === character) length += 1 + return length +} + +function isLeftFlanking(before: string, after: string): boolean { + if (isWhitespace(after)) return false + if (!isPunctuation(after)) return true + return isWhitespace(before) || isPunctuation(before) +} + +function isRightFlanking(before: string, after: string): boolean { + if (isWhitespace(before)) return false + if (!isPunctuation(before)) return true + return isWhitespace(after) || isPunctuation(after) +} + +function isPunctuation(character: string): boolean { + return character !== '' && unicodePunctuation.test(character) +} + +function isWhitespace(character: string): boolean { + return character === '' || unicodeWhitespace.test(character) +} + +function charAt(text: string, index: number): string { + return index < 0 ? '' : text.charAt(index) +} diff --git a/src/markdown-inline.ts b/src/markdown-inline.ts new file mode 100644 index 0000000..739682f --- /dev/null +++ b/src/markdown-inline.ts @@ -0,0 +1,163 @@ +import type { AdfMark, AdfNode } from './adf-document.ts' +import { assembleInlineLine, lineOpensBlock, type InlineSegment, type LineContainer } from './markdown-escaping.ts' +import { failure, success, type Result } from './result.ts' +import { longestBacktickRun } from './backtick-runs.ts' +import { serializeCanonicalJson } from './canonical-json.ts' + +type InlineContext = { + atBlockEnd: boolean + container: LineContainer + inLinkText: boolean +} + +const autolink = /^[A-Za-z][A-Za-z0-9+.-]{1,31}:[^\s<>\u0000-\u001f\u007f]*$/ +const controlCharacter = /[\u0000-\u001f\u007f]/ +const entityReference = /&(?:[A-Za-z][A-Za-z0-9]{1,31}|#\d{1,7}|#[Xx][A-Fa-f0-9]{1,6});/ +const linkAttributes = ['href', 'title'] + +export function emitInlineLine(nodes: readonly AdfNode[], container: LineContainer): Result { + const segments = emitRun(nodes, 0, { atBlockEnd: true, container, inLinkText: false }) + if (!segments.ok) return segments + const line = assembleInlineLine(segments.value, container) + for (const single of line.split('\n')) { + if (/^[ \t]|[ \t]$/.test(single)) { + return failure('unspellable-whitespace', 'a line begins or ends with whitespace CommonMark strips') + } + if (container === 'paragraph' && lineOpensBlock(single)) { + return failure('unspellable-line-start', `block parsing would claim the emitted line ${JSON.stringify(single)}`) + } + } + return success(line) +} + +function emitRun(nodes: readonly AdfNode[], depth: number, context: InlineContext): Result { + const segments: InlineSegment[] = [] + let index = 0 + while (index < nodes.length) { + const node = nodes[index] + if (node === undefined) return failure('unsupported-node-shape', 'the inline content holds a hole') + const mark = (node.marks ?? [])[depth] + if (mark === undefined) { + const leaf = emitLeaf(node, { ...context, atBlockEnd: context.atBlockEnd && index === nodes.length - 1 }) + if (!leaf.ok) return leaf + segments.push(...leaf.value) + index += 1 + continue + } + let end = index + 1 + while (end < nodes.length && sameMark((nodes[end]?.marks ?? [])[depth], mark)) end += 1 + const wrapped = emitMarkedRun(nodes.slice(index, end), mark, depth, { ...context, atBlockEnd: context.atBlockEnd && end === nodes.length }) + if (!wrapped.ok) return wrapped + segments.push(...wrapped.value) + index = end + } + return success(segments) +} + +function emitLeaf(node: AdfNode, context: InlineContext): Result { + if (node.type !== 'hardBreak' && node.type !== 'text') { + return failure('unsupported-node-type', `the canonical form spells no inline node of type ${node.type}`) + } + const unspelled = Object.keys(node.attrs ?? {})[0] + if (unspelled !== undefined) { + return failure('unspelled-node-attribute', `the ${node.type} attribute ${unspelled} has no canonical markdown spelling`) + } + if (node.type === 'hardBreak') { + if (context.container === 'heading' || context.atBlockEnd) return success([{ kind: 'syntax', text: ':hardBreak{}' }]) + return success([{ kind: 'syntax', text: '\\\n' }]) + } + if (typeof node.text !== 'string') return failure('unsupported-node-shape', 'a text node carries no text') + if ((node.content ?? []).length > 0) return failure('unsupported-node-shape', 'a text node carries content') + if (/[\n\r]/.test(node.text)) return failure('unspellable-whitespace', 'a text node holds a newline CommonMark cannot spell') + return success([{ kind: context.inLinkText ? 'link-text' : 'literal', text: node.text }]) +} + +function emitMarkedRun(nodes: readonly AdfNode[], mark: AdfMark, depth: number, context: InlineContext): Result { + if (mark.type === 'code') return emitCodeSpan(nodes, depth) + if (mark.type === 'link') return emitLink(nodes, mark, depth, context) + const spelling = mark.type === 'em' ? '_' : mark.type === 'strike' ? '~~' : mark.type === 'strong' ? '**' : undefined + if (spelling === undefined) return failure('unspellable-mark', `no markdown spelling holds the ${mark.type} mark`) + if (Object.keys(mark.attrs ?? {}).length > 0) return failure('unspellable-mark', `the ${mark.type} spelling holds no attributes`) + const inner = emitRun(nodes, depth + 1, context) + if (!inner.ok) return inner + const text = inner.value.map((segment) => segment.text).join('') + if (/^[ \t]|[ \t]$/.test(text)) return failure('unspellable-whitespace', `the ${mark.type} spelling cannot open or close beside whitespace`) + if (mark.type === 'em') { + return success([{ kind: 'emphasis-open', text: spelling }, ...inner.value, { kind: 'emphasis-close', text: spelling }]) + } + return success([{ kind: 'syntax', text: spelling }, ...inner.value, { kind: 'syntax', text: spelling }]) +} + +function emitCodeSpan(nodes: readonly AdfNode[], depth: number): Result { + const node = nodes[0] + if (nodes.length !== 1 || node === undefined || node.type !== 'text' || typeof node.text !== 'string') { + return failure('unspellable-mark', 'a code span holds exactly one text node') + } + if ((node.marks ?? []).length !== depth + 1) return failure('unspellable-mark', 'a code span cannot sit inside the marks it carries') + if (/[\n\r]/.test(node.text)) return failure('unspellable-mark', 'a code span holds no newline') + const fence = '`'.repeat(longestBacktickRun(node.text) + 1) + const padded = needsPadding(node.text) ? ` ${node.text} ` : node.text + return success([{ kind: 'syntax', text: `${fence}${padded}${fence}` }]) +} + +function needsPadding(text: string): boolean { + if (text.startsWith('`') || text.endsWith('`')) return true + return text.startsWith(' ') && text.endsWith(' ') && text.trim() !== '' +} + +function emitLink(nodes: readonly AdfNode[], mark: AdfMark, depth: number, context: InlineContext): Result { + const unspelled = Object.keys(mark.attrs ?? {}).find((key) => !linkAttributes.includes(key)) + if (unspelled !== undefined) return failure('unspellable-mark', `the link spelling holds no ${unspelled} attribute`) + const href = mark.attrs?.['href'] + const title = mark.attrs?.['title'] + if (typeof href !== 'string') return failure('unsupported-node-shape', 'a link mark carries no href') + if (title !== undefined && typeof title !== 'string') return failure('unsupported-node-shape', 'a link title is no string') + const node = nodes[0] + const bare = nodes.length === 1 && node !== undefined && node.type === 'text' && node.text === href && (node.marks ?? []).length === depth + 1 + if (bare && title === undefined && autolink.test(href)) return success([{ kind: 'syntax', text: `<${href}>` }]) + const destination = spellDestination(href) + if (!destination.ok) return destination + const spelledTitle = title === undefined ? success('') : spellTitle(title) + if (!spelledTitle.ok) return spelledTitle + const inner = emitRun(nodes, depth + 1, { ...context, inLinkText: true }) + if (!inner.ok) return inner + return success([{ kind: 'syntax', text: '[' }, ...inner.value, { kind: 'syntax', text: `](${destination.value}${spelledTitle.value})` }]) +} + +function spellDestination(href: string): Result { + if (controlCharacter.test(href)) return failure('unspellable-link-destination', 'a link destination holds a control character') + if (href.includes('\\')) return failure('unspellable-link-destination', 'no canonical escape spells a backslash in a link destination') + if (entityReference.test(href)) { + return failure('unspellable-link-destination', 'a link destination shaped like an entity reference decodes on the way back') + } + if (href.includes(' ')) { + if (/[<>]/.test(href)) { + return failure('unspellable-link-destination', 'no canonical escape spells an angle bracket beside a space in a link destination') + } + return success(`<${href}>`) + } + if (href.startsWith('<')) return failure('unspellable-link-destination', 'a bare link destination cannot begin with an angle bracket') + if (!balanced(href)) return failure('unspellable-link-destination', 'no canonical escape spells an unbalanced parenthesis in a link destination') + return success(href) +} + +function spellTitle(title: string): Result { + if (/["\n\r\\]/.test(title)) return failure('unspellable-link-title', 'no canonical escape spells a quote, backslash or newline in a link title') + if (entityReference.test(title)) return failure('unspellable-link-title', 'a link title shaped like an entity reference decodes on the way back') + return success(` "${title}"`) +} + +function balanced(href: string): boolean { + let depth = 0 + for (const character of href) { + if (character === '(') depth += 1 + if (character === ')') depth -= 1 + if (depth < 0) return false + } + return depth === 0 +} + +function sameMark(candidate: AdfMark | undefined, mark: AdfMark): boolean { + if (candidate === undefined || candidate.type !== mark.type) return false + return serializeCanonicalJson(candidate.attrs ?? {}, 'compact') === serializeCanonicalJson(mark.attrs ?? {}, 'compact') +} diff --git a/src/result.ts b/src/result.ts new file mode 100644 index 0000000..75fc667 --- /dev/null +++ b/src/result.ts @@ -0,0 +1,33 @@ +export type ConvertErrorCode = + | 'ambiguous-empty-code-block-language' + | 'ambiguous-ordered-list-start' + | 'not-an-adf-document' + | 'reserved-adf-language' + | 'unspellable-adjacent-lists' + | 'unspellable-code-block-language' + | 'unspellable-line-start' + | 'unspellable-link-destination' + | 'unspellable-link-title' + | 'unspellable-mark' + | 'unspellable-whitespace' + | 'unspelled-block-marks' + | 'unspelled-node-attribute' + | 'unsupported-document-version' + | 'unsupported-heading-level' + | 'unsupported-node-shape' + | 'unsupported-node-type' + +export type ConvertError = { + code: ConvertErrorCode + message: string +} + +export type Result = { error: ConvertError; ok: false } | { ok: true; value: T } + +export function failure(code: ConvertErrorCode, message: string): Result { + return { error: { code, message }, ok: false } +} + +export function success(value: T): Result { + return { ok: true, value } +} diff --git a/todo.md b/todo.md index 7fbbb26..b198e56 100644 --- a/todo.md +++ b/todo.md @@ -20,7 +20,7 @@ detail is settled at its own milestone. escape-based, never literal, since pipe cells trim and pad. At `mediaInline`, check real payloads for external-URL support — if it exists, revisit the media section's mid-text-image error and its "no slot" ground. -- [ ] **1d — Corpus start** (§10): checked-in fixtures per spec'd node, in `corpus/`, one +- [x] **1d — Corpus start** (§10): checked-in fixtures per spec'd node, in `corpus/`, one directory per contract kind (`corpus/README.md`). **Blocked on the maintainer** (§15), not to be guessed: Canonical form has no totality guard. Per `@atlaskit/adf-schema` 57.1.0 every block node it spells — `blockquote`, @@ -37,8 +37,8 @@ detail is settled at its own milestone. break §2 silently — href `https://example.com/a)b` emits `[t](https://example.com/a)b)`, read back as href `…/a` plus literal `b)`; title `He said "hi"` emits `[t](u "He said "hi"")`, which holds no title. Two defensible spellings each — angle - brackets or a backslash escape, and for titles `'…'` or `(…)` besides — so §8 leaves the - pick here. **Also blocked**: block separation is unstated for a CommonMark block beside a + brackets or a backslash escape, and for titles `'…'` or `(…)` besides — so §8 leaves + the pick here. **Also blocked**: block separation is unstated for a CommonMark block beside a directive block in a container body — an `expand` whose content is `paragraph` "A" then a `panel` (`panelType` `warning`) holding "B" spells `A` and `:::panel warning` either on consecutive lines or with a blank line between. Two defensible spellings, so §8 leaves the @@ -52,28 +52,35 @@ detail is settled at its own milestone. - [x] **1d3 — Inline nodes and marks**: date, emoji, inlineCard, mediaInline, mention, status; border, subsup, textColor, underline; the content slot's `text` attribute and the `:text{text="…"}` whitespace spelling. - - [ ] **1d4 — Opaque carry** (§3): an unknown node in both positions, the reserved `adf` info - string, and the `codeBlock` whose language is `adf`. - - [ ] **1d5 — Carve-outs and combinations**: the three carve-outs and their escapes, mark runs — - the longest-run rule, attributes included — and the runs a carry breaks, a mark spelling - that cannot open where it sits (`un**-real**istic`; the spec owes the carry a trigger), +- [ ] **2 — `adfToMarkdown`.** First real code. Each sub-item turns one corpus directory green; + the two that have no fixtures yet write them in the same chunk, tests first (§10). + - [ ] **2a — The runner and the CommonMark subset.** The corpus runner: walk + `corpus/round-trip/`, assert `adfToMarkdown` emits each `.md` byte for byte. Decide here + where §10's coverage check lives, and gate that every `corpus/**/*.json` re-serializes to + itself under the library's own canonical serializer — one implementation, keys sorted, two + spellings: two-space indent for the corpus files and the block carry's body, compact for + the inline carry. `commonmark-subset/` green. + - [ ] **2b — Block nodes.** `block-nodes/` green. + - [ ] **2c — Inline nodes and marks.** `inline-nodes/` green. + - [ ] **2d — The opaque carry** (§3). Fixtures and emitter together, into + `corpus/round-trip/opaque-carry/`: an unknown node in both positions, the reserved `adf` + info string, and the `codeBlock` whose language is `adf`. + - [ ] **2e — Carve-outs and combinations.** Fixtures and emitter together, into + `corpus/round-trip/combinations/`: the three carve-outs and their escapes, mark runs — the + longest-run rule, attributes included — and the runs a carry breaks, a mark spelling that + cannot open where it sits (`un**-real**istic`; the spec owes the carry a trigger), attribute canonicalization, a pipe cell's whitespace edges and `\u007c` for a `|` inside a - quoted attribute value, documents combining nodes rather than isolating one, and a paragraph - line inside a container body shaped like a closing fence (`:::`, `::: x`). - - [ ] **1d6 — Input normalization**: one-way markdown→ADF fixtures, not pairs — setext - headings, indented code, loose lists, `*`/`+` bullets, entity references, soft wraps. - - [ ] **1d7 — Error input**: also one-way, a markdown input per named error, asserting only - that conversion fails — malformed directives, the image gap, a claimed pipe-table line - that does not parse, the content slot, raw HTML with no mapping. Which error each returns - is pinned at milestone 3, where they are named. -- [ ] **2 — `adfToMarkdown`.** First real code — decide here where §10's coverage check lives, and - gate that every `corpus/**/*.json` re-serializes to itself under the library's own canonical - serializer: one implementation, keys sorted, two spellings — two-space indent for the corpus - files and the block carry's body, compact for the inline carry. -- [ ] **3 — `markdownToAdf`.** The CommonMark parser is the largest single component. The raw-HTML - element mapping is empty until milestone 6, so at `0.1.0` every raw-HTML construct in input - is an error result. The CommonMark spec suite runs against it from here (§10), and - `corpus/errors/` gains the error each fixture must return (1d7). + quoted attribute value, documents combining nodes rather than isolating one, and a + paragraph line inside a container body shaped like a closing fence (`:::`, `::: x`). +- [ ] **3 — `markdownToAdf`.** The CommonMark parser is the largest single component; split it + into sub-items before starting (§15). Fixtures land with the code that reads them: + `corpus/normalization/` (setext, indented code, loose lists, `*`/`+` bullets, entity + references, soft wraps — one-way, the markdown not canonical) and `corpus/errors/` (a + markdown input per named error — malformed directives, the image gap, a claimed pipe-table + line that does not parse, the content slot, raw HTML with no mapping — each with the error + it must return). The raw-HTML element mapping is empty until milestone 6, so at `0.1.0` + every raw-HTML construct in input is an error result. The CommonMark spec suite runs + against it from here (§10). - [ ] **4 — Round-trip property tests** over the corpus, both ways — the thing that proves 2 and 3. Generators emit editor-normal ADF (§2). Real sanitized ADF from live Atlassian APIs lands here too (§10), in `corpus/real-payloads/`: an ADF→markdown→ADF check with no expected