From 6f5b11800b4be0bd783f96f4315f082d87eec8a1 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Sat, 3 Oct 2026 22:09:42 +0200 Subject: [PATCH 01/23] test: cover blank lines inside an enclosing header span --- tests/fixtures/decode/blank-lines.json | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/tests/fixtures/decode/blank-lines.json b/tests/fixtures/decode/blank-lines.json index b8eb4f9..b79f9fe 100644 --- a/tests/fixtures/decode/blank-lines.json +++ b/tests/fixtures/decode/blank-lines.json @@ -211,6 +211,26 @@ "input": "m[2:]{v}:\n\n a: 1\n b: 2", "expected": { "m": { "a": { "v": 1 }, "b": { "v": 2 } } }, "specSection": "12" + }, + { + "name": "throws on blank line between a nested header and its first row inside a list item", + "input": "outer[2]:\n - inner[1]{a}:\n\n 1\n - x", + "expected": null, + "shouldError": true, + "options": { + "strict": true + }, + "specSection": "12" + }, + { + "name": "throws on blank line after a nested object inside a list item", + "input": "a[2]:\n - x:\n y: 1\n\n - 2", + "expected": null, + "shouldError": true, + "options": { + "strict": true + }, + "specSection": "12" } ] } From 0d03c77dd84169008475dbcb9bd7a8a4003a84ad Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Sat, 3 Oct 2026 22:09:47 +0200 Subject: [PATCH 02/23] =?UTF-8?q?docs:=20make=20two=20root=20scalars=20an?= =?UTF-8?q?=20error=20in=20any=20mode=20in=20=C2=A75,=20as=20=C2=A714.2=20?= =?UTF-8?q?already=20says?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- SPEC.md | 2 +- tests/fixtures/decode/validation-errors.json | 11 +++++++++++ 2 files changed, 12 insertions(+), 1 deletion(-) diff --git a/SPEC.md b/SPEC.md index 676f3ec..f612042 100644 --- a/SPEC.md +++ b/SPEC.md @@ -281,7 +281,7 @@ TOON is a deterministic, line-oriented, indentation-based notation. - Otherwise, decode an object. - An empty document (no non-blank lines after comment removal, §5.1) decodes to an empty object `{}`. A document consisting only of comment and blank lines is therefore `{}`. - The root form spans the whole document: once a root array, an empty root array (`[]`), or a keyed tabular root object is complete, no further non-comment, non-blank line may follow. In strict mode, decoders MUST error on such trailing content (§14.2) – it MUST NOT be silently discarded. In non-strict mode, decoders MAY ignore it. (A root object extends to the last line of the document, so this case does not arise for object roots.) - - In strict mode, if there are two or more non-blank depth-0 lines that are neither headers nor key-value lines, the document is invalid. Example of invalid input (strict mode): + - If there are two or more non-blank depth-0 lines that are neither headers nor key-value lines, the document is invalid in strict and non-strict mode alike (§14.2). Example of invalid input: ``` hello world diff --git a/tests/fixtures/decode/validation-errors.json b/tests/fixtures/decode/validation-errors.json index 5e20b3e..4104483 100644 --- a/tests/fixtures/decode/validation-errors.json +++ b/tests/fixtures/decode/validation-errors.json @@ -566,6 +566,17 @@ "shouldError": true, "specSection": "7.4", "minSpecVersion": "4.1" + }, + { + "name": "throws on two primitives at root depth in non-strict mode", + "input": "hello\nworld", + "expected": null, + "shouldError": true, + "options": { + "strict": false + }, + "specSection": "5", + "minSpecVersion": "4.1" } ] } From aed7ca10340461036baa0ef3aa315e0905478fe2 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Sat, 3 Oct 2026 22:09:58 +0200 Subject: [PATCH 03/23] docs: keep scalar lines out of the non-strict skip for over-indented lines --- SPEC.md | 2 +- tests/fixtures/decode/validation-errors.json | 11 +++++++++++ 2 files changed, 12 insertions(+), 1 deletion(-) diff --git a/SPEC.md b/SPEC.md index f612042..57b9ab7 100644 --- a/SPEC.md +++ b/SPEC.md @@ -458,7 +458,7 @@ Decoding of value tokens follows §4 (unquoted type inference, quoted strings, n - Lines in an object body are classified per §5.2; the rules below cover its key-value class. - A line "key:" with nothing after the colon at depth d opens an object; subsequent lines at depth > d belong to that object until the depth decreases to ≤ d. - In strict mode, the first line of a non-empty nested scope MUST be at exactly depth d+1; a depth increase of more than one level relative to the enclosing scope MUST error (§14.2). Conforming encoders never produce depth jumps; §10's depth model governs fields carried on a list-item hyphen line. - - A line deeper than the content depth of its enclosing scope whose preceding line did not open a scope belongs to no scope (e.g., a depth d+1 line directly under a depth-d primitive field). In strict mode, decoders MUST error (§14.2) – such lines MUST NOT be silently discarded. In non-strict mode, decoders MAY skip them. + - A line deeper than the content depth of its enclosing scope whose preceding line did not open a scope belongs to no scope (e.g., a depth d+1 line directly under a depth-d primitive field). In strict mode, decoders MUST error (§14.2) – such lines MUST NOT be silently discarded. In non-strict mode, decoders MAY skip them, except scalar lines, which are an error in any mode (§5.2). - A bare `key:` (no value after the colon) MUST decode as an empty or nested object, not an empty array. Empty arrays use the explicit `key: []` form (§9.1). - Lines "key: value" at the same depth are sibling fields. - Duplicate sibling keys at the same depth: see §14.3 for strict/non-strict behavior. diff --git a/tests/fixtures/decode/validation-errors.json b/tests/fixtures/decode/validation-errors.json index 4104483..213be94 100644 --- a/tests/fixtures/decode/validation-errors.json +++ b/tests/fixtures/decode/validation-errors.json @@ -577,6 +577,17 @@ }, "specSection": "5", "minSpecVersion": "4.1" + }, + { + "name": "throws on an over-indented bare token in non-strict mode", + "input": "a: 1\n hello", + "expected": null, + "shouldError": true, + "options": { + "strict": false + }, + "specSection": "8", + "minSpecVersion": "4.1" } ] } From f355fe0d3d3f715e93b166ec6d79a3d64507f8b8 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Sat, 3 Oct 2026 22:10:03 +0200 Subject: [PATCH 04/23] docs: reject surrogate escapes in pairs as well as alone --- SPEC.md | 2 +- tests/fixtures/decode/validation-errors.json | 7 +++++++ 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/SPEC.md b/SPEC.md index 57b9ab7..427691d 100644 --- a/SPEC.md +++ b/SPEC.md @@ -393,7 +393,7 @@ In quoted strings and keys, codepoints are encoded according to the following ta | CR (U+000D) | MUST emit `\r` | MUST decode `\r` → CR | | HTAB (U+0009) | MUST emit `\t` | MUST decode `\t` → HTAB | | Other U+0000–U+001F controls | MUST emit `\uXXXX` (lowercase hex SHOULD) | MUST decode `\uXXXX` (case-insensitive hex) | -| U+D800–U+DFFF lone surrogates | (not produced by valid encoders) | MUST reject when decoded from `\uXXXX` | +| U+D800–U+DFFF surrogates | (not produced by valid encoders) | MUST reject when decoded from `\uXXXX`, lone or paired | | Other BMP codepoints (U+0020–U+D7FF, U+E000–U+FFFF) | SHOULD emit literal UTF-8; MAY emit `\uXXXX` | MUST accept either form | | Supplementary scalar values (U+10000–U+10FFFF) | MUST emit as literal UTF-8 | MUST accept literal UTF-8; surrogate `\uXXXX` escapes MUST be rejected (see row above) | diff --git a/tests/fixtures/decode/validation-errors.json b/tests/fixtures/decode/validation-errors.json index 213be94..d6a529e 100644 --- a/tests/fixtures/decode/validation-errors.json +++ b/tests/fixtures/decode/validation-errors.json @@ -588,6 +588,13 @@ }, "specSection": "8", "minSpecVersion": "4.1" + }, + { + "name": "throws on a surrogate pair escape", + "input": "val: \"\\ud83d\\ude00\"", + "expected": null, + "shouldError": true, + "specSection": "7.1" } ] } From 1098ab01224c69b14cbfb0f7f49e16271e601399 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Sat, 3 Oct 2026 22:10:08 +0200 Subject: [PATCH 05/23] =?UTF-8?q?docs:=20render=20any=20other=20array=20el?= =?UTF-8?q?ement=20as=20a=20nested=20list=20in=20=C2=A79.4?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- SPEC.md | 2 +- tests/fixtures/encode/arrays-nested.json | 16 ++++++++++++++++ 2 files changed, 17 insertions(+), 1 deletion(-) diff --git a/SPEC.md b/SPEC.md index 427691d..ae2a12e 100644 --- a/SPEC.md +++ b/SPEC.md @@ -536,7 +536,7 @@ When tabular requirements are not met (encoding; including any column that is ne - Each element is rendered as a list item at depth +1 under the header: - Primitive: `- ` - Primitive array: `- [M]: v1…` - - Array of objects or non-uniform array: `- [M]:` on the hyphen line, followed by the nested array's list items at depth +1 relative to the hyphen line (i.e. +2 from the outer array header). Items are encoded recursively per §9.1–§9.4 as each item's shape requires; tabular form (§9.3) is not available in this position (a keyless fields-bearing header is valid only at the document root, §6) – encoders MUST use list form. + - Any other array (of objects, of arrays, or mixed): `- [M]:` on the hyphen line, followed by the nested array's list items at depth +1 relative to the hyphen line (i.e. +2 from the outer array header). Items are encoded recursively per §9.1–§9.4 as each item's shape requires; tabular form (§9.3) is not available in this position (a keyless fields-bearing header is valid only at the document root, §6) – encoders MUST use list form. - Object: formatted per §10 (objects as list items). Decoding: diff --git a/tests/fixtures/encode/arrays-nested.json b/tests/fixtures/encode/arrays-nested.json index bc2abc4..5cf7a45 100644 --- a/tests/fixtures/encode/arrays-nested.json +++ b/tests/fixtures/encode/arrays-nested.json @@ -139,6 +139,22 @@ ], "expected": "[1]:\n - [1]:\n - [1]: 1", "specSection": "9.4" + }, + { + "name": "encodes a non-uniform array as a list item", + "input": { + "a": [ + [ + 1, + { + "x": 1 + } + ], + 2 + ] + }, + "expected": "a[2]:\n - [2]:\n - 1\n - x: 1\n - 2", + "specSection": "9.4" } ] } From df012dca872d772dcdb43d2e642e9e78a3f7aed6 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Sat, 3 Oct 2026 22:10:15 +0200 Subject: [PATCH 06/23] docs: allow spaces between a quoted key and its colon --- SPEC.md | 2 +- tests/fixtures/decode/objects.json | 9 +++++++++ 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/SPEC.md b/SPEC.md index ae2a12e..1540ace 100644 --- a/SPEC.md +++ b/SPEC.md @@ -441,7 +441,7 @@ Keys requiring quoting per the above rules MUST be quoted in all contexts, inclu Decoding of value tokens follows §4 (unquoted type inference, quoted strings, numeric rules). This section adds key-specific requirements: - Quoted keys MUST be unescaped per §7.1; any other escape MUST error. -- Keys (quoted or unquoted) MUST be followed by ":"; missing colon MUST error (see also §14.2). +- Keys (quoted or unquoted) MUST be followed by ":", optionally after spaces (§12); missing colon MUST error (see also §14.2). - Unquoted key token (normative): an unquoted key token is the text before the first unquoted colon of a key-value line (§5.2) or entry row (§9.5), with surrounding spaces trimmed (§12); the text before a header's bracket segment; or a field name in a field list (§6). Decoders MUST accept any such token as a literal key, in strict and non-strict mode alike, even when it does not match §7.3's unquoted-key pattern: `foo-bar: 1`, `foo-bar[2]: 1,2`, and `items[1]{2key}:` are valid input. §7.3 constrains what encoders may emit unquoted, not what decoders accept. - Quoted-token boundary (normative): a token whose first character, after the trimming of §12, is `"` MUST be a complete quoted token – its closing `"` MUST be the token's last character. This applies wherever a token is extracted; any character after the closing quote MUST error. It overrides §4's "Otherwise → string" fallback. - Symmetrically for values: an unquoted value token that an encoder would have been required to quote (§7.2) is not an error. Decoders, strict mode included, MUST decode it per §4 – unless another rule of this specification assigns the token structural meaning (§5.2, §6, §9.1). Example: `key: -x` decodes to the string `-x`. §7.2 governs encoder output; it adds no decoder-side rejection. diff --git a/tests/fixtures/decode/objects.json b/tests/fixtures/decode/objects.json index af52a61..cdee58d 100644 --- a/tests/fixtures/decode/objects.json +++ b/tests/fixtures/decode/objects.json @@ -538,6 +538,15 @@ ] }, "specSection": "15" + }, + { + "name": "parses a quoted key followed by spaces before the colon", + "input": "\"a\" : 1", + "expected": { + "a": 1 + }, + "specSection": "7.4", + "minSpecVersion": "4.1" } ] } From fa732755a71d408015ea42d0367d0ef348d4ee00 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Sat, 3 Oct 2026 22:10:15 +0200 Subject: [PATCH 07/23] docs: trim field names like other key tokens --- SPEC.md | 2 +- tests/fixtures/decode/arrays-tabular.json | 14 ++++++++++++++ 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/SPEC.md b/SPEC.md index 1540ace..5eb7cc6 100644 --- a/SPEC.md +++ b/SPEC.md @@ -637,7 +637,7 @@ For an object appearing as a list item: - Depth MAY be computed as floor(indentSpaces / indentSize). - Implementations MAY accept tab characters in indentation. When they do, leading tabs are indentation and MUST be removed from the line's content before classification (§5.2). Depth computation for tabs is implementation-defined and MUST be documented. - Trailing spaces: trailing spaces (U+0020) at the end of a line are not part of the line's content. Decoders MUST strip them after the CR exclusion above and before line classification (§5.2); a line whose content is `-` followed only by spaces is therefore the bare marker for an empty-object list item (§9.4, §10), not a list item carrying an empty token. - - Token trimming: when a token is extracted – a key token before a key-value colon or an entry key's colon (§7.4, §9.5), or a value token after a key-value colon, after an array-header colon, or around each delimiter-separated token – decoders MUST trim surrounding spaces, exactly U+0020, no other characters. Any other whitespace (e.g., NBSP, or HTAB outside its delimiter role) is part of the token; internal semantics follow quoting rules. This trimming does not apply between a key and its bracket segment, where whitespace is a header syntax error (§6). + - Token trimming: when a token is extracted – a key token before a key-value colon or an entry key's colon (§7.4, §9.5), a field name in a field list (§6), or a value token after a key-value colon, after an array-header colon, or around each delimiter-separated token – decoders MUST trim surrounding spaces, exactly U+0020, no other characters. Any other whitespace (e.g., NBSP, or HTAB outside its delimiter role) is part of the token; internal semantics follow quoting rules. This trimming does not apply between a key and its bracket segment, where whitespace is a header syntax error (§6). - Comment lines are removed before any check in this section applies (§5.1). - Blank lines: - A line whose content trims to empty is blank, regardless of leading-space count; the indentation checks above do not apply to blank lines. diff --git a/tests/fixtures/decode/arrays-tabular.json b/tests/fixtures/decode/arrays-tabular.json index 64830aa..ef9c7eb 100644 --- a/tests/fixtures/decode/arrays-tabular.json +++ b/tests/fixtures/decode/arrays-tabular.json @@ -230,6 +230,20 @@ ] }, "specSection": "9.3" + }, + { + "name": "trims spaces around field names", + "input": "items[1]{ a , b }:\n 1,2", + "expected": { + "items": [ + { + "a": 1, + "b": 2 + } + ] + }, + "specSection": "12", + "minSpecVersion": "4.1" } ] } From 8285bbbff7397a7239934034dd900f6c781cc10e Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Sat, 3 Oct 2026 22:10:28 +0200 Subject: [PATCH 08/23] docs: exclude only one CR before the line end --- SPEC.md | 2 +- tests/fixtures/decode/whitespace.json | 9 +++++++++ 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/SPEC.md b/SPEC.md index 5eb7cc6..e004e44 100644 --- a/SPEC.md +++ b/SPEC.md @@ -629,7 +629,7 @@ For an object appearing as a list item: - Encoders MUST NOT emit a trailing newline at the end of the document. - Decoding: - Byte-order mark: a single U+FEFF at the very start of the document is a byte-order mark, not content – decoders MUST remove it before any processing in §5.1 and this section. A U+FEFF anywhere else is content. Encoders MUST NOT emit one. - - Line terminators: a CR (U+000D) at the end of a line is part of the line terminator, not of the line's content – decoders MUST exclude it before any processing in §5.1 and this section, thereby accepting CRLF input. A CR anywhere else in a line is content. + - Line terminators: a single CR (U+000D) at the end of a line is part of the line terminator, not of the line's content – decoders MUST exclude it before any processing in §5.1 and this section, thereby accepting CRLF input. A CR anywhere else in a line is content, including a second CR before the line end. - Strict mode: - The number of leading spaces on a line MUST be an exact multiple of indentSize; otherwise MUST error. - Tabs used as indentation MUST error (see §7.1 for tabs in quoted strings and as the HTAB delimiter). diff --git a/tests/fixtures/decode/whitespace.json b/tests/fixtures/decode/whitespace.json index 3667076..8f6aa6b 100644 --- a/tests/fixtures/decode/whitespace.json +++ b/tests/fixtures/decode/whitespace.json @@ -151,6 +151,15 @@ "a": "x\ry" }, "specSection": "12" + }, + { + "name": "keeps a second carriage return before CRLF as content", + "input": "a: x\r\r\nb: 1", + "expected": { + "a": "x\r", + "b": 1 + }, + "specSection": "12" } ] } From 140ce45f19460b7d6fbb141a50902e23367aa0cc Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Sat, 3 Oct 2026 22:10:28 +0200 Subject: [PATCH 09/23] test: cover content under a legacy `[0]` header in non-strict mode --- tests/fixtures/decode/arrays-nested.json | 14 ++++++++++++++ tests/fixtures/decode/arrays-primitive.json | 15 +++++++++++++++ 2 files changed, 29 insertions(+) diff --git a/tests/fixtures/decode/arrays-nested.json b/tests/fixtures/decode/arrays-nested.json index 69e3a40..d98e71f 100644 --- a/tests/fixtures/decode/arrays-nested.json +++ b/tests/fixtures/decode/arrays-nested.json @@ -301,6 +301,20 @@ ] }, "specSection": "9.4" + }, + { + "name": "keeps list items under a legacy [0] header in non-strict mode", + "input": "a[0]:\n - x", + "expected": { + "a": [ + "x" + ] + }, + "options": { + "strict": false + }, + "specSection": "14.1", + "minSpecVersion": "4.1" } ] } diff --git a/tests/fixtures/decode/arrays-primitive.json b/tests/fixtures/decode/arrays-primitive.json index fd5f610..dafeb40 100644 --- a/tests/fixtures/decode/arrays-primitive.json +++ b/tests/fixtures/decode/arrays-primitive.json @@ -167,6 +167,21 @@ }, "specSection": "9.1", "note": "An escaped quote does not close a quoted key, so the [2] inside it is not a bracket segment" + }, + { + "name": "keeps inline values under a legacy [0] header in non-strict mode", + "input": "a[0]: 1,2", + "expected": { + "a": [ + 1, + 2 + ] + }, + "options": { + "strict": false + }, + "specSection": "14.1", + "minSpecVersion": "4.1" } ] } From ba381a2b0b8924efed508ec685f425e00f233a6c Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Sat, 3 Oct 2026 22:10:28 +0200 Subject: [PATCH 10/23] test: cover rules that had no fixture --- tests/fixtures/decode/arrays-primitive.json | 15 ++++++++++ tests/fixtures/decode/arrays-tabular.json | 17 +++++++++++ tests/fixtures/decode/objects.json | 18 ++++++++++++ tests/fixtures/decode/validation-errors.json | 31 ++++++++++++++++++++ 4 files changed, 81 insertions(+) diff --git a/tests/fixtures/decode/arrays-primitive.json b/tests/fixtures/decode/arrays-primitive.json index dafeb40..fcc4e98 100644 --- a/tests/fixtures/decode/arrays-primitive.json +++ b/tests/fixtures/decode/arrays-primitive.json @@ -182,6 +182,21 @@ }, "specSection": "14.1", "minSpecVersion": "4.1" + }, + { + "name": "keeps every inline value when the count mismatches in non-strict mode", + "input": "tags[1]: a,b", + "expected": { + "tags": [ + "a", + "b" + ] + }, + "options": { + "strict": false + }, + "specSection": "14.1", + "minSpecVersion": "4.1" } ] } diff --git a/tests/fixtures/decode/arrays-tabular.json b/tests/fixtures/decode/arrays-tabular.json index ef9c7eb..edd87e8 100644 --- a/tests/fixtures/decode/arrays-tabular.json +++ b/tests/fixtures/decode/arrays-tabular.json @@ -244,6 +244,23 @@ }, "specSection": "12", "minSpecVersion": "4.1" + }, + { + "name": "materializes a nested field group without cells as an empty object in non-strict mode", + "input": "items[1]{a,n{x}}:\n 1", + "expected": { + "items": [ + { + "a": 1, + "n": {} + } + ] + }, + "options": { + "strict": false + }, + "specSection": "14.1", + "minSpecVersion": "4.1" } ] } diff --git a/tests/fixtures/decode/objects.json b/tests/fixtures/decode/objects.json index cdee58d..cce7d23 100644 --- a/tests/fixtures/decode/objects.json +++ b/tests/fixtures/decode/objects.json @@ -547,6 +547,24 @@ }, "specSection": "7.4", "minSpecVersion": "4.1" + }, + { + "name": "parses a hyphen-led line outside a list as a key-value line", + "input": "- a: 1", + "expected": { + "- a": 1 + }, + "specSection": "5.2" + }, + { + "name": "keeps keys that differ only in normalization form distinct", + "input": "é: 1\né: 2", + "expected": { + "é": 1, + "é": 2 + }, + "specSection": "16", + "minSpecVersion": "4.1" } ] } diff --git a/tests/fixtures/decode/validation-errors.json b/tests/fixtures/decode/validation-errors.json index d6a529e..3ac0484 100644 --- a/tests/fixtures/decode/validation-errors.json +++ b/tests/fixtures/decode/validation-errors.json @@ -595,6 +595,37 @@ "expected": null, "shouldError": true, "specSection": "7.1" + }, + { + "name": "throws on a key-value line at row depth inside a tabular array", + "input": "items[2]{a,b}:\n 1,2\n c: 3,4", + "expected": null, + "shouldError": true, + "specSection": "9.3" + }, + { + "name": "throws on a header delimiter mismatch that row widths do not expose", + "input": "items[1|]{a,b}:\n 1,2", + "expected": null, + "shouldError": true, + "options": { + "strict": true + }, + "specSection": "6" + }, + { + "name": "throws on a hyphen without a following space at item depth", + "input": "items[2]:\n - a\n -b", + "expected": null, + "shouldError": true, + "specSection": "5.2" + }, + { + "name": "throws on tabular rows at the hyphen-line field depth", + "input": "x[1]:\n - a[1]{p}:\n 1", + "expected": null, + "shouldError": true, + "specSection": "10" } ] } From d1b41293ca4c372d1195338d1821a86f4ffccd8f Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:00 +0200 Subject: [PATCH 11/23] docs: keep scalar lines out of the non-strict leniency for trailing root content --- SPEC.md | 2 +- tests/fixtures/decode/root-form.json | 12 ++++++++++++ 2 files changed, 13 insertions(+), 1 deletion(-) diff --git a/SPEC.md b/SPEC.md index e004e44..5344ebc 100644 --- a/SPEC.md +++ b/SPEC.md @@ -280,7 +280,7 @@ TOON is a deterministic, line-oriented, indentation-based notation. - Else if the document has exactly one non-blank line and it is neither a valid array header nor a key-value line (quoted or unquoted key), decode a single primitive (examples: `hello`, `42`, `true`). - Otherwise, decode an object. - An empty document (no non-blank lines after comment removal, §5.1) decodes to an empty object `{}`. A document consisting only of comment and blank lines is therefore `{}`. - - The root form spans the whole document: once a root array, an empty root array (`[]`), or a keyed tabular root object is complete, no further non-comment, non-blank line may follow. In strict mode, decoders MUST error on such trailing content (§14.2) – it MUST NOT be silently discarded. In non-strict mode, decoders MAY ignore it. (A root object extends to the last line of the document, so this case does not arise for object roots.) + - The root form spans the whole document: once a root array, an empty root array (`[]`), or a keyed tabular root object is complete, no further non-comment, non-blank line may follow. In strict mode, decoders MUST error on such trailing content (§14.2) – it MUST NOT be silently discarded. In non-strict mode, decoders MAY ignore it, except scalar lines, which are an error in any mode (§5.2). (A root object extends to the last line of the document, so this case does not arise for object roots.) - If there are two or more non-blank depth-0 lines that are neither headers nor key-value lines, the document is invalid in strict and non-strict mode alike (§14.2). Example of invalid input: ``` hello diff --git a/tests/fixtures/decode/root-form.json b/tests/fixtures/decode/root-form.json index 6378932..c5fc94c 100644 --- a/tests/fixtures/decode/root-form.json +++ b/tests/fixtures/decode/root-form.json @@ -87,6 +87,18 @@ "strict": true }, "specSection": "5" + }, + { + "name": "throws on a trailing bare token after a root array in non-strict mode", + "input": "[2]: 1,2\njunk", + "expected": null, + "shouldError": true, + "options": { + "strict": false + }, + "specSection": "5", + "note": "The non-strict leniency for trailing content excludes scalar lines (§5.2)", + "minSpecVersion": "4.1" } ] } From 18a778503d82fcc97c2b3dd92370298d71460272 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:00 +0200 Subject: [PATCH 12/23] docs: trim field entries rather than the names inside them --- SPEC.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/SPEC.md b/SPEC.md index 5344ebc..1d7c95d 100644 --- a/SPEC.md +++ b/SPEC.md @@ -637,7 +637,7 @@ For an object appearing as a list item: - Depth MAY be computed as floor(indentSpaces / indentSize). - Implementations MAY accept tab characters in indentation. When they do, leading tabs are indentation and MUST be removed from the line's content before classification (§5.2). Depth computation for tabs is implementation-defined and MUST be documented. - Trailing spaces: trailing spaces (U+0020) at the end of a line are not part of the line's content. Decoders MUST strip them after the CR exclusion above and before line classification (§5.2); a line whose content is `-` followed only by spaces is therefore the bare marker for an empty-object list item (§9.4, §10), not a list item carrying an empty token. - - Token trimming: when a token is extracted – a key token before a key-value colon or an entry key's colon (§7.4, §9.5), a field name in a field list (§6), or a value token after a key-value colon, after an array-header colon, or around each delimiter-separated token – decoders MUST trim surrounding spaces, exactly U+0020, no other characters. Any other whitespace (e.g., NBSP, or HTAB outside its delimiter role) is part of the token; internal semantics follow quoting rules. This trimming does not apply between a key and its bracket segment, where whitespace is a header syntax error (§6). + - Token trimming: when a token is extracted – a key token before a key-value colon or an entry key's colon (§7.4, §9.5), a field entry in a field list (§6), or a value token after a key-value colon, after an array-header colon, or around each delimiter-separated token – decoders MUST trim surrounding spaces, exactly U+0020, no other characters. Any other whitespace (e.g., NBSP, or HTAB outside its delimiter role) is part of the token; internal semantics follow quoting rules. This trimming does not apply between a key and its bracket segment, where whitespace is a header syntax error (§6). - Comment lines are removed before any check in this section applies (§5.1). - Blank lines: - A line whose content trims to empty is blank, regardless of leading-space count; the indentation checks above do not apply to blank lines. From 249f2e0a52d80ff47b42af001791c3efe703a8e7 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:00 +0200 Subject: [PATCH 13/23] docs: drop the second-CR example from the line terminator rule --- SPEC.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/SPEC.md b/SPEC.md index 1d7c95d..007377a 100644 --- a/SPEC.md +++ b/SPEC.md @@ -629,7 +629,7 @@ For an object appearing as a list item: - Encoders MUST NOT emit a trailing newline at the end of the document. - Decoding: - Byte-order mark: a single U+FEFF at the very start of the document is a byte-order mark, not content – decoders MUST remove it before any processing in §5.1 and this section. A U+FEFF anywhere else is content. Encoders MUST NOT emit one. - - Line terminators: a single CR (U+000D) at the end of a line is part of the line terminator, not of the line's content – decoders MUST exclude it before any processing in §5.1 and this section, thereby accepting CRLF input. A CR anywhere else in a line is content, including a second CR before the line end. + - Line terminators: a single CR (U+000D) at the end of a line is part of the line terminator, not of the line's content – decoders MUST exclude it before any processing in §5.1 and this section, thereby accepting CRLF input. A CR anywhere else in a line is content. - Strict mode: - The number of leading spaces on a line MUST be an exact multiple of indentSize; otherwise MUST error. - Tabs used as indentation MUST error (see §7.1 for tabs in quoted strings and as the HTAB delimiter). From ca181c25e0db0b3ab7d09704129fa7eae0517b20 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:01 +0200 Subject: [PATCH 14/23] test: drop the inline CR case that the double-CR case covers --- tests/fixtures/decode/whitespace.json | 8 -------- 1 file changed, 8 deletions(-) diff --git a/tests/fixtures/decode/whitespace.json b/tests/fixtures/decode/whitespace.json index 8f6aa6b..69f0f47 100644 --- a/tests/fixtures/decode/whitespace.json +++ b/tests/fixtures/decode/whitespace.json @@ -144,14 +144,6 @@ }, "specSection": "12" }, - { - "name": "keeps a carriage return inside a line as content", - "input": "a: x\ry", - "expected": { - "a": "x\ry" - }, - "specSection": "12" - }, { "name": "keeps a second carriage return before CRLF as content", "input": "a: x\r\r\nb: 1", From a539192644b7d3e47eace1f718478d612adb7d04 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:01 +0200 Subject: [PATCH 15/23] test: move the non-strict orphan scalar case next to its strict twin --- tests/fixtures/decode/indentation-errors.json | 12 ++++++++++++ tests/fixtures/decode/validation-errors.json | 11 ----------- 2 files changed, 12 insertions(+), 11 deletions(-) diff --git a/tests/fixtures/decode/indentation-errors.json b/tests/fixtures/decode/indentation-errors.json index 96097d0..2c32119 100644 --- a/tests/fixtures/decode/indentation-errors.json +++ b/tests/fixtures/decode/indentation-errors.json @@ -196,6 +196,18 @@ "shouldError": true, "specSection": "14.2", "note": "A scalar line is valid only at root primitive position" + }, + { + "name": "throws on orphan scalar line under a primitive field in non-strict mode", + "input": "a: 1\n hello", + "expected": null, + "shouldError": true, + "options": { + "strict": false + }, + "specSection": "8", + "note": "The non-strict skip for over-indented lines excludes scalar lines (§5.2)", + "minSpecVersion": "4.1" } ] } diff --git a/tests/fixtures/decode/validation-errors.json b/tests/fixtures/decode/validation-errors.json index 3ac0484..baa0ca2 100644 --- a/tests/fixtures/decode/validation-errors.json +++ b/tests/fixtures/decode/validation-errors.json @@ -578,17 +578,6 @@ "specSection": "5", "minSpecVersion": "4.1" }, - { - "name": "throws on an over-indented bare token in non-strict mode", - "input": "a: 1\n hello", - "expected": null, - "shouldError": true, - "options": { - "strict": false - }, - "specSection": "8", - "minSpecVersion": "4.1" - }, { "name": "throws on a surrogate pair escape", "input": "val: \"\\ud83d\\ude00\"", From 819d849361a9609cf21d589b4d47824dac21b133 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:01 +0200 Subject: [PATCH 16/23] test: note why the key-value line ends the tabular rows --- tests/fixtures/decode/validation-errors.json | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/fixtures/decode/validation-errors.json b/tests/fixtures/decode/validation-errors.json index baa0ca2..729625a 100644 --- a/tests/fixtures/decode/validation-errors.json +++ b/tests/fixtures/decode/validation-errors.json @@ -590,7 +590,8 @@ "input": "items[2]{a,b}:\n 1,2\n c: 3,4", "expected": null, "shouldError": true, - "specSection": "9.3" + "specSection": "9.3", + "note": "The colon precedes the delimiter, so the line ends the rows instead of becoming a second row" }, { "name": "throws on a header delimiter mismatch that row widths do not expose", From fd92d3f0928d00171de39b7f4c4be80150f62574 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:01 +0200 Subject: [PATCH 17/23] test: name the list-item header depth case in the spec's depth terms --- tests/fixtures/decode/validation-errors.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/fixtures/decode/validation-errors.json b/tests/fixtures/decode/validation-errors.json index 729625a..71f31a7 100644 --- a/tests/fixtures/decode/validation-errors.json +++ b/tests/fixtures/decode/validation-errors.json @@ -611,7 +611,7 @@ "specSection": "5.2" }, { - "name": "throws on tabular rows at the hyphen-line field depth", + "name": "throws on tabular rows at depth +1 under a list-item first-field header", "input": "x[1]:\n - a[1]{p}:\n 1", "expected": null, "shouldError": true, From ef751dbb6d0d00bb3ad2174d3f66c1c45b66db72 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:01 +0200 Subject: [PATCH 18/23] test: note why content under a legacy `[0]` header counts --- tests/fixtures/decode/arrays-nested.json | 1 + tests/fixtures/decode/arrays-primitive.json | 1 + 2 files changed, 2 insertions(+) diff --git a/tests/fixtures/decode/arrays-nested.json b/tests/fixtures/decode/arrays-nested.json index d98e71f..7a549e9 100644 --- a/tests/fixtures/decode/arrays-nested.json +++ b/tests/fixtures/decode/arrays-nested.json @@ -314,6 +314,7 @@ "strict": false }, "specSection": "14.1", + "note": "A legacy [0] is a declared length, so its content is decoded (§14.1)", "minSpecVersion": "4.1" } ] diff --git a/tests/fixtures/decode/arrays-primitive.json b/tests/fixtures/decode/arrays-primitive.json index fcc4e98..2397bab 100644 --- a/tests/fixtures/decode/arrays-primitive.json +++ b/tests/fixtures/decode/arrays-primitive.json @@ -181,6 +181,7 @@ "strict": false }, "specSection": "14.1", + "note": "A legacy [0] is a declared length, so its content is decoded (§14.1)", "minSpecVersion": "4.1" }, { From 6a7f437fe0016ac7e2ea069f898ba626bbec7910 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:01 +0200 Subject: [PATCH 19/23] test: align the list-item blank-line cases with their siblings --- tests/fixtures/decode/blank-lines.json | 14 +++++--------- 1 file changed, 5 insertions(+), 9 deletions(-) diff --git a/tests/fixtures/decode/blank-lines.json b/tests/fixtures/decode/blank-lines.json index b79f9fe..74e04cf 100644 --- a/tests/fixtures/decode/blank-lines.json +++ b/tests/fixtures/decode/blank-lines.json @@ -217,20 +217,16 @@ "input": "outer[2]:\n - inner[1]{a}:\n\n 1\n - x", "expected": null, "shouldError": true, - "options": { - "strict": true - }, - "specSection": "12" + "specSection": "14.2", + "note": "The blank is between the inner header and its first row but inside the outer array's span (§12)" }, { - "name": "throws on blank line after a nested object inside a list item", + "name": "throws on blank line between list items after a nested object", "input": "a[2]:\n - x:\n y: 1\n\n - 2", "expected": null, "shouldError": true, - "options": { - "strict": true - }, - "specSection": "12" + "specSection": "14.2", + "note": "The blank is after the nested object's content but inside the outer array's span (§12)" } ] } From 5b2dd820ad2c3e8af7fea4d3dfd893bb11fdbf2e Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:01 +0200 Subject: [PATCH 20/23] test: spell hyphen-leading like the other cases --- tests/fixtures/decode/objects.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/fixtures/decode/objects.json b/tests/fixtures/decode/objects.json index cce7d23..9582823 100644 --- a/tests/fixtures/decode/objects.json +++ b/tests/fixtures/decode/objects.json @@ -549,7 +549,7 @@ "minSpecVersion": "4.1" }, { - "name": "parses a hyphen-led line outside a list as a key-value line", + "name": "parses a hyphen-leading line outside a list as a key-value line", "input": "- a: 1", "expected": { "- a": 1 From 80dabc3be4b8b6b6cacce05a3c11fdbf75cfe260 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:01 +0200 Subject: [PATCH 21/23] test: write the normalization-form keys as escapes --- tests/fixtures/decode/objects.json | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/fixtures/decode/objects.json b/tests/fixtures/decode/objects.json index 9582823..971d7a0 100644 --- a/tests/fixtures/decode/objects.json +++ b/tests/fixtures/decode/objects.json @@ -558,10 +558,10 @@ }, { "name": "keeps keys that differ only in normalization form distinct", - "input": "é: 1\né: 2", + "input": "\u00e9: 1\ne\u0301: 2", "expected": { - "é": 1, - "é": 2 + "\u00e9": 1, + "e\u0301": 2 }, "specSection": "16", "minSpecVersion": "4.1" From 2598bb277edbc4af033dbc06ba4d6f8707889c29 Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:01 +0200 Subject: [PATCH 22/23] test: name the mixed list-item array case after its form --- tests/fixtures/encode/arrays-nested.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/fixtures/encode/arrays-nested.json b/tests/fixtures/encode/arrays-nested.json index 5cf7a45..55e0715 100644 --- a/tests/fixtures/encode/arrays-nested.json +++ b/tests/fixtures/encode/arrays-nested.json @@ -141,7 +141,7 @@ "specSection": "9.4" }, { - "name": "encodes a non-uniform array as a list item", + "name": "uses list form for an array mixing primitives and objects in list-item position", "input": { "a": [ [ From 968733f54cda36b52010c34f95774a14da1b7e8e Mon Sep 17 00:00:00 2001 From: Johann Schopplich Date: Mon, 5 Oct 2026 08:38:01 +0200 Subject: [PATCH 23/23] test: cover more rules that had no fixture --- tests/fixtures/decode/arrays-primitive.json | 10 +++++++ tests/fixtures/decode/arrays-tabular.json | 29 ++++++++++++++++++++ tests/fixtures/decode/objects.json | 20 ++++++++++++++ tests/fixtures/decode/validation-errors.json | 15 ++++++++++ tests/fixtures/decode/whitespace.json | 9 ++++++ tests/fixtures/encode/arrays-objects.json | 22 +++++++++++++++ 6 files changed, 105 insertions(+) diff --git a/tests/fixtures/decode/arrays-primitive.json b/tests/fixtures/decode/arrays-primitive.json index 2397bab..f84fb8c 100644 --- a/tests/fixtures/decode/arrays-primitive.json +++ b/tests/fixtures/decode/arrays-primitive.json @@ -198,6 +198,16 @@ }, "specSection": "14.1", "minSpecVersion": "4.1" + }, + { + "name": "decodes the inline element [] as a string, not an empty array", + "input": "a[1]: []", + "expected": { + "a": [ + "[]" + ] + }, + "specSection": "9.3" } ] } diff --git a/tests/fixtures/decode/arrays-tabular.json b/tests/fixtures/decode/arrays-tabular.json index edd87e8..269bffa 100644 --- a/tests/fixtures/decode/arrays-tabular.json +++ b/tests/fixtures/decode/arrays-tabular.json @@ -261,6 +261,35 @@ }, "specSection": "14.1", "minSpecVersion": "4.1" + }, + { + "name": "drops surplus cells when the width mismatches in non-strict mode", + "input": "items[1]{a}:\n 1,2", + "expected": { + "items": [ + { + "a": 1 + } + ] + }, + "options": { + "strict": false + }, + "specSection": "14.1", + "minSpecVersion": "4.1" + }, + { + "name": "parses unquoted colon after the delimiter in tabular row as data", + "input": "items[1]{id,note}:\n 1,a:b", + "expected": { + "items": [ + { + "id": 1, + "note": "a:b" + } + ] + }, + "specSection": "9.3" } ] } diff --git a/tests/fixtures/decode/objects.json b/tests/fixtures/decode/objects.json index 971d7a0..a53710c 100644 --- a/tests/fixtures/decode/objects.json +++ b/tests/fixtures/decode/objects.json @@ -539,6 +539,26 @@ }, "specSection": "15" }, + { + "name": "materializes __proto__ entry key as an ordinary own key", + "input": "u[1:]{x}:\n __proto__: 1", + "expected": { + "u": { + "__proto__": { + "x": 1 + } + } + }, + "specSection": "15" + }, + { + "name": "parses an unquoted key followed by spaces before the colon", + "input": "a : 1", + "expected": { + "a": 1 + }, + "specSection": "7.4" + }, { "name": "parses a quoted key followed by spaces before the colon", "input": "\"a\" : 1", diff --git a/tests/fixtures/decode/validation-errors.json b/tests/fixtures/decode/validation-errors.json index 71f31a7..a55d2b8 100644 --- a/tests/fixtures/decode/validation-errors.json +++ b/tests/fixtures/decode/validation-errors.json @@ -616,6 +616,21 @@ "expected": null, "shouldError": true, "specSection": "10" + }, + { + "name": "throws on invalid escape sequence in a quoted key", + "input": "\"a\\x\": 1", + "expected": null, + "shouldError": true, + "specSection": "7.4" + }, + { + "name": "throws on a [1] header with nothing after the colon", + "input": "a[1]:", + "expected": null, + "shouldError": true, + "specSection": "9.1", + "note": "Nothing after the colon is list form with zero items, not one empty inline value" } ] } diff --git a/tests/fixtures/decode/whitespace.json b/tests/fixtures/decode/whitespace.json index 69f0f47..f527885 100644 --- a/tests/fixtures/decode/whitespace.json +++ b/tests/fixtures/decode/whitespace.json @@ -152,6 +152,15 @@ "b": 1 }, "specSection": "12" + }, + { + "name": "keeps a second leading byte-order mark as content", + "input": "\ufeff\ufeffa: 1", + "expected": { + "\ufeffa": 1 + }, + "specSection": "12", + "minSpecVersion": "4.1" } ] } diff --git a/tests/fixtures/encode/arrays-objects.json b/tests/fixtures/encode/arrays-objects.json index c462ded..bc46a04 100644 --- a/tests/fixtures/encode/arrays-objects.json +++ b/tests/fixtures/encode/arrays-objects.json @@ -199,6 +199,28 @@ }, "expected": "items[1]:\n - a:\n b: 1", "specSection": "10" + }, + { + "name": "uses tabular form for a later field of a list-item object", + "input": { + "items": [ + { + "id": 1, + "users": [ + { + "id": 1, + "name": "Ada" + }, + { + "id": 2, + "name": "Bob" + } + ] + } + ] + }, + "expected": "items[1]:\n - id: 1\n users[2]{id,name}:\n 1,Ada\n 2,Bob", + "specSection": "10" } ] }