From 7a13f1efdcbf9dc6725d55488c2d1d5899ecdae7 Mon Sep 17 00:00:00 2001 From: Glenn Fiedler Date: Sat, 5 Sep 2026 08:17:28 -0700 Subject: [PATCH 1/7] tables: the C++ codec for unbounded arrays, work in progress (#531) Co-Authored-By: Claude Fable 5.1 --- Makefile | 251 ++- compiler/cook.go | 3 - compiler/tableslists.go | 41 +- compiler/tableslists_test.go | 79 +- compiler/target_cpp.go | 10 +- docs/COMPARISON-TABLES.md | 4 +- docs/SPEC-TABLES.md | 126 +- docs/USAGE.md | 11 +- .../bench/tables/cpp/BenchTableTable.cpp | 52 +- internal/check/tablelist.go | 23 +- internal/check/tables_test.go | 18 +- internal/codegen/cpptable/arena.go | 58 +- internal/codegen/cpptable/codecs.go | 63 +- internal/codegen/cpptable/cookwrite.go | 69 +- internal/codegen/cpptable/cpptable.go | 103 +- internal/codegen/cpptable/extent.go | 847 ++++++++++ internal/codegen/cpptable/json.go | 228 ++- internal/codegen/cpptable/lists.go | 852 ++++++++++ internal/codegen/cpptable/maps.go | 756 +-------- internal/codegen/cpptable/pointers.go | 136 +- internal/tablecook/check.go | 59 + internal/tablecook/list_test.go | 90 ++ ir/tablelist.go | 15 + tables/lists/Holders.schema | 44 + tables/lists/Migrate.schema | 21 + tables/lists/Report.schema | 25 + tables/lists/Save.schema | 48 + tables/lists/Shared.schema | 19 + tables/lists/tables.baseline | 107 ++ test/tables/lists_main.cpp | 1403 +++++++++++++++++ testdata/wire/tables/list_before_pointer.bin | Bin 0 -> 82 bytes testdata/wire/tables/list_empty.bin | Bin 0 -> 47 bytes testdata/wire/tables/list_erased.bin | Bin 0 -> 71 bytes testdata/wire/tables/list_migrates.bin | Bin 0 -> 69 bytes testdata/wire/tables/list_mixed.bin | Bin 0 -> 173 bytes testdata/wire/tables/list_nested.bin | Bin 0 -> 189 bytes testdata/wire/tables/list_nested_cook.bin | Bin 0 -> 248 bytes testdata/wire/tables/list_of_maps.bin | Bin 0 -> 163 bytes testdata/wire/tables/list_of_maps_cook.bin | Bin 0 -> 184 bytes testdata/wire/tables/list_scalars.bin | Bin 0 -> 35 bytes testdata/wire/tables/list_shared.bin | Bin 0 -> 77 bytes testdata/wire/tables/list_tables.bin | Bin 0 -> 179 bytes 42 files changed, 4569 insertions(+), 992 deletions(-) create mode 100644 internal/codegen/cpptable/extent.go create mode 100644 internal/codegen/cpptable/lists.go create mode 100644 internal/tablecook/list_test.go create mode 100644 tables/lists/Holders.schema create mode 100644 tables/lists/Migrate.schema create mode 100644 tables/lists/Report.schema create mode 100644 tables/lists/Save.schema create mode 100644 tables/lists/Shared.schema create mode 100644 tables/lists/tables.baseline create mode 100644 test/tables/lists_main.cpp create mode 100644 testdata/wire/tables/list_before_pointer.bin create mode 100644 testdata/wire/tables/list_empty.bin create mode 100644 testdata/wire/tables/list_erased.bin create mode 100644 testdata/wire/tables/list_migrates.bin create mode 100644 testdata/wire/tables/list_mixed.bin create mode 100644 testdata/wire/tables/list_nested.bin create mode 100644 testdata/wire/tables/list_nested_cook.bin create mode 100644 testdata/wire/tables/list_of_maps.bin create mode 100644 testdata/wire/tables/list_of_maps_cook.bin create mode 100644 testdata/wire/tables/list_scalars.bin create mode 100644 testdata/wire/tables/list_shared.bin create mode 100644 testdata/wire/tables/list_tables.bin diff --git a/Makefile b/Makefile index e9f360adf..9ceed6c54 100644 --- a/Makefile +++ b/Makefile @@ -36,6 +36,8 @@ SCHEMAS_TABLES_BLOBS := $(wildcard tables/blobs/*.schema) SCHEMAS_TABLES_SCALARS := $(wildcard tables/scalars/*.schema) # the MAP corpus (docs/SPEC-TABLES.md §2.8) SCHEMAS_TABLES_MAPS := $(wildcard tables/maps/*.schema) +# the UNBOUNDED ARRAY corpus (docs/SPEC-TABLES.md §2.9) +SCHEMAS_TABLES_LISTS := $(wildcard tables/lists/*.schema) # the MESSAGE FORM's corpora (docs/SPEC-TABLES.md §3.3): the three backend # messages the ruling measured, and the WIDE-VOCABULARY unit test/vocabgen # writes, whose vocabulary passes 127 ids so its message names slots on both @@ -134,6 +136,7 @@ define tables_generate $(1) generate --lang cpp --out $(2)/a2 test/tables/A2.schema $(1) generate --lang cpp --out $(2)/scalars tables/scalars $(1) generate --lang cpp --out $(2)/maps tables/maps + $(1) generate --lang cpp --out $(2)/lists tables/lists $(1) generate --lang cpp --out $(2)/scalars2 test/tables/Scalars2.schema $(1) generate --lang cpp --out $(2)/backend tables/backend $(1) generate --lang cpp --out $(2)/vocab tables/vocab @@ -141,9 +144,9 @@ endef tables_includes = -I$(1)/examples -I$(1)/pointers -I$(1)/block -I$(1)/blockhome -Itest/tables \ -I$(1)/v1 -I$(1)/v2 -I$(1)/p1 -I$(1)/p2 -I$(1)/p3 -I$(1)/jsonkeys \ - -I$(1)/messages -I$(1)/stream -I$(1)/blobs -I$(1)/m1 -I$(1)/m2 -I$(1)/a1 -I$(1)/a2 -I$(1)/g1 -I$(1)/k1 -I$(1)/k2 -I$(1)/scalars -I$(1)/scalars2 -I$(1)/maps -I$(1)/backend -I$(1)/vocab -I$(SERIALIZE) + -I$(1)/messages -I$(1)/stream -I$(1)/blobs -I$(1)/m1 -I$(1)/m2 -I$(1)/a1 -I$(1)/a2 -I$(1)/g1 -I$(1)/k1 -I$(1)/k2 -I$(1)/scalars -I$(1)/scalars2 -I$(1)/maps -I$(1)/lists -I$(1)/backend -I$(1)/vocab -I$(SERIALIZE) -build/tables-generated/.stamp: bin/schema $(SCHEMAS_TABLES) $(SCHEMAS_TABLES_POINTERS) $(SCHEMAS_TABLES_BLOCK) $(SCHEMAS_TABLES_MESSAGES) $(SCHEMAS_TABLES_BLOBS) $(SCHEMAS_TABLES_SCALARS) $(SCHEMAS_TABLES_MAPS) $(SCHEMAS_TABLES_BACKEND) $(SCHEMAS_TABLES_VOCAB) test/tables/V1.schema test/tables/V2.schema test/tables/P1.schema test/tables/P2.schema test/tables/P3.schema test/tables/JsonKeys.schema test/tables/M1.schema test/tables/M2.schema test/tables/A1.schema test/tables/A2.schema test/tables/G1.schema test/tables/K1.schema test/tables/K2.schema test/tables/Scalars2.schema +build/tables-generated/.stamp: bin/schema $(SCHEMAS_TABLES) $(SCHEMAS_TABLES_POINTERS) $(SCHEMAS_TABLES_BLOCK) $(SCHEMAS_TABLES_MESSAGES) $(SCHEMAS_TABLES_BLOBS) $(SCHEMAS_TABLES_SCALARS) $(SCHEMAS_TABLES_MAPS) $(SCHEMAS_TABLES_LISTS) $(SCHEMAS_TABLES_BACKEND) $(SCHEMAS_TABLES_VOCAB) test/tables/V1.schema test/tables/V2.schema test/tables/P1.schema test/tables/P2.schema test/tables/P3.schema test/tables/JsonKeys.schema test/tables/M1.schema test/tables/M2.schema test/tables/A1.schema test/tables/A2.schema test/tables/G1.schema test/tables/K1.schema test/tables/K2.schema test/tables/Scalars2.schema @mkdir -p build/tables-generated $(call tables_generate,./bin/schema,build/tables-generated) @touch $@ @@ -165,7 +168,7 @@ build/tables-generated/.stamp: bin/schema $(SCHEMAS_TABLES) $(SCHEMAS_TABLES_POI # THE SCAN IS BY SYMBOL, not by line, and `TableNode` is matched with its whole # spelling so a node symbol nobody has written yet is still refused. Exactly one # of those spellings is ALLOWED in a pointer-free unit and it is named below. -TABLES_ZERO_COST_SYMBOLS := TableArena|TableSlot|TableWorker|TableRef|TableRegion|kTableSegment|kTableSlab|TablePack|[A-Za-z_]*TableNode[A-Za-z_]*|is_pointer|Builder|PackMeasure|LoadMeasure|TableBlob|TableBytesView|TableStringView|AllocBytes|AllocString|BytesEmplace|StringEmplace|TableMap|TableMapHead|TableMapSegment|TableMapOrder|TableMapCursor|TableEntryKey|TableKeyOrder|TableResetMapValue|TableEntrySetKey|kTableMapSegment +TABLES_ZERO_COST_SYMBOLS := TableArena|TableSlot|TableWorker|TableRef|TableRegion|kTableSegment|kTableSlab|TablePack|[A-Za-z_]*TableNode[A-Za-z_]*|is_pointer|Builder|PackMeasure|LoadMeasure|TableBlob|TableBytesView|TableStringView|AllocBytes|AllocString|BytesEmplace|StringEmplace|TableMap|TableMapHead|TableMapSegment|TableMapOrder|TableMapCursor|TableEntryKey|TableKeyOrder|TableResetMapValue|TableEntrySetKey|kTableMapSegment|TableList|TableListHead|TableListSegment|TableListCursor|TableListElements|TableListPlace|kTableListSegment|TableExtentCarve|TableExtentUnreachedEmpty|TableWireExtent|TableRefuseReason|count_over_length|count_over_extent_cap # THE ONE NODE SPELLING A POINTER-FREE UNIT CARRIES, and it is the FORM's and # not the pointer machinery's (docs/SPEC-TABLES.md §3, §3.1): the reserved @@ -190,10 +193,10 @@ tables-zero-cost: build/tables-generated/.stamp @for f in $(TABLES_ZERO_COST_HEADERS); do \ if grep -ohE "$(TABLES_ZERO_COST_SYMBOLS)" $$f | grep -vxE "$(TABLES_ZERO_COST_ALLOWED)" | sort -u | grep -q .; then \ grep -nE "$(TABLES_ZERO_COST_SYMBOLS)" $$f | grep -vE "$(TABLES_ZERO_COST_ALLOWED)"; \ - echo "ZERO-COST GATE FAILED: pointer or map machinery leaked into $$f"; exit 1; \ + echo "ZERO-COST GATE FAILED: pointer, map or list machinery leaked into $$f"; exit 1; \ fi; \ done - @echo "tables zero-cost gate: value-only tables carry no pointer or map machinery" + @echo "tables zero-cost gate: value-only tables carry no pointer, map or list machinery" # THE NEGATIVE CONTROL. The gate above sanctions ONE node spelling, so it owes a # demonstration that it still refuses the others: the nearest neighbour of the @@ -1011,7 +1014,7 @@ tables-block-zero-cost: build/tables-generated/.stamp build/tables-generated-cs/ testdata/golden/tables/block/*Table.* testdata/golden/tables/blockhome/*Table.* \ testdata/golden/tables/messages/*Table.* testdata/golden/tables/stream/*Table.* \ testdata/golden/tables/blobs/*Table.* testdata/golden/tables/scalars/*Table.* \ - testdata/golden/tables/maps/*Table.* ; do \ + testdata/golden/tables/maps/*Table.* testdata/golden/tables/lists/*Table.* ; do \ dir=$$(basename $$(dirname $$f)); \ n=$$(( n + 1 )); \ cmp -s $$f build/tables-generated/$$dir/$$(basename $$f) || \ @@ -2066,6 +2069,12 @@ build/schema_test_maps_be: build/tables-generated/.stamp test/tables/maps_main.c @mkdir -p build $(BE_CXX) $(TABLES_CXXFLAGS) -static $(TABLES_INCLUDES) test/tables/maps_main.cpp $(MAPS_SOURCES) -o $@ +# and the LIST gate on the same host: a list's element array is the same bytes +# on both hosts because the count and the reference are the region's own scalars +build/schema_test_lists_be: build/tables-generated/.stamp test/tables/lists_main.cpp + @mkdir -p build + $(BE_CXX) $(TABLES_CXXFLAGS) -static $(TABLES_INCLUDES) test/tables/lists_main.cpp $(LISTS_SOURCES) -o $@ + # The COOK's read side, for a BIG-ENDIAN target. A cook is produced in the byte # order of the build it is cooked for (docs/SPEC-TABLES.md §7), so this is where that # stops being a sentence: the big-endian build opens the big-endian cook @@ -2088,10 +2097,11 @@ build/schema_test_block_endian_be: build/tables-generated/.stamp test/tables/blo $(BE_CXX) $(BLOCK_CXXFLAGS) -static $(BLOCK_INCLUDES) test/tables/block_endian_main.cpp $(BLOCK_SOURCES) -o $@ .PHONY: tables-big-endian -tables-big-endian: build/schema_test_tables_be build/schema_test_maps_be build/schema_test_block_endian build/schema_test_block_endian_be build/schema_test_cook build/schema_test_cook_be build/cook-open/.stamp +tables-big-endian: build/schema_test_tables_be build/schema_test_maps_be build/schema_test_lists_be build/schema_test_block_endian build/schema_test_block_endian_be build/schema_test_cook build/schema_test_cook_be build/cook-open/.stamp $(BE_RUN) ./build/schema_test_tables_be $(BE_RUN) ./build/schema_test_maps_be - @echo "big-endian leg: the wire crosses the byte order, a map's framing and its sorted entry array with it" + $(BE_RUN) ./build/schema_test_lists_be + @echo "big-endian leg: the wire crosses the byte order, a map's framing and its sorted entry array with it, and a list's element array" ./build/schema_test_block_endian write build/block-host.bin $(BE_RUN) ./build/schema_test_block_endian_be write build/block-target.bin $(BE_RUN) ./build/schema_test_block_endian_be accept build/block-target.bin @@ -2421,6 +2431,10 @@ test: build/schema_test build/schema_test_guard build/schema_test_tables build/s $(MAKE) tables-maps $(MAKE) tables-json-map-walk $(MAKE) tables-maps-negative-controls + $(MAKE) tables-lists + $(MAKE) tables-list-measure-refusals + $(MAKE) tables-json-list-walk + $(MAKE) tables-lists-negative-controls $(MAKE) tables-json-walk $(MAKE) tables-json-graph-walk $(MAKE) tables-json-negative-control @@ -2663,17 +2677,234 @@ tables-maps-negative-controls: tables-maps-sort-negative-control \ tables-maps-text-order-negative-control \ tables-maps-unreached-negative-control +# ---- THE LIST GATE (docs/SPEC-TABLES.md §2.9) ------------------------------ +# +# One binary over the `tables/lists` corpus: the builder's three, the four +# writing walks in INDEX order, the node extent a region and a cook carry, +# every reader rule §2.9 states, the migration golden, and the clamp control +# at 100,000. Its wire goldens are the reference's, pinned like every other +# table golden, and the cooks it writes are read by `schema cook-check`, whose +# scan carries §7.4's element-array clause, beside a FORGERY whose list slot +# points its array past the holder's extent, which the tool must refuse. +# +# THE SANITIZED TWIN rides beside it for the map gate's reason: segments, an +# element array carved from a node's own extent and indexing over mapped +# bytes are lifetime and bounds questions -Werror cannot see. + +LISTS_SOURCES = $$(ls build/tables-generated/lists/*Table.cpp) + +build/schema_test_lists: build/tables-generated/.stamp test/tables/lists_main.cpp + @mkdir -p build + $(CXX) $(TABLES_CXXFLAGS) $(TABLES_INCLUDES) test/tables/lists_main.cpp $(LISTS_SOURCES) -o $@ + +build/schema_test_lists_asan: build/tables-generated/.stamp test/tables/lists_main.cpp + @mkdir -p build + $(CXX) $(TABLES_CXXFLAGS) -fsanitize=address,undefined -fno-omit-frame-pointer -g \ + $(TABLES_INCLUDES) test/tables/lists_main.cpp $(LISTS_SOURCES) -o $@ + +.PHONY: tables-lists +tables-lists: build/schema_test_lists build/schema_test_lists_asan + @rm -rf build/lists-cooks && mkdir -p build/lists-cooks + SCHEMA_LIST_COOK_DIR=build/lists-cooks ./build/schema_test_lists + ./build/schema_test_lists_asan + # `schema cook-check` reads what the runtime cooked (§7.4): the root's list + # slot, every element's own slots and companions, a pointed-at holder's + # list, and an element's map, and refuses the forgery beside them + ./bin/schema cook-check --root Save build/lists-cooks/save.cook tables/lists + ./bin/schema cook-check --root Sheet build/lists-cooks/sheet.cook tables/lists + ./bin/schema cook-check --root Army build/lists-cooks/army.cook tables/lists + @if ./bin/schema cook-check --root Sheet build/lists-cooks/sheet-forged.cook tables/lists > build/lists-cooks/forged.log 2>&1; then \ + echo "LIST GATE FAILED: cook-check accepted a list slot pointing past its holder's extent"; exit 1; \ + fi + @grep -q "leaves\|extent" build/lists-cooks/forged.log || { echo "LIST GATE FAILED: the forgery was refused, but not on the element-array clause"; cat build/lists-cooks/forged.log; exit 1; } + @echo "list gate: cook-check reads three cooks the runtime wrote and refuses the forged list slot" + +# THE FOUR LoadMeasure REFUSALS are a unit test and not a `report` row (§2.9, +# §6.5): each wire is built in memory with a SYNTHETIC count, and the answer +# and the REASON are asserted, with a clean wire beside them that must measure. +.PHONY: tables-list-measure-refusals +tables-list-measure-refusals: build/schema_test_lists + ./build/schema_test_lists measure-refusals + +# THE LIST-WALK GATE (docs/SPEC-TABLES.md §2.9, §16): the list's half of the +# text form is emitted only in a unit that declares one, it is ONE half, the +# same bytes in every list-bearing .cpp, and none of it reaches a list-free +# unit, which is the zero-cost property (§2.2) holding for the text form. +.PHONY: tables-json-list-walk +tables-json-list-walk: build/tables-generated/.stamp + @rm -rf build/json-list-walk && mkdir -p build/json-list-walk + @for f in build/tables-generated/lists/*Table.cpp; do \ + out=build/json-list-walk/$$(echo $$f | tr / _); \ + awk '/---- json list walk: begin ----/,/---- json list walk: end ----/' $$f > $$out; \ + if [ ! -s $$out ]; then echo "LIST-WALK GATE FAILED: no list half in $$f"; exit 1; fi; \ + done + @first=""; for f in build/json-list-walk/*; do \ + if [ -z "$$first" ]; then first=$$f; else \ + cmp -s $$first $$f || { echo "LIST-WALK GATE FAILED: the list half in $$f is not the list half in $$first"; exit 1; }; \ + fi; \ + done + @for f in build/tables-generated/examples/*Table.cpp build/tables-generated/pointers/*Table.cpp build/tables-generated/maps/*Table.cpp; do \ + if grep -q "json list walk: begin" $$f; then \ + echo "LIST-WALK GATE FAILED: the list half reached the list-free unit $$f"; exit 1; \ + fi; \ + done + @echo "tables list-walk gate: one list half, byte-identical in $$(ls build/json-list-walk | wc -l | tr -d ' ') list-bearing .cpp files, and none in a list-free one" + +# ---- the NEGATIVE CONTROLS §2.9 names ------------------------------------ +# +# The map controls' shape: each names the sabotage, patches the GENERATOR +# through a Go overlay, regenerates the corpus, rebuilds the gate and requires +# it to go RED on a CHECK. A sabotage that patches nothing is itself a failure. +# +# $(1) the control's short name, $(2) the sed script, $(3) the file to patch, +# $(4) the sentence a reader gets when the gate stayed green. +# a comma inside a $(call) argument, spelled so the call does not split on it +comma := , + +define list_negative_control + @mkdir -p build + @sed -e $(2) $(3) > build/list-$(1).gotext + @cmp -s build/list-$(1).gotext $(3) && \ + { echo "NEGATIVE CONTROL FAILED: the $(1) sabotage patched nothing"; exit 1; } || true + @printf '{"Replace":{"%s/$(3)":"%s/build/list-$(1).gotext"}}\n' "$(CURDIR)" "$(CURDIR)" > build/list-$(1)-overlay.json + @go build -overlay=build/list-$(1)-overlay.json -o build/schema-list-$(1) ./cmd/schema + @rm -rf build/tables-list-$(1) && mkdir -p build/tables-list-$(1) + @./build/schema-list-$(1) generate --lang cpp --out build/tables-list-$(1)/lists tables/lists + @$(CXX) $(TABLES_CXXFLAGS) -Ibuild/tables-list-$(1)/lists -Itest/tables test/tables/lists_main.cpp \ + build/tables-list-$(1)/lists/*Table.cpp -o build/schema_test_lists_$(1) + @if ./build/schema_test_lists_$(1) > build/list-$(1).log 2>&1; then \ + echo "NEGATIVE CONTROL FAILED: $(4)"; exit 1; \ + fi + @grep -q "^FAIL" build/list-$(1).log || \ + { echo "NEGATIVE CONTROL FAILED: the gate went red, but not on a CHECK"; cat build/list-$(1).log; exit 1; } + @echo "negative control: $(1) turns the LIST GATE red: $$(grep -c '^FAIL' build/list-$(1).log) failures" +endef + +# THE WRITER EMITS THE ELEMENTS OUT OF ORDER. The cursor is made to step the +# builder's segments from the LAST slot back; `list_scalars` meets it, and the +# byte compare against its pinned wire goes red while measure == save holds. +.PHONY: tables-lists-order-negative-control +tables-lists-order-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,order,'s@return segment->elements + within;@return segment->elements + ( segment->used - 1 - within ); // SABOTAGED@',internal/codegen/cpptable/lists.go,the writer emitting elements out of order left the list gate GREEN) + +# `Save` EMITS A DEAD ELEMENT. `list_erased` meets it, an erase from the +# MIDDLE with an add after it, and the byte compare goes red while +# measure == save still holds, which says the sabotage is the skip. +.PHONY: tables-lists-dead-element-negative-control +tables-lists-dead-element-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,dead,'s@if ( !TableListSegmentDead( segment->dead, within ) ) { break; }@break; // SABOTAGED@',internal/codegen/cpptable/lists.go,a dead element riding on the wire left the list gate GREEN) + +# THE ELEMENT ARRAY IS LAID OUT AFTER A NESTED CONTAINER'S, breaking the +# pre-order rule: the cook's extent writer no longer steps past the element +# array before it lays each element's map, so the maps land where the elements +# are. `list_of_maps` meets it: the pinned cook's byte compare goes red, and +# `schema cook-check`'s no-overlap clause refuses the file the gate wrote. +.PHONY: tables-lists-preorder-negative-control +tables-lists-preorder-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,preorder,'s@g.pf(" at += (int64_t) cursor.count \* %d; // the whole array FIRST\\n", size)@g.pf(" // SABOTAGED: %d\\n", size)@',internal/codegen/cpptable/extent.go,laying the element array after a nested container left the list gate GREEN) + + +# THE WALK VISITS LISTS OUT OF DECLARATION ORDER, grouped after the pointer +# fields: the edge walk is made to take every list field last, so +# `list_before_pointer`'s `cover` reaches the shared node before the list +# does, and the pinned wire goes red on the node numbering. +.PHONY: tables-lists-walk-order-negative-control +tables-lists-walk-order-negative-control: bin/schema build/tables-generated/.stamp + @mkdir -p build + @printf 'func listsLast(fields []*ir.Field) []*ir.Field {\n\tvar first, last []*ir.Field\n\tfor _, f := range fields {\n\t\tif f.IsList() {\n\t\t\tlast = append(last, f)\n\t\t} else {\n\t\t\tfirst = append(first, f)\n\t\t}\n\t}\n\treturn append(first, last...)\n}\n' > build/list-walkorder-helper.txt + $(call list_negative_control,walkorder,'/guards := guardWalk(st$(comma) v.read+".")/$(comma)/^}/ s@for _$(comma) f := range st.Fields {@for _$(comma) f := range listsLast(st.Fields) { // SABOTAGED@' -e '$$r build/list-walkorder-helper.txt',internal/codegen/cpptable/pointers.go,grouping the lists after the pointer fields left the list gate GREEN) + +# A SHARED NODE IS WRITTEN TWICE: `list_shared`, whose two slots name one +# node, meets it, and the region's byte count and the text round trip's &node +# resolution go red. The sabotage reaches a fresh map entry per visit. +.PHONY: tables-lists-shared-negative-control +tables-lists-shared-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,shared,'s@const TablePackEntry \* entry = TablePackMapReach( seen, (const void \*) pointee, 0, taken, slot );@const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, 0, taken, slot ); taken = true; // SABOTAGED@',internal/codegen/cpptable/pointers.go,writing a shared node twice left the list gate GREEN) + +# THE READER CLAMPS THE COUNT against something: the 100,000-element row +# meets it, and the decoded count goes red. The sabotage clamps at 2^16. +.PHONY: tables-lists-clamp-negative-control +tables-lists-clamp-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,clamp,'s@ fill.capacity = (int32_t) n;@ fill.capacity = n > 65536 ? 65536 : (int32_t) n; // SABOTAGED@',internal/codegen/cpptable/lists.go,clamping the count left the list gate GREEN) + +# THE ELEMENT-KIND RULE DECODES ANYWAY: the `Ints`-as-`Floats` row meets it, +# and the decoded values go red. +.PHONY: tables-lists-element-kind-negative-control +tables-lists-element-kind-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,elemkind,'s@else if ( elem_kind != %d ) { r.report->kind_mismatch++; r.offset = body_end; break; }\\n", ind, elemKind)@else if ( elem_kind != %d \&\& false ) { r.report->kind_mismatch++; r.offset = body_end; break; }\\n", ind, elemKind)@',internal/codegen/cpptable/lists.go,decoding under a changed element kind left the list gate GREEN) + +# LoadMeasure OVER A LIST OF TABLES HOLDING LISTS, summed at ONE DEPTH only: +# `list_nested` meets it, and the measure goes red against the region Load +# fills. +.PHONY: tables-lists-depth-negative-control +tables-lists-depth-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,depth,'s@if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term@return true; // SABOTAGED@',internal/codegen/cpptable/lists.go,summing the extent at one depth only left the list gate GREEN) + +# THE FIT CHECK IS DROPPED: an N the list's L cannot carry measures, and the +# refusals battery goes red on the reason. +.PHONY: tables-lists-fit-negative-control +tables-lists-fit-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,fit,'s@if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; }@if ( n > (uint64_t) ( rest / elem_floor ) \&\& false ) { reason = count_over_length; return false; }@',internal/codegen/cpptable/lists.go,an N the list L cannot carry left the list gate GREEN) + +# AN UNREACHED NON-EMPTY LIST SLOT IS REFUSED by Cook and by Lock: the `Deck` +# instance whose counted array holds a list PAST ITS LIVE COUNT meets it. +.PHONY: tables-lists-unreached-negative-control +tables-lists-unreached-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,unreached,'s@inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; }@inline bool TableExtentUnreachedEmpty( int64_t ) { return true; }@',internal/codegen/cpptable/extent.go,writing an unreached non-empty list left the list gate GREEN) + +# `schema cook-check`'S ELEMENT-ARRAY CLAUSE IS DROPPED (§7.4): the forged +# cook whose list slot points past its holder's extent passes the tool, and the +# Go test that holds the clause goes red. +.PHONY: tables-lists-cook-check-negative-control +tables-lists-cook-check-negative-control: + @rm -rf build/list-cook-check-control && mkdir -p build/list-cook-check-control + @sed -e 's@if start < s.base || end > s.extent {@if ( start < s.base || end > s.extent ) \&\& false { // SABOTAGED@' \ + internal/tablecook/check.go > build/list-cook-check-control/check.go.txt + + @cmp -s internal/tablecook/check.go build/list-cook-check-control/check.go.txt && \ + { echo "NEGATIVE CONTROL FAILED: the cook-check list sabotage patched nothing"; exit 1; } || true + @printf '{"Replace":{"%s/internal/tablecook/check.go":"%s/build/list-cook-check-control/check.go.txt"}}\n' \ + "$(CURDIR)" "$(CURDIR)" > build/list-cook-check-control/overlay.json + @if go test -count=1 -overlay=build/list-cook-check-control/overlay.json \ + -run 'TestCookCheckListSlot' ./internal/tablecook/ > build/list-cook-check-control/log 2>&1; then \ + echo "NEGATIVE CONTROL FAILED: dropping cook-check's element-array clause left its test GREEN"; exit 1; \ + fi + @echo "negative control: dropping cook-check's element-array clause turns its test RED" + +# AN ALLOCATION IS PLANTED IN Load: the gate's operator new counter sees it on +# the reading path, and the allocation audit goes red. +.PHONY: tables-lists-allocation-negative-control +tables-lists-allocation-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,allocation,'s@ Element \* element = fill.array + fill.list->count;@ Element * element = fill.array + fill.list->count; delete new int; // SABOTAGED@',internal/codegen/cpptable/lists.go,an allocation planted in Load left the allocation audit GREEN) + +.PHONY: tables-lists-negative-controls +tables-lists-negative-controls: tables-lists-allocation-negative-control \ + tables-lists-order-negative-control \ + + tables-lists-dead-element-negative-control \ + tables-lists-preorder-negative-control \ + tables-lists-walk-order-negative-control \ + tables-lists-shared-negative-control \ + tables-lists-clamp-negative-control \ + tables-lists-element-kind-negative-control \ + tables-lists-depth-negative-control \ + tables-lists-fit-negative-control \ + tables-lists-unreached-negative-control \ + tables-lists-cook-check-negative-control + # Re-pin the goldens DELIBERATELY (SPEC §7.2 gates 1, 2, 7). A wire golden # breaking under an unchanged schema is stop-the-line, never a quiet re-pin # (SPEC §3.1) — this target is for intentional emitter/schema changes only. -update-goldens: build/schema_test build/schema_test_ludicrous build/schema_test_bench build/schema_test_bench_table build/schema_test_tables build/schema_test_block build/schema_test_maps +update-goldens: build/schema_test build/schema_test_ludicrous build/schema_test_bench build/schema_test_bench_table build/schema_test_tables build/schema_test_block build/schema_test_maps build/schema_test_lists @mkdir -p testdata/golden testdata/wire testdata/wire/tables go test ./internal/goldens -update -run 'TestGolden' SCHEMA_UPDATE_WIRE_GOLDENS=1 ./build/schema_test SCHEMA_UPDATE_WIRE_GOLDENS=1 ./build/schema_test_tables SCHEMA_UPDATE_WIRE_GOLDENS=1 ./build/schema_test_block SCHEMA_UPDATE_WIRE_GOLDENS=1 ./build/schema_test_maps - @for d in examples pointers block blockhome messages stream blobs scalars maps; do \ + SCHEMA_UPDATE_WIRE_GOLDENS=1 ./build/schema_test_lists + @for d in examples pointers block blockhome messages stream blobs scalars maps lists; do \ + mkdir -p testdata/golden/tables/$$d; \ cp build/tables-generated/$$d/*Table.h build/tables-generated/$$d/*Table.cpp testdata/golden/tables/$$d/ 2>/dev/null || true; \ done diff --git a/compiler/cook.go b/compiler/cook.go index 107e86ed1..c2e67e36a 100644 --- a/compiler/cook.go +++ b/compiler/cook.go @@ -120,9 +120,6 @@ func (c *Compiler) CookCheck(u *ir.Unit, root string, file []byte) (CookReport, if err := refuseToolMaps(u); err != nil { return CookReport{}, err } - if err := refuseToolLists(u); err != nil { - return CookReport{}, err - } m := tabletext.NewModel(u) res, err := tablecook.Check(m, file) if err != nil { diff --git a/compiler/tableslists.go b/compiler/tableslists.go index fc9b0eacb..41604e4d2 100644 --- a/compiler/tableslists.go +++ b/compiler/tableslists.go @@ -1,8 +1,8 @@ // The UNBOUNDED ARRAY cross-target refusal (docs/SPEC-TABLES.md §2.9, §11): -// its own file, per the registry split — a construct's refusal adds a file -// beside builtin.go rather than growing it. Every target's Generate calls -// [refuseLists], because no code generator carries the construct yet; the -// carrier registry the map's file has lands here with the first carrier. +// its own file, per the registry split: a construct's carrier registry and +// its refusal add a file beside builtin.go rather than growing it. A target +// that carries the construct registers through [registerListCarrier] from its +// own file's init, and every other target's Generate calls [refuseLists]. package compiler import ( @@ -11,29 +11,40 @@ import ( "github.com/mas-bandwidth/schema/v2/ir" ) -// refuseLists is the named refusal every target gives a unit whose table +// listTargets is the canonical name of every built-in target whose table +// backend carries an UNBOUNDED ARRAY (docs/SPEC-TABLES.md §2.9). refuseLists +// names them. +var listTargets []string + +// registerListCarrier is what a carrying target's file calls from its init, +// beside its registerBuiltin call. +func registerListCarrier(name string) { listTargets = append(listTargets, name) } + +// refuseLists is the named refusal every PORT gives a unit whose table // closure declares a `[]T` (docs/SPEC-TABLES.md §2.9, §11, §15). // // An unbounded array is a VARIABLE-CLASS construct, and the variable class is // the C++ reference's alone — the arena, the region, the node extent and the -// walks a list's elements ride in are all the reference's. The LANGUAGE takes -// the spelling and the tool's WIRE and TEXT halves carry it, so `pack` and -// `unpack` read and write one, and a generator that emitted a codec for it -// would emit one that never met an element array. So every target refuses by -// name until the reference lands the codec. +// walks a list's elements ride in are all the reference's. So the reference +// carries the codec, registers through [registerListCarrier] from its own +// init and never reaches here, and every port refuses loudly rather than +// emitting a codec that never met an element array. func refuseLists(u *ir.Unit, target string) error { fields := ir.ListFields(u) if len(fields) == 0 { return nil } - return fmt.Errorf("unit declares an unbounded array in a table closure (%s) — no code generator carries `[]T` yet, %s included: the language takes the spelling and the tool's `pack` and `unpack` read and write it, and the C++ reference lands the codec first (docs/SPEC-TABLES.md §2.9, §11, §15). Declare the array at a bound, [..N]T, which is the same bytes, and remove the bound when the reference carries it", - englishList(fields), target) + carry, flags := carriers(listTargets) + return fmt.Errorf("unit declares an unbounded array in a table closure (%s): a []T is %s only today, and the %s form is a named follow-on. Generate with %s, or declare the array at a bound, [..N]T, which is the same bytes (docs/SPEC-TABLES.md §2.9, §11, §15)", + englishList(fields), englishList(carry), target, englishList(flags)) } // refuseToolLists is the TOOL's COOK refusal (docs/SPEC-TABLES.md §2.9, §15), // the map's own, one construct over: a unit whose table closure declares a -// `[]T` is refused by name at the tool's COOK surfaces, because -// internal/tablecook does not lay out the element arrays yet. +// `[]T` is refused by name at the tool's COOK and UNCOOK surfaces, because +// internal/tablecook does not lay out the element arrays yet. `cook-check` +// is not among them: its scan carries §7.4's element-array clause, and a cook +// the C++ reference wrote is checked there. // // It is here, at the surface, rather than in the engine, and it is NAMED // rather than left to the layout. Without it the engine lays out a region @@ -44,6 +55,6 @@ func refuseToolLists(u *ir.Unit) error { if len(fields) == 0 { return nil } - return fmt.Errorf("unit declares an unbounded array in a table closure (%s) — the tool's WIRE and TEXT halves carry the construct, and its COOK half does not, so this command would lay out a region short of the element arrays rather than refusing; the C++ reference lands the cook (--lang cpp) (docs/SPEC-TABLES.md §2.9, §15)", + return fmt.Errorf("unit declares an unbounded array in a table closure (%s): the tool's WIRE and TEXT halves carry the construct and `cook-check` reads one, and its COOK half does not, so this command would lay out a region short of the element arrays rather than refusing. The C++ reference carries the cook (--lang cpp) (docs/SPEC-TABLES.md §2.9, §15)", englishList(fields)) } diff --git a/compiler/tableslists_test.go b/compiler/tableslists_test.go index 9abd614cb..aaf8fb861 100644 --- a/compiler/tableslists_test.go +++ b/compiler/tableslists_test.go @@ -1,9 +1,8 @@ package compiler // The UNBOUNDED ARRAY's cross-target refusals (docs/SPEC-TABLES.md §2.9, §11, -// §15): no code generator carries the construct yet, so every one of them -// refuses a unit that declares one BY NAME, and none of them refuses a -// list-free unit for it. +// §15): the C++ reference carries the codec, every port refuses a unit that +// declares one BY NAME, and none of them refuses a list-free unit for it. import ( "strings" @@ -26,22 +25,38 @@ table Save } ` -// TestEveryTargetRefusesAList: the refusal is a REFUSAL and not a silent -// emission of an array whose elements no backend laid out. -func TestEveryTargetRefusesAList(t *testing.T) { +// TestListsAreRefusedByEveryPort: the refusal is a REFUSAL and not a silent +// emission of an array whose elements no port laid out, and the reference +// does not refuse. +func TestListsAreRefusedByEveryPort(t *testing.T) { u := unitFromSource(t, listSrc) c := New() for _, target := range c.Targets() { - _, err := c.Generate(u, target, Options{}) - if err == nil { - t.Errorf("--lang %s emitted for a unit declaring an unbounded array", target) - continue - } - for _, want := range []string{"unbounded array", "Save.placements", "Save.scores", "[..N]T"} { - if !strings.Contains(err.Error(), want) { - t.Errorf("--lang %s: the refusal does not name %q: %v", target, want, err) + t.Run(target, func(t *testing.T) { + _, err := c.Generate(u, target, Options{}) + if target == "cpp" { + if err != nil { + t.Fatalf("--lang cpp refused an unbounded array: the reference carries the codec (schema#531): %v", err) + } + return } - } + if err == nil { + t.Fatalf("--lang %s emitted for a unit declaring an unbounded array", target) + } + for _, want := range []string{"unbounded array", "Save.placements", "Save.scores", "[..N]T", "cpp"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("--lang %s: the refusal does not name %q: %v", target, want, err) + } + } + }) + } +} + +// TestListCarrierIsTheReferenceAlone: exactly one target carries the +// construct, and it is the C++ reference (docs/SPEC-TABLES.md §2.9, §15). +func TestListCarrierIsTheReferenceAlone(t *testing.T) { + if len(listTargets) != 1 || listTargets[0] != "cpp" { + t.Fatalf("listTargets = %v, want exactly [cpp]: the variable class is the reference's (docs/SPEC-TABLES.md §2.9, §15)", listTargets) } } @@ -75,11 +90,27 @@ func TestListFieldsNamesWhatAnAuthorWrote(t *testing.T) { } } +// TestListRefusalNamesTheCarrier: what a port's refusal says: the carrier, +// the flag that generates, and the fields an author wrote. +func TestListRefusalNamesTheCarrier(t *testing.T) { + err := refuseLists(unitFromSource(t, listSrc), "go") + if err == nil { + t.Fatalf("refuseLists accepted a list-bearing unit for a non-carrier") + } + for _, want := range []string{"a []T is cpp only today", "Save.placements", "--lang cpp"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("the carrier-form refusal does not name %q: %v", want, err) + } + } +} + // TestTheToolsCookRefusesAList: the tool's WIRE and TEXT halves carry the -// construct and its COOK half does not, so the cook surfaces refuse by name -// rather than laying out a region short of the element arrays. +// construct and `cook-check` reads one, and its COOK half does not, so the +// cook and uncook surfaces refuse by name rather than laying out a region +// short of the element arrays. func TestTheToolsCookRefusesAList(t *testing.T) { - err := refuseToolLists(unitFromSource(t, listSrc)) + u := unitFromSource(t, listSrc) + err := refuseToolLists(u) if err == nil { t.Fatal("the tool's cook accepted a unit declaring an unbounded array") } @@ -91,4 +122,16 @@ func TestTheToolsCookRefusesAList(t *testing.T) { if err := refuseToolLists(unitFromSource(t, mapSrc)); err != nil { t.Fatalf("the tool refused a list-free unit: %v", err) } + c := New() + if _, _, _, err := c.Cook(u, "Save", nil, CookOptions{}); err == nil || !strings.Contains(err.Error(), "unbounded array") { + t.Errorf("the tool's cook did not refuse a list-bearing unit by name: %v", err) + } + if _, err := c.Uncook(u, "Save", nil); err == nil || !strings.Contains(err.Error(), "unbounded array") { + t.Errorf("the tool's uncook did not refuse a list-bearing unit by name: %v", err) + } + // cook-check reaches its scan: the refusal it answers for an empty file is + // the header's, not the construct's + if _, err := c.CookCheck(u, "Save", nil); err == nil || strings.Contains(err.Error(), "unbounded array") { + t.Errorf("cook-check refused a list-bearing unit by construct rather than reading the file: %v", err) + } } diff --git a/compiler/target_cpp.go b/compiler/target_cpp.go index 6023346ab..392d2e53f 100644 --- a/compiler/target_cpp.go +++ b/compiler/target_cpp.go @@ -22,13 +22,6 @@ func (cppTarget) Generate(u *ir.Unit, _ Options) (map[string][]byte, error) { if err := refusePacketVoidArms(u, "cpp"); err != nil { return nil, err } - // the UNBOUNDED ARRAY is OWED in this target (docs/SPEC-TABLES.md §2.9): - // the reference lands the codec first and registers through - // registers a carrier when it does. Until then it refuses one by name - // rather than emitting an array whose elements it never laid out. - if err := refuseLists(u, "cpp"); err != nil { - return nil, err - } files, err := cpp.Generate(u) if err != nil { return nil, err @@ -53,5 +46,6 @@ func init() { registerBuiltin(cppTarget{}, true, true, true, true) registerWideTextCarrier("cpp") // the C++ reference carries wstring(N) on the packet wire (SPEC §4.12) registerOptionalArrayCarrier("cpp") - registerMapCarrier("cpp") // the C++ reference carries the map codecs (docs/SPEC-TABLES.md §2.8) + registerMapCarrier("cpp") // the C++ reference carries the map codecs (docs/SPEC-TABLES.md §2.8) + registerListCarrier("cpp") // and the unbounded array codec (docs/SPEC-TABLES.md §2.9) } diff --git a/docs/COMPARISON-TABLES.md b/docs/COMPARISON-TABLES.md index cdc500742..bcab29d17 100644 --- a/docs/COMPARISON-TABLES.md +++ b/docs/COMPARISON-TABLES.md @@ -214,7 +214,7 @@ the source list at the end. | Optional fields | `?T` on a nested table, a type, an enum, a flags mask, a scalar and a bounded array: the value plus a `_present` bool, fixed size, no allocation (§2.3); on a string, on `bytes` and on a value whose closure is variable, a named follow-on (§15) | optional scalars via `= null`; references null when absent (FB-schema) | explicit presence: proto2 all, proto3 `optional`, editions explicit by default (PB-presence) | | Defaults | `= v` on scalars; part of the wire contract because a default is elided; changing one is a silent edit the baseline refuses (§4, §4.1, §18) | scalar defaults; "don't change existing default values" (FB-evolution) | proto2 `[default]`; proto3 zero, not serialized; editions serialize set defaults (PB-presence) | | Union | `union` with implicit `None`; arms by name hash, so add, remove, reorder freely; an arm IS a field line, so its type is any type a field's is — a `table` inside a table closure included — and an arm may carry no payload at all (§2.6, §5) | `union` of tables; structs and strings experimental; vectors of unions C++ only (FB-schema) | `oneof`; `Any` for open typing (PB-proto3) | -| Arrays | `[N]T`, `[..N]T`, `[A..B]T` on both wires, and `[]T` unbounded in a table body, whose count the data decides (§2.9, taken by the front end and by the tool, emitted by no backend yet); the bound is not wire identity, so `[]T` and `[..N]T` are the same bytes (§2, §4) | `[T]` unbounded; `[T:N]` in structs only (FB-schema) | `repeated`, unbounded, packed scalars (PB-proto3) | +| Arrays | `[N]T`, `[..N]T`, `[A..B]T` on both wires, and `[]T` unbounded in a table body, whose count the data decides (§2.9, carried by the C++ reference and the tool, refused by every port by name); the bound is not wire identity, so `[]T` and `[..N]T` are the same bytes (§2, §4) | `[T]` unbounded; `[T:N]` in structs only (FB-schema) | `repeated`, unbounded, packed scalars (PB-proto3) | | Enum-keyed arrays | `[E]T`: one slot per variant, no `None` slot, bad keys refused in every build, slots ride by name (§2.4, §3.2) | — | — | | Maps | `map[K]V` in a table body, a lookup over entries the wire carries as a sorted array of one generated `{ key, value }` table; keys are `string(N)` and the integer kinds, every other key refused by name; makes the holder variable (§2.8), in the reference and the tool | sorted vector of tables with `key` plus `LookupByKey` (FB-cpp) | `map`, integral or string keys, order undefined (PB-proto3) | | Sorted lookup in a buffer | the map's `Find`: a binary search in place over the sorted entry array, the same call in a locked region, a loaded one and an opened cook, plus an optional index the caller builds at load and never stores (§2.8) | `key`, `CreateVectorOfSortedTables`, `LookupByKey` (FB-cpp) | — | @@ -268,7 +268,7 @@ the source list at the end. | Union evolution | arms by name; add anywhere, remove, reorder (§2.6, §5) | append or explicit discriminant (FB-evolution) | adding is fine; moving an existing field into a oneof is unsafe (PB-editions) | | Flags evolution | append only, retire in place; the baseline refuses the rest (§4.1) | explicit values, any order (FB-schema) | — | | Array bound change | prefix kept, `clamped` counted; a short array fills with defaults (§4) | unbounded | unbounded | -| Unbounded arrays | `[]T` and `[]*T` in a TABLE body, whose count the data decides (§2.9, taken by the front end and by the tool, emitted by no backend yet); the same bytes as `[..N]T`, so the bound is a declaration-side fact and moving between them is silent or a clamp. Refused in a `type` body by name, which is what keeps the packet wire bounded | unbounded vectors | unbounded repeated fields | +| Unbounded arrays | `[]T` and `[]*T` in a TABLE body, whose count the data decides (§2.9, carried by the C++ reference and the tool, refused by every port by name); the same bytes as `[..N]T`, so the bound is a declaration-side fact and moving between them is silent or a clamp. Refused in a `type` body by name, which is what keeps the packet wire bounded | unbounded vectors | unbounded repeated fields | | `T` to `?T` to `*T` | `T` and `?T` are byte-identical for non-default content; to or from `*T` is a counted mismatch (§2.3, §4) | changes default semantics; required/optional changes break (FB-schema) | presence changes round-trip of defaults (PB-presence) | | Unknown fields on read | skipped by length, counted (§3, §4) | ignored (FB-evolution) | retained in the unknown set (PB-proto3) | | Unknown fields on rewrite | dropped and counted, by design; the writer has a schema | the buffer keeps them if forwarded whole (FB-evolution) | preserved and re-serialized (PB-proto3) | diff --git a/docs/SPEC-TABLES.md b/docs/SPEC-TABLES.md index 18dad63ca..04057105e 100644 --- a/docs/SPEC-TABLES.md +++ b/docs/SPEC-TABLES.md @@ -2268,12 +2268,15 @@ table that holds `[]*Self` is the ordinary legal recursion through a pointer. A hold one node, one index on the wire (§3.1), one body in a region (§6.3), one `&node` in the text (§16.7). -**THE COUNT IS THE DATA'S, and what bounds it is stated.** There is no `| max` -on a `[]T` and there is no `?[]T`, for the reasons §2.8 gives a map: a bound -would buy only a CLAMP, which drops a tail, and a fresh list is empty and an -empty list is elided under §3's by-value elision rule, the rule that elides an -empty counted array. What bounds it is three things and each belongs to one -side: +**THE COUNT IS THE DATA'S, and what bounds it is stated.** There is no +attribute that bounds a `[]T`'s COUNT and there is no `?[]T`, for the reasons +§2.8 gives a map: a bound would buy only a CLAMP, which drops a tail, and a +fresh list is empty and an empty list is elided under §3's by-value elision +rule, the rule that elides an empty counted array. **A bar attribute on a +`[]T` qualifies the ELEMENT, exactly as it does on a `[..N]T`**: `scores +[]int32 | min = 0, max = 100` bounds each score, `was` renames the field and +`json` keys its text, and none of them is a count. What bounds the count is +three things and each belongs to one side: - **On the AUTHORING side, the ARENA.** Elements are carved from the builder's arena in bulk segments (below), so an `Add` that cannot carve one answers @@ -2754,20 +2757,25 @@ each: | byte-stable output | `measure == save`, index order, one image from one value (§9) | field order is the writer's | the builder's order | | a fixed-table user pays nothing | a list-free unit carries no list machinery, held by the zero-cost gate's header scan (§2.2) | every runtime carries the repeated codec | every runtime carries the vector | -**BACKEND STATUS: OWED, not emitted.** This section is specified ahead of its -implementation, on the same terms §3.3 and §6.6 take: **the FRONT END takes the -spelling and holds every refusal above, and the TOOL's WIRE and TEXT halves -carry the construct**, so `pack` and `unpack` read and write a `[]T` and the -projections render it. **No CODE GENERATOR carries it**, every one of them -refuses a unit that declares one by name (§11), and the corpus holds no -`tables/lists`. The C++ REFERENCE lands the codec next and every other backend -keeps refusing, with the ports a named follow-on (§15). The corpus the implementation owes is `tables/lists`: -`list_empty` (an empty list beside a full one), `list_scalars`, `list_tables`, -`list_shared` (two slots naming one node beside a null slot), -`list_before_pointer` (the walk-order control above), `list_erased` (an erase -from the middle with an add after it), `list_of_maps` and `list_nested` (a list -of tables that hold lists), each crossing the wire, the text and the cook in -the harness, with the report rows the negative controls above name and +**WHERE IT IS CARRIED.** The FRONT END takes the spelling and holds every +refusal above. The C++ REFERENCE carries the codec: the `TableList` runtime, +the builder's three, the five element classes on the wire, the node extent, the +cook's write side, the text form and the descriptors, held by the corpus below. +The TOOL's WIRE and TEXT halves carry the construct, so `pack` and `unpack` +read and write a `[]T` and the projections render it, and `schema cook-check` +reads one (§7.4); the tool's COOK and UNCOOK halves do not lay out the element +arrays yet and refuse a unit that declares one by name, beside the map's own +refusal there (§15). **Every PORT refuses a unit that declares one by name** +(§11), naming the reference as the carrier, with the ports a named follow-on +(§15). The corpus is `tables/lists`: `list_empty` (an empty list beside a full +one), `list_scalars`, `list_tables`, `list_mixed` (an enum, a flags mask, a +union and a bounded scalar as elements), `list_shared` (two slots naming one +node beside a null slot), `list_before_pointer` (the walk-order control above), +`list_erased` (an erase from the middle with an add after it), `list_of_maps` +and `list_nested` (a list of tables that hold lists, and a pointed-at holder +with a list of its own), each crossing the wire, the text and the cook in +`test/tables/lists_main.cpp`, with the report rows the negative controls above +name, the two cooks pinned beside the wires, and `make tables-list-measure-refusals` beside them. **AND ONE GOLDEN IS THE MIGRATION ITSELF**, `list_migrates`, because "the same @@ -6241,13 +6249,15 @@ The builder is designed to go wide, lock-free by ownership: with the accelerators' refusal and lands with it**, so a build that has one has the other. - **BACKEND STATUS: OWED, not emitted.** The enum is specified ahead of its - implementation, on the terms §3.3 and §6.6 take. `TableRefuseReason` is - spelled in no target, in no runtime and in no tool, a bare `-1` is the whole - of a measure's answer today, and the name is not claimed either, so a unit - declaring a table or a type called `TableRefuseReason` compiles. - Owed as schema#523, with §7's check order and §19.2's block clauses, and - this line is deleted by the implementation PR that lands the behavior. + **WHERE IT IS CARRIED.** The C++ reference spells `TableRefuseReason` with + the two values a map's and an unbounded array's framing can raise, + `count_over_length` and `count_over_extent_cap`, as a native enum a unit + that declares either construct emits, and `LoadMeasure` there takes it as a + trailing out-parameter, `TableRefuseReason * reason_out = NULL`, so a caller + that does not ask keeps the signature it had. A unit with neither construct + carries neither the enum nor the parameter (§2.2). The other three values, + §7's check order and §19.2's block clauses are owed as schema#523, and no + port spells the enum yet. - **Into a builder** — the tool's path. The same tolerant decode into a fresh builder, so loaded data can be edited and locked again. **Its own refusal is a NULL** rather than a `-1`, and the report it leaves behind is @@ -8172,8 +8182,12 @@ leaves all three NULL. INLINE, and `array_bound = 0` is what says so.** Neither had a written rule before this section, so the rule is here, one shape for both: -- **`kind` is `14`** and the ELEMENT kind is `13` for a map, the element's own - kind for a list, exactly as the wire carries them (§2.8, §2.9). +- **`kind` is the ELEMENT kind, as on every array line**: `13` for a map, the + generated entry being a table; the element's own kind for a list, and `17` + for a `[]*T`, whose elements are node indices. The wire's kind `14` is what + `is_array` says, exactly as it says it for a `[..N]T`, and the `type_name` + says which of the two constructs it is, `map[...]` or the element's own, + the way `bytes` separates a byte buffer from an array of `u8` (below). - **`element_size` is the pitch**: `sizeof( Entry )` for a map, `sizeof( T )` for a list. It is the stride a walker steps, as on every array line. - **`counted` is set and `count_offset` names the `int32` count**, which sits @@ -8194,21 +8208,17 @@ before this section, so the rule is here, one shape for both: map's key is a field of the entry and a walker meets it there, and a list has no key at all. -**BACKEND STATUS, because the reference does not carry this yet.** The C++ -map descriptor emitted today leaves `kind` at `0` and `is_array` false and -describes the map through the ENTRY's own `TableTypeInfo` beside three -map-specific FUNCTION columns, `map_count`, `map_at` and `map_insert`, which -is what let the text walk reach a `TableMap` it has no name for. **It -moves to the columns above when the list lands**, and the two land together -for one reason: a second out-of-line shape would otherwise need a second set -of function columns, and three per construct is how a descriptor becomes a -per-construct API instead of a vocabulary. **The function columns do not all -go**: what the walk cannot spell for itself it still cannot spell, so a -resolver stays where a resolver is needed, and what changes is that the SHAPE -is read from `kind`, `is_array`, `counted`, `element_size` and -`array_bound = 0` like every other array's rather than inferred from a -non-NULL `entry`. A port that has neither construct carries neither column -set, which is §2.2's gate doing its job. +**ONE FUNCTION COLUMN SERVES BOTH, and it is `place`.** What the ONE text walk +cannot spell for itself it still cannot spell: placing an entry by key in a +`TableMap` or appending an element to a `TableList` needs the type +the walk has no name for, so a resolver stays where a resolver is needed, and +it is one column, `place( worker, slot, key, key_length, key_value )`, which a +map reads as an insert by key and a list reads as an append. The SHAPE is read +from `kind`, `is_array`, `counted`, `element_size` and `array_bound = 0` like +every other array's, the count from `count_offset`, and the array from the +reference at `offset`, so nothing about the two constructs is inferred from a +column of their own. A unit that has neither construct carries no `place` +column, which is §2.2's gate doing its job. **The public currency is the KEY; the storage index is private** (§2.4). `array_bound` on a keyed field is the STORAGE EXTENT, `E.Max` — derived @@ -9049,8 +9059,10 @@ in build version (§20.5). diagnostic naming the table wrapper that serves, which is the refusal a `map` takes there on the same ground; the near-miss spellings `[..]T` and `[0..]T`, each naming `[]T` as the fix, because a count bound is - a range literal and never a truncated one (SPEC.md §4.2); `?[]T`, a specified - default on one, and `| max` on one; the bounded spellings of the construct + a range literal and never a truncated one (SPEC.md §4.2); `?[]T` and a + specified default on one, while the bar attributes qualify the ELEMENT + exactly as they do on a `[..N]T` and no attribute names a count bound + (§2.9); the bounded spellings of the construct itself, `[..N][]T` and `[N][]T`, which are arrays of arrays and refused as those are (SPEC.md §4.3); a table that holds a `[]` of ITSELF by value, directly or through any chain (the @@ -10556,13 +10568,14 @@ inspects everything in the schema built: own call, because it is never stored and no golden names it (§2.8's memory layout): what is deferred is the BENCH NUMBER that says the size above which a caller should reach for it, not the surface. -- **UNBOUNDED ARRAYS IN EVERY BACKEND** (§2.9). The LANGUAGE carries the - construct, which is the parser's `[]T`, the checker's refusals and its two - claimed names, and the record's reference-and-count slot, and every backend - refuses a unit that declares one, by name (§11), until its codec lands. The C++ - reference and the tool are first: the builder's segments and `Add`, the four - walks in index order, the region load, the const `TableList` surface, the - text form's array and `schema cook-check`'s element-array clause. What a port +- **UNBOUNDED ARRAYS IN EVERY PORT** (§2.9). The LANGUAGE carries the + construct, which is the parser's `[]T`, the checker's refusals and its three + claimed names, and the record's reference-and-count slot, and every port + refuses a unit that declares one, by name (§11), until its codec lands. The + C++ reference carries it: the builder's segments and `Add`, the four walks + in index order, the region load, the const `TableList` surface, the text + form's array, and `schema cook-check`'s element-array clause in the tool; + the tool's COOK and UNCOOK halves are owed beside the map's. What a port needs is SMALLER than what a map needed, and by exactly the key: the element is an ordinary array element its measure, save and load already carry, there is no sort, no key compare and neither of the map's two reader events, and @@ -13123,7 +13136,12 @@ of declaration it names"*: carries a `bound=`**, because neither declares an extent and both take their count from the wire. An unbounded array generates no second record, because it generates no entry (§2.9), so its element's own `record` line is the only - one it needs and that line is already there under the element's name. + one it needs and that line is already there under the element's name. **A + list's line carries `kind=14`**, the array's own kind, as a map's does, + where a fixed or bounded array's line carries its ELEMENT's kind: the two + spellings are one wire (§2.9) and two storages, and this projection digests + storage. + - **A MAP's generated ENTRY takes a `record` line of its own, and it is ANONYMOUS.** The line carries the HOLDER's wire id and the MAP FIELD's wire id, joined by a dot, in place of a name, and it sorts with the named records diff --git a/docs/USAGE.md b/docs/USAGE.md index 21fee28c0..3b68f7708 100644 --- a/docs/USAGE.md +++ b/docs/USAGE.md @@ -2337,10 +2337,13 @@ its quoted decimal spelling, entries in ascending key order. ### Unbounded arrays: `placements []Placement` -*Partly landed. The front end takes the `[]T` spelling and holds every -refusal below, and `pack` and `unpack` read and write one. No backend carries -the construct — every one of them refuses a unit that declares one, by name — -and the corpus holds no `tables/lists` (SPEC-TABLES.md §2.9).* +*Carried by the C++ reference: the `TableList` runtime, the builder's `Add`, +`Each` and `Erase`, the wire, the region, the cook, the text form and the +descriptors, held by the `tables/lists` corpus. `pack`, `unpack` and +`cook-check` read and write one; the tool's `cook` and `uncook` halves do not +yet, and every port refuses a unit that declares one, by name +(SPEC-TABLES.md §2.9, §15).* + **An unbounded array is a counted array whose count the DATA decides.** It is the map with the key and the sort taken out: the same kind `14` body a diff --git a/generated/bench/tables/cpp/BenchTableTable.cpp b/generated/bench/tables/cpp/BenchTableTable.cpp index 72b431655..a431bb721 100644 --- a/generated/bench/tables/cpp/BenchTableTable.cpp +++ b/generated/bench/tables/cpp/BenchTableTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace benchtable #endif // BENCHTABLE_SCHEMA_TABLE_JSON diff --git a/internal/check/tablelist.go b/internal/check/tablelist.go index 8fc23d688..10b2fdc87 100644 --- a/internal/check/tablelist.go +++ b/internal/check/tablelist.go @@ -5,8 +5,8 @@ // almost nothing here is new machinery: the ELEMENT resolves through the // ordinary array path, so whatever `[..N]T` admits `[]T` admits and whatever // `[..N]T` refuses `[]T` refuses on the bounded array's own diagnostic. What -// this file adds is the placements the construct is refused in and the -// qualifications it does not take. +// this file adds is the placements the construct is refused in and the two +// qualifications it does not take, `?` and a specified default. package check import ( @@ -42,18 +42,11 @@ func (c *checker) checkListSpelling(f *ast.Field, inTable bool) bool { f.Name) return false } - for i := range f.Attrs { - a := &f.Attrs[i] - switch a.Key { - case "was", "json": - // a list is renamed under `was` as any field is, and takes a - // `json` key as any field does: both are about the field, not the - // construct (docs/SPEC-TABLES.md §2.9, §5, §16.4) - default: - c.errf(a.Pos, "field %s: %s does not apply to an unbounded array — THE COUNT IS THE DATA'S, and a bound would buy only a CLAMP, which drops a tail; drop the qualification, or declare the array at a bound, [..N]T, which is the same bytes with a bound (docs/SPEC-TABLES.md §2.9, §11)", - f.Name, a.Key) - return false - } - } + // THE BAR ATTRIBUTES QUALIFY THE ELEMENT, exactly as they do on a `[..N]T` + // (docs/SPEC-TABLES.md §2.9, §11): `min` and `max` bound each element, `was` + // renames the field and `json` keys its text. What the construct has no + // spelling for is a COUNT bound, and there is no attribute that names one, + // so nothing is refused here and the element path judges each attribute + // on the bounded array's own terms. return true } diff --git a/internal/check/tables_test.go b/internal/check/tables_test.go index afed786e3..5df514076 100644 --- a/internal/check/tables_test.go +++ b/internal/check/tables_test.go @@ -349,8 +349,6 @@ func TestTableRefusals(t *testing.T) { src: "package t\ntable E { a uint32 }\ntable Tab { xs ?[]E }\n"}, {name: "a default on a []T is refused by name", want: "a []T takes no specified default", src: "package t\ntable Tab { xs []int32 = 0 }\n"}, - {name: "a qualification on a []T is refused by name", want: "does not apply to an unbounded array", - src: "package t\ntable Tab { xs []int32 | max = 4 }\n"}, // the bounded spellings of the construct itself, and the element set's // own four refusals, each on the bounded array's own diagnostic {name: "[][]T is an array of arrays", want: "an array of arrays is not supported in v1", @@ -1307,3 +1305,19 @@ func TestTheIdRefusalsUnderAPlantedCollision(t *testing.T) { } } } + +// TestListQualificationBoundsTheElement: a bar attribute on a `[]T` qualifies +// the ELEMENT, exactly as it does on a `[..N]T` (docs/SPEC-TABLES.md §2.9, +// §11). The construct has no spelling for a count bound, and `max` is not +// one: it is the element's range, and the checker judges it on the bounded +// array's own terms. +func TestListQualificationBoundsTheElement(t *testing.T) { + u := buildUnit(t, "package t\ntable Tab { xs []int32 | min = 0, max = 4 }\n") + f := u.Tables["Tab"].Fields[0] + if !f.IsList() { + t.Fatalf("xs is not an unbounded array: %+v", f.Array) + } + if !f.HasIntRange || f.IntMin.Int64() != 0 || f.IntMax.Int64() != 4 { + t.Fatalf("the qualification did not bound the element: has=%v min=%v max=%v", f.HasIntRange, f.IntMin, f.IntMax) + } +} diff --git a/internal/codegen/cpptable/arena.go b/internal/codegen/cpptable/arena.go index da8befe43..dde48a91e 100644 --- a/internal/codegen/cpptable/arena.go +++ b/internal/codegen/cpptable/arena.go @@ -25,29 +25,20 @@ import ( // tableArenaRuntime is the variable-length runtime, guarded per package like // tablePrimitives so one definition survives any include order. -func tableArenaRuntime(pkg string, anyMap bool) string { +func tableArenaRuntime(pkg string, anyExtent bool) string { guard := strings.ToUpper(pkg) + "_SCHEMA_TABLE_ARENA" - // A MAP-FREE UNIT CARRIES NOT ONE SYMBOL OF THE MAP MACHINERY - // (docs/SPEC-TABLES.md §2.2, §2.8), the node map's extent cursor included: - // a pointered unit with no map emits exactly the arena runtime it always - // emitted, to the byte. - // AllocRaw is a MAP symbol (docs/SPEC-TABLES.md §2.8) and stays out of a - // map-free unit's header with the rest of them: nothing else allocates + // A UNIT WITH NEITHER A MAP NOR A LIST CARRIES NOT ONE SYMBOL OF THE EXTENT + // MACHINERY (docs/SPEC-TABLES.md §2.2, §2.8, §2.9), the node map's extent + // cursor included: a pointered unit with neither emits exactly the arena + // runtime it always emitted, to the byte. + // AllocRaw is an EXTENT symbol (docs/SPEC-TABLES.md §2.8, §2.9) and stays + // out of such a unit's header with the rest of them: nothing else allocates // storage that is not a node. - // and the refusal a node's storage answers when the FRAMING ITSELF is bad - // rather than merely unnameable — a map whose N its L cannot carry. - refusedConstant := "" - if anyMap { - refusedConstant = "\n\n// What a node's storage answers when the FRAMING ITSELF is refused rather than\n" + - "// merely unnameable: a map whose N cannot fit in its L (docs/SPEC-TABLES.md\n" + - "// §2.8). An unnameable type id commands no storage and keeps its index; this\n" + - "// one makes the whole measure answer -1 (§7.6).\nstatic const int64_t kTableNodeRefused = -2;" - } allocRaw := "" carveDecl, carveMember := "\n", "" - if anyMap { - allocRaw = ` // RAW, ZEROED storage of the bytes asked for, at the alignment asked for — a MAP's builder head and its - // entry segments (docs/SPEC-TABLES.md §2.8). It is not a node: it carries + if anyExtent { + allocRaw = ` // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries // no type id, takes no index and has no Reset, so it goes through the same // slab and span the blob path uses rather than through Alloc. uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) @@ -75,18 +66,23 @@ func tableArenaRuntime(pkg string, anyMap bool) string { return TableArenaAt( *arena, at ); // the segment came back zeroed } ` - carveDecl = "\n\n// a map's extent cursor, defined with the map runtime (docs/SPEC-TABLES.md\n// §2.8); the node map names it only through a pointer.\nstruct TableMapCarve;\n" - carveMember = "\n // WHERE A MAP'S ENTRIES LAND while this node's body decodes\n" + - " // (docs/SPEC-TABLES.md §2.8): the node's own extent on the region path\n" + - " // and the builder's arena on the tool's. It is MUTABLE because the\n" + - " // cursor belongs to ONE node's decode and the dispatch that owns that\n" + - " // node holds the map by const reference, exactly as it did before maps\n" + - " // existed — the decoder's signature does not move for a construct it\n" + - " // may not carry.\n mutable TableMapCarve * carve = NULL;\n" + - " // and the TOOL's path's allocation front, set once: there a map's\n" + - " // entries are the builder's arena's rather than a node's extent.\n" + - " TableWorker * worker = NULL;" + carveDecl = "\n\n// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md\n// §2.8, §2.9); the node map names it only through a pointer.\nstruct TableExtentCarve;\n" + carveMember = "\n // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body\n" + + " // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the\n" + + " // region path and the builder's arena on the tool's. It is MUTABLE\n" + + " // because the cursor belongs to ONE node's decode and the dispatch that\n" + + " // owns that node holds the map by const reference, exactly as it did\n" + + " // before either construct existed. The decoder's signature does not\n" + + " // move for a construct it may not carry.\n mutable TableExtentCarve * carve = NULL;\n" + + " // and the TOOL's path's allocation front, set once: there the arrays\n" + + " // are the builder's arena's rather than a node's extent.\n" + + " TableWorker * worker = NULL;\n" + + " // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the\n" + + " // int32 cap met while a body decoded. LoadBuilder answers NULL for it\n" + + " // and moves no counter; mutable for the reason the cursor is.\n" + + " mutable bool refused = false;" } + return `#ifndef ` + guard + ` #define ` + guard + ` @@ -731,7 +727,7 @@ static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts th // The not-materialized sentinel (§6.3): a record whose type id this build could // not name. Distinct from every real offset including the root's 0, so an index // resolving through it yields NULL and can never fabricate the root. -static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull;` + refusedConstant + ` +static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; // ---- the numbering, on the SAVE side ---- // diff --git a/internal/codegen/cpptable/codecs.go b/internal/codegen/cpptable/codecs.go index c430e9164..292331510 100644 --- a/internal/codegen/cpptable/codecs.go +++ b/internal/codegen/cpptable/codecs.go @@ -219,6 +219,15 @@ func (g *tableGen) emitTableStorageField(f *ir.Field) { g.pf(" TableMap<%s> %s; // %s — the sorted entry array, empty until an insert\n", f.MapEntry.Name, f.Name, ir.TableTypeSpelling(f)) return } + if f.IsList() { + // AN UNBOUNDED ARRAY FIELD IS SIXTEEN BYTES (docs/SPEC-TABLES.md §2.9, + // §7.2): the map's slot exactly, a self-relative reference to the + // element array and the live count, then padding to eight. The ELEMENTS + // are not here: they are by-value records inside the holder's node + // extent, laid after the record's own storage. + g.pf(" %s %s; // %s: the element array, empty until an Add\n", g.listStorageType(f), f.Name, ir.TableTypeSpelling(f)) + return + } if f.Type.Pointer { // a pointer is EIGHT BYTES and no address: an arena offset while the // builder is mutable, a self-relative delta once packed. That is what @@ -346,7 +355,15 @@ func (g *tableGen) emitTableResetField(f *ir.Field) { // null in both encodings and the live count is zero. The builder's // head and its segments are the arena's, and Reset does not free them // — the arena's own reset is what reclaims a dead entry's storage. - g.pf(" value.%s.entries.value = 0; // %s — empty\n", f.Name, ir.TableTypeSpelling(f)) + g.pf(" value.%s.entries.value = 0; // %s: empty\n", f.Name, ir.TableTypeSpelling(f)) + g.pf(" value.%s.count = 0;\n", f.Name) + g.pf(" value.%s.padding = 0;\n", f.Name) + return + } + if f.IsList() { + // a fresh list is EMPTY (docs/SPEC-TABLES.md §2.9), on the map's terms: + // the reference is null in both encodings and the live count is zero + g.pf(" value.%s.elements.value = 0; // %s: empty\n", f.Name, ir.TableTypeSpelling(f)) g.pf(" value.%s.count = 0;\n", f.Name) g.pf(" value.%s.padding = 0;\n", f.Name) return @@ -712,6 +729,10 @@ func (g *tableGen) emitTableMeasureField(f *ir.Field) { g.emitMapMeasureField(f) return } + if f.IsList() { + g.emitListMeasureField(f) + return + } id := tableFieldWireId(f) kind := tableScalarKind(f) width := tableKindWidth(kind) @@ -1103,6 +1124,10 @@ func (g *tableGen) emitTableWriteField(f *ir.Field) { g.emitMapWriteField(f) return } + if f.IsList() { + g.emitListWriteField(f) + return + } id := tableFieldWireId(f) kind := tableScalarKind(f) elemKind := kind @@ -1579,6 +1604,10 @@ func (g *tableGen) emitTableReadField(f *ir.Field, kind int) { g.emitMapReadField(f) return } + if f.IsList() { + g.emitListReadField(f) + return + } switch { case f.KeyEnum != "": // each triple is placed by its KEY REFERENCE, so a slot lands by name @@ -2146,8 +2175,38 @@ func (g *tableGen) emitFieldInfo(f *ir.Field, sp fieldSpelling, hoisted bool) { elemSize = elemSizeOverride countOffset = "0xffffffffu" } + if f.IsMap() || f.IsList() { + // AN OUT-OF-LINE ARRAY (docs/SPEC-TABLES.md §8.1): an array field whose + // elements are not inline, and array_bound = 0 is what says so. kind is + // the ELEMENT's, as on every array line: 13 for a map's entry, the + // element's own for a list, 17 for a []*T. is_array and counted are + // set, elem_size is the pitch, count_offset names the int32 count + // beside the reference in the sixteen-byte slot, and offset names the + // REFERENCE, which a walker resolves before it steps. + isArray = true + counted = true + bound = "0" + countOffset = fmt.Sprintf("(uint32_t) offsetof( %s, %s.count )", sp.owner, sp.member) + if f.IsMap() { + kind = tkTable + elemSize = fmt.Sprintf("(uint32_t) sizeof( %s )", f.MapEntry.Name) + } else { + kind = listElementWireKind(f) + elemSize = fmt.Sprintf("(uint32_t) sizeof( %s )", g.listElementType(f)) + } + } table := "NULL" + if f.IsMap() { + // the generated ENTRY's descriptor: fields[0] is the key and fields[1] + // the value (§2.8, §8.1) + if hoisted { + table = "&" + f.MapEntry.Name + "TableInfo" + } else { + table = fmt.Sprintf("%sTableType()", f.MapEntry.Name) + } + } if _, isStruct := f.Type.Ref.(*ir.Struct); f.Type.Kind == ir.TNamed && isStruct { + if hoisted { // an ADDRESS, not a call: constant-initialisable, so a // self-reference (Node -> *Node) is simply &NodeTableInfo @@ -2269,5 +2328,5 @@ func (g *tableGen) emitFieldInfo(f *ir.Field, sp fieldSpelling, hoisted bool) { sp.indent, f.Name, ir.TableFieldJsonKey(f), tableFieldTypeName(f), id, kind, isArray, pointerColumn, counted, f.Type.Optional, bound, sp.owner, sp.member, elemSize, countOffset, presentOffset, table, hasRange, rangeMin, rangeMax, fracBits, wide, enumMax, enumName, variantId, - keyTypeName, keyName, keyId, arms, g.mapColumn(f), sp.guard) + keyTypeName, keyName, keyId, arms, g.placeColumn(f), sp.guard) } diff --git a/internal/codegen/cpptable/cookwrite.go b/internal/codegen/cpptable/cookwrite.go index aa1200fab..77d815af4 100644 --- a/internal/codegen/cpptable/cookwrite.go +++ b/internal/codegen/cpptable/cookwrite.go @@ -220,7 +220,8 @@ func (g *tableGen) emitCookWriteSurface(members []*ir.Struct) { for _, st := range bodies { g.emitCookWriteBody(st) } - g.emitCookMapSurface(members) + g.emitCookExtentSurface(members) + for _, st := range bodies { if !st.IsTable || st.IsMapEntry() { // a `type` is no root, and a map's generated ENTRY is §2.8's one @@ -250,18 +251,16 @@ func (g *tableGen) emitCookWriteBody(st *ir.Struct) { g.pf(" (void) at; (void) value; (void) order; // a record with no field writes nothing\n") } } - if variable && len(ml.Fields) > 0 && g.noVariableEdges(st) { - g.pf(" (void) ctx; (void) region; // no reference below this node: the class was decided by a pointer elsewhere in its closure\n") + if variable && len(ml.Fields) > 0 && g.noCookRefs(st) { + g.pf(" (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure\n") } - if len(ml.Fields) > 0 && onlyMapFields(st) { - // every field is a MAP, whose slot the extent writer fills: this body - // writes the empty sixteen bytes and reads nothing off the value, and - // no reference of its own resolves here + if len(ml.Fields) > 0 && onlyExtentFields(st) { + // every field is a LIST or a MAP, whose slot the extent writer fills: + // this body writes the empty sixteen bytes and reads nothing off the + // value g.pf(" (void) value;\n") - if variable && !g.noVariableEdges(st) { - g.pf(" (void) ctx; (void) region;\n") - } } + for i := range ml.Fields { fl := &ml.Fields[i] g.emitCookWriteField(st, fl.Field, fl.Offset) @@ -281,13 +280,14 @@ func (g *tableGen) emitCookWriteField(st *ir.Struct, f *ir.Field, offset int64) // in a member of the record (docs/SPEC-TABLES.md §2.6), and its pieces sit at // the arms' shared offset. func (g *tableGen) emitCookWriteFieldAs(st *ir.Struct, f *ir.Field, offset int64, name, base, sfx string) { - if f.IsMap() { - // THE SLOT IS THE EXTENT WRITER'S (docs/SPEC-TABLES.md §2.8): the + if f.IsMap() || f.IsList() { + // THE SLOT IS THE EXTENT WRITER'S (docs/SPEC-TABLES.md §2.8, §2.9): the // reference is a delta to an array this record's own extent holds, and - // only CookMaps knows where that landed. The record's sixteen bytes + // only CookExtent knows where that landed. The record's sixteen bytes // are written EMPTY here, which is what a node the walk never reaches - // keeps — and is why an unreached non-empty map is refused (§7.6). - g.pf(" table_cook_put( %s + %d, 0, 8, order ); // %s: the entry array's delta, filled by the extent writer\n", base, offset, f.Name) + // keeps, and is why an unreached non-empty list or map is refused + // (§7.6). + g.pf(" table_cook_put( %s + %d, 0, 8, order ); // %s: the array's delta, filled by the extent writer\n", base, offset, f.Name) g.pf(" table_cook_put( %s + %d, 0, 4, order ); // and its count\n", base, offset+8) return } @@ -531,22 +531,22 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf("// eight. The offsets go into the region's table when it has one, and are only\n") g.pf("// summed when it does not (a measure). A type id the numbering carries that\n") g.pf("// this root cannot name is the two walks disagreeing, and it is refused.\n") - if g.anyMap { - g.pf("// A NODE'S SIZE DEPENDS ON ITS VALUE where a map rides in its extent\n") + if g.anyExtent { + g.pf("// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent\n") g.pf("// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context\n") - g.pf("// the numbering walked and reads the same maps that walk read.\n") + g.pf("// the numbering walked and reads the same arrays that walk read.\n") g.pf("template \ninline bool %sCookLayout( const Ctx & ctx, const %s & root, const TableNumbering & numbering, TableCookRegion & region )\n{\n", n, n) } else { g.pf("inline bool %sCookLayout( const TableNumbering & numbering, TableCookRegion & region )\n{\n", n) } g.pf(" region.numbering = &numbering;\n") g.pf(" region.count = numbering.count + 1;\n") - if g.anyMap && g.hasMapExtent(st) { - g.pf(" const int64_t root_extent = %sMapExtent( ctx, root );\n", n) + if g.anyExtent && g.hasExtent(st) { + g.pf(" const int64_t root_extent = %sExtent( ctx, root );\n", n) g.pf(" if ( root_extent < 0 ) { return false; }\n") g.pf(" int64_t offset = %d + root_extent; // the root at zero, its extent behind it\n", cookAlignUp(ml.Size, ir.RegionAlignFloor)) } else { - if g.anyMap { + if g.anyExtent { g.pf(" (void) root;\n") } g.pf(" int64_t offset = %d; // the root, at zero\n", ml.Size) @@ -559,7 +559,7 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf(" switch ( numbering.entries[k].type_id )\n {\n") for _, t := range reachable { tl := ir.RecordLayout(g.unit, t) - if g.anyMap && g.hasMapExtent(t) { + if g.anyExtent && g.hasExtent(t) { g.pf(" case 0x%016xull: // %s\n", ir.TableWireId(t.Name), t.Name) g.emitCookNodeBytes(t, " ", fmt.Sprintf("*(const %s *) numbering.entries[k].node", t.Name), "return false;") g.pf(" break;\n") @@ -596,7 +596,7 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf(" TableNumberingInit( numbering, allocator );\n") g.pf(" TableCookRegion region;\n") g.pf(" int64_t bytes = -1;\n") - if g.anyMap { + if g.anyExtent { g.pf(" if ( %sNumberFrom( ctx, numbering, root ) && %sCookLayout( ctx, root, numbering, region ) )\n {\n", n, n) } else { g.pf(" if ( %sNumberFrom( ctx, numbering, root ) && %sCookLayout( numbering, region ) )\n {\n", n, n) @@ -627,7 +627,7 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf(" bool ok = %sNumberFrom( ctx, numbering, root );\n", n) g.pf(" if ( ok )\n {\n") g.pf(" region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) );\n") - if g.anyMap { + if g.anyExtent { g.pf(" ok = region.offsets != NULL && %sCookLayout( ctx, root, numbering, region );\n", n) } else { g.pf(" ok = region.offsets != NULL && %sCookLayout( numbering, region );\n", n) @@ -644,7 +644,7 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf(" region.base = raw + data_offset;\n") g.pf(" // the DATA part: the root at the region's base, then every numbered\n") g.pf(" // node at the offset the layout gave it, each through its own writer\n") - if g.anyMap { + if g.anyExtent { g.pf(" ok = %sCookNode( ctx, region, region.base, root, order );\n", n) } else { g.pf(" ok = %sCookBody( ctx, region, region.base, root, order );\n", n) @@ -660,7 +660,7 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf(" switch ( numbering.entries[k].type_id )\n {\n") } for _, t := range reachable { - if g.anyMap { + if g.anyExtent { g.pf(" case 0x%016xull: ok = %sCookNode( ctx, region, at, *(const %s *) node, order ); break; // %s\n", ir.TableWireId(t.Name), t.Name, t.Name, t.Name) continue } @@ -737,3 +737,20 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf(" TableArenaCtx ctx = { &builder.arena };\n") g.pf(" return %sCookFrom( ctx, *(const %s *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator );\n}\n\n", n, n) } + +// noCookRefs reports a record whose cook body resolves no reference through +// the context: no pointer or byte buffer slot, no by-value nesting or union arm +// that holds one. A list's and a map's slots are not references here, the +// extent writer fills them (§2.8, §2.9), so a record whose only edges are +// those has nothing to resolve, and its ctx and region are named unused. +func (g *tableGen) noCookRefs(st *ir.Struct) bool { + for _, f := range st.Fields { + if f.IsMap() || f.IsList() { + continue + } + if g.edgeOf(f) != edgeNone { + return false + } + } + return true +} diff --git a/internal/codegen/cpptable/cpptable.go b/internal/codegen/cpptable/cpptable.go index 544cd25ce..da168373a 100644 --- a/internal/codegen/cpptable/cpptable.go +++ b/internal/codegen/cpptable/cpptable.go @@ -89,6 +89,13 @@ type tableGen struct { // §2.8). It gates the map runtime and every map-shaped walk, so not one // symbol of the machinery reaches a map-free unit's header (§2.2). anyMap bool + // anyList is the unit declaring at least one `[]T` (docs/SPEC-TABLES.md + // §2.9). It gates the list runtime and the builder's three the same way. + anyList bool + // anyExtent is either: the unit carries the NODE EXTENT machinery both + // constructs share: the carve, the framing walk, the extent walks and the + // refusal reason (§2.8, §2.9, §6.5). A unit with neither carries none of it. + anyExtent bool // blocks is the unit's BLOCK FORM surface (docs/SPEC-TABLES.md §19), nil when // no table is marked `| block`. Nil is what makes the zero-cost gate // answerable by asking one question (§2.2). @@ -355,7 +362,7 @@ struct TableKeyed // same way would be a redefinition. func tableInlineMacro(pkg string) string { return strings.ToUpper(pkg) + "_TABLE_INLINE" } -func tablePrimitives(pkg string, anyVariable bool, anyKeyed bool, anyMap bool, idCap int, u *ir.Unit) string { +func tablePrimitives(pkg string, anyVariable bool, anyKeyed bool, anyExtent bool, idCap int, u *ir.Unit) string { // THE ID TABLE'S CAPACITY IS A COMPILE-TIME FACT of the unit (§3): the // distinct names its table closure can spell, so a save allocates nothing. // The bucket count is the next power of two at twice the capacity, so the @@ -379,24 +386,35 @@ func tablePrimitives(pkg string, anyVariable bool, anyKeyed bool, anyMap bool, i // the two pointer-era descriptor members exist only in a unit that HAS // pointers: a unit of value-only tables emits the descriptor surface it // always emitted, to the byte (docs/SPEC-TABLES.md §2, the zero-cost gate) - // the MAP columns (docs/SPEC-TABLES.md §2.8, §16), emitted only into a unit - // that declares one: the generated ENTRY's descriptor, whose two fields - // ARE the key and the value, and the three thunks the ONE walk cannot - // spell for itself because TableMap is a type it has no name for. - mapFieldMember := "" - if anyMap { - mapFieldMember = "\n // a MAP (docs/SPEC-TABLES.md §2.8): the generated ENTRY's descriptor —\n" + - " // fields[0] is the key and fields[1] the value — and the three the ONE\n" + - " // text walk cannot spell for itself, because TableMap is a type\n" + - " // it has no name for. NULL on every field that is not a map.\n" + - " const TableTypeInfo * entry;\n" + - " int32_t ( * map_count )( const void * slot );\n" + - " const void * ( * map_at )( const void * slot, int32_t index );\n" + - " // place one entry BY KEY and hand back the entry, at its defaults: a\n" + - " // string key comes in as the bytes and the length, an integer key as\n" + - " // the value, and NULL is NOT INSERTED — a key past the bound, or an\n" + - " // arena that could not carve another segment.\n" + - " void * ( * map_insert )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value );" + // the PLACE column (docs/SPEC-TABLES.md §8.1, §16), emitted only into a + // unit that declares a map or an unbounded array: the one resolver the ONE + // text walk cannot spell for itself, because TableMap and + // TableList are types it has no name for. The SHAPE of either is read + // from the array columns like every other array's, with array_bound = 0 + // the one tell that the offset names a reference and not the first element. + placeMember := "" + refuseReason := "" + if anyExtent { + placeMember = "\n // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and\n" + + " // hand it back at its defaults. A MAP places BY KEY, a string key comes\n" + + " // in as the bytes and the length, an integer key as the value, and NULL\n" + + " // is NOT INSERTED: a key past the bound, or an arena that could not carve\n" + + " // another segment. A LIST ignores the key and APPENDS, NULL at the arena\n" + + " // or the int32 cap. NULL on every field that is neither.\n" + + " void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value );" + refuseReason = ` + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +};` } pointerFieldMember, pointerTypeMember, pointerForward := "", "", "" if anyVariable { @@ -479,7 +497,7 @@ struct TableReport // must not look at then (docs/SPEC-TABLES.md §3.3). TableMessageReason reason = newer_form; }; - +` + refuseReason + ` // ---- reflection (tables only, docs/SPEC-TABLES.md) ---- // // Static field descriptors for every type in the table closure: name, wire @@ -580,7 +598,7 @@ struct TableWideRange // descriptor stays CONSTANT-INITIALISED (a captureless lambda converts to // a function pointer at compile time; the arms themselves are a static // inside it). NULL for every other kind. - const TableUnionInfo * (*arms)();` + mapFieldMember + ` + const TableUnionInfo * (*arms)();` + placeMember + ` const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded }; @@ -971,6 +989,8 @@ func Generate(u *ir.Unit) (map[string][]byte, error) { anyVariable := len(variable) > 0 anyKeyed := unitHasKeyedArray(u, closure) anyMap := unitHasMap(u, closure) + anyList := unitHasList(u, closure) + anyExtent := anyMap || anyList blocks := ir.Blocks(u) // The BLOCK FORM (docs/SPEC-TABLES.md §19) is emitted ON THE SIDE, into @@ -987,7 +1007,7 @@ func Generate(u *ir.Unit) (map[string][]byte, error) { slots[id] = uint64(i + 1) } for _, f := range u.Files { - g := &tableGen{unit: u, file: f, anyVariable: anyVariable, anyKeyed: anyKeyed, anyMap: anyMap, blocks: blocks, variable: variable, targets: targets, + g := &tableGen{unit: u, file: f, anyVariable: anyVariable, anyKeyed: anyKeyed, anyMap: anyMap, anyList: anyList, anyExtent: anyExtent, blocks: blocks, variable: variable, targets: targets, includes: map[string]bool{}, nativeIncludes: map[string]bool{}, slots: slots} var members []*ir.Struct members = append(members, orderTables(f.Tables)...) @@ -1047,6 +1067,7 @@ func Generate(u *ir.Unit) (map[string][]byte, error) { } g.pf("\n") g.emitCookLayoutAsserts(members) + g.emitListAlignAsserts(members) g.pf("// ---- reflection descriptors (tables only, docs/SPEC-TABLES.md) ----\n\n") for _, st := range members { g.pf("inline const TableTypeInfo * %sTableType();\n", st.Name) @@ -1097,11 +1118,13 @@ func Generate(u *ir.Unit) (map[string][]byte, error) { // The relocatability and standard-layout asserts read the COMPILER // INTRINSICS, so — 124 headers on its own — is not here. h.WriteString("#pragma once\n\n#include \n#include // the prefill's scalar-array fills\n#include // offsetof, for the reflection descriptors\n") - if anyKeyed { - // ENUM-KEYED arrays only: indexing one by None is a program error in - // EVERY configuration, and the accessor is where a runtime key can - // first be caught (docs/SPEC-TABLES.md §2.4). It is the runtime's - // only refusal, so it is the only reason these two hooks are here. + if anyKeyed || anyList { + // ENUM-KEYED arrays and UNBOUNDED arrays only: indexing a keyed array + // by None, or a list past its count, is a program error in EVERY + // configuration, and the accessor is where a runtime key or index can + // first be caught (docs/SPEC-TABLES.md §2.4, §2.9). Those are the + // runtime's only refusals, so they are the only reason these two + // hooks are here. h.WriteString(tableHooks) } if anyVariable { @@ -1137,19 +1160,32 @@ func Generate(u *ir.Unit) (map[string][]byte, error) { fmt.Fprintf(&h, "#include \"%s\"\n", n) } h.WriteString("\n") - h.WriteString(tablePrimitives(u.Package, anyVariable, anyKeyed, anyMap, ir.TableWireIdCapacity(u), u)) + h.WriteString(tablePrimitives(u.Package, anyVariable, anyKeyed, anyExtent, ir.TableWireIdCapacity(u), u)) if anyVariable { h.WriteString("\n") - h.WriteString(tableArenaRuntime(u.Package, anyMap)) + h.WriteString(tableArenaRuntime(u.Package, anyExtent)) + } + if anyExtent { + // the NODE EXTENT runtime (docs/SPEC-TABLES.md §2.8, §2.9): what a + // map and an unbounded array share once the key and the sort are + // taken out. Either makes its holder variable-length, so it always + // follows the arena runtime it is spelled in terms of. + h.WriteString("\n") + h.WriteString(tableExtentRuntime(u.Package)) } if anyMap { // the MAP runtime (docs/SPEC-TABLES.md §2.8): the storage type, the // order, the builder's head and segments, and the optional index. - // A map makes its holder variable-length, so it always follows the - // arena runtime it is spelled in terms of. h.WriteString("\n") h.WriteString(tableMapRuntime(u.Package)) } + if anyList { + // the LIST runtime (docs/SPEC-TABLES.md §2.9): the storage type and + // its const surface, the builder's head and segments, the index-order + // cursor and the load side's fill. + h.WriteString("\n") + h.WriteString(tableListRuntime(u.Package)) + } // the COOKED FORM's read side (docs/SPEC-TABLES.md §7) and the BUILD VERSION // it matches against, in EVERY unit that declares a table: every table // cooks and any table may be a cook's root, so there is no unit with @@ -1188,9 +1224,10 @@ func Generate(u *ir.Unit) (map[string][]byte, error) { c.WriteString("#include // the text form: number formatting\n") c.WriteString("#include // the text form: exact number conversion\n") c.WriteString("#include // the text form: the runtime's decimal point\n\n") - c.WriteString(tableJsonWalk(u.Package, anyVariable, anyMap)) + c.WriteString(tableJsonWalk(u.Package, anyVariable, anyMap, anyList)) fmt.Fprintf(&c, "\nnamespace %s {\n\n", u.Package) - cg := &tableGen{unit: u, file: f, anyVariable: anyVariable, anyMap: anyMap, blocks: blocks, variable: variable, targets: targets, + cg := &tableGen{unit: u, file: f, anyVariable: anyVariable, anyMap: anyMap, anyList: anyList, anyExtent: anyExtent, blocks: blocks, variable: variable, targets: targets, + includes: map[string]bool{}, nativeIncludes: map[string]bool{}} for _, st := range members { cg.emitJsonDefinitions(st) diff --git a/internal/codegen/cpptable/extent.go b/internal/codegen/cpptable/extent.go new file mode 100644 index 000000000..95098ea97 --- /dev/null +++ b/internal/codegen/cpptable/extent.go @@ -0,0 +1,847 @@ +// The NODE EXTENT's shared runtime (docs/SPEC-TABLES.md §2.8, §2.9, §6.3): +// what a map and an unbounded array have in common once the key and the sort +// are taken out of the map. Both put their arrays in the holder's node extent +// after the record's own storage, both are carved from that extent as the +// node's body decodes, and both make LoadMeasure walk the wire's framing for +// every N at every depth. So the cursor, the framing walkers over the by-value +// edges that hold them, and the one test an unreached slot takes live here, +// emitted into a unit that declares either construct and into no other. +package cpptable + +import ( + "fmt" + "slices" + "strings" + + "github.com/mas-bandwidth/schema/v2/ir" +) + +// tableExtentRuntime is the extent half of the variable-length runtime. It +// follows the arena runtime it is spelled in terms of and precedes the map +// and list runtimes that are spelled in terms of it. +func tableExtentRuntime(pkg string) string { + guard := strings.ToUpper(pkg) + "_SCHEMA_TABLE_EXTENT" + return `#ifndef ` + guard + ` +#define ` + guard + ` + +namespace ` + pkg + ` { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace ` + pkg + ` + +#endif // ` + guard + ` +` +} + +// ---- the NODE EXTENT's emitters (docs/SPEC-TABLES.md §2.8, §2.9, §6.3) ---- +// +// A map's entries and a list's elements are BY-VALUE RECORDS INSIDE THE +// HOLDER'S NODE EXTENT, laid after the record's own storage: count x sizeof +// at the element's alignment, zero slack, one array per container reachable +// BY VALUE from the record, which includes one inside a nested table and one +// inside an element or an entry, in depth-first field order. The placement +// is PRE-ORDER and it interleaves lists and maps on ONE rule: a container's +// whole array first, then, element by element in the container's own order, +// the arrays of any list or map that element holds by value. +// +// Two emitters walk that layout and they are ONE walk: the measure advances a +// running offset, and the pack advances the same one and copies. Nothing +// passes between them, which is what makes `used == total` a real check. + +// hasExtent reports a member with any list or map reachable by value: the +// members that carry an extent, and the ones whose extent walks are emitted. +func (g *tableGen) hasExtent(st *ir.Struct) bool { + for _, f := range st.Fields { + if f.IsMap() || f.IsList() { + return true + } + // A CONTAINER REACHABLE BY VALUE IS THIS RECORD'S EXTENT, whichever + // by-value edge reaches it: a nested table, an array of them, an + // enum-keyed array of them, or a union arm. + switch g.edgeOf(f) { + case edgeNested: + if ref, ok := f.Type.Ref.(*ir.Struct); ok && g.hasExtent(ref) { + return true + } + case edgeArm: + for _, v := range f.Type.Ref.(*ir.Union).Variants { + if ref := memberOf(g.unit, v.Type); ref != nil && g.hasExtent(ref) { + return true + } + } + } + // a nested table this walk does not call an edge can still hold a + // container, because a container makes its holder VARIABLE and every + // variable nesting is an edge, so there is nothing else to look at + } + return false +} + +// memberOf resolves one closure member by name. +func memberOf(u *ir.Unit, name string) *ir.Struct { + if st := u.Tables[name]; st != nil { + return st + } + return u.Structs[name] +} + +// alignOfEntry is the C ABI alignment of one generated entry record, the +// alignment its array is laid at, and the same model §20.3 commits the +// compiler to for every record in the closure. +func alignOfEntry(u *ir.Unit, entry *ir.Struct) int64 { + if ml := ir.RecordLayout(u, entry); ml != nil && ml.Align > 0 { + return ml.Align + } + return 8 +} + +// extentVisitor is what one extent emitter does at the two containers and at +// a by-value nesting that holds one. +type extentVisitor struct { + mapField func(f *ir.Field, expr, ind string) + listField func(f *ir.Field, expr, ind string) + descend func(table, expr, ind string) +} + +// emitExtentWalk is the ONE walk every extent emitter takes: the record's +// fields in declaration order, every container at its own position, +// descending each by-value edge in place. A pointer is NOT an edge here, a +// pointee is its own node with its own extent, and neither is a union arm +// that is one, for the same reason. +func (g *tableGen) emitExtentWalk(st *ir.Struct, subject string, v extentVisitor) { + ev := edgeVisitor{read: subject} + for _, f := range st.Fields { + if f.IsMap() { + v.mapField(f, subject+"."+f.Name, " ") + continue + } + if f.IsList() { + v.listField(f, subject+"."+f.Name, " ") + continue + } + switch g.edgeOf(f) { + case edgeNested: + ref, _ := f.Type.Ref.(*ir.Struct) + if ref == nil || !g.hasExtent(ref) { + continue + } + g.emitVariableByValueWalk(f, ev, func(expr edgeExpr) { v.descend(f.Type.Name, expr.Src, " ") }) + g.emitUnreachedExtentRefusal(f, ref, subject) + case edgeArm: + un := f.Type.Ref.(*ir.Union) + any := false + for _, arm := range un.Variants { + if ref := memberOf(g.unit, arm.Type); ref != nil && g.hasExtent(ref) { + any = true + } + } + if !any { + continue + } + // only the ARMS that hold a container are descended: a pointer arm + // and a byte buffer arm reach nodes, not this node's extent + armed := edgeVisitor{read: subject, + pointer: func(*ir.Field, edgeExpr) {}, + blob: func(*ir.Field, edgeExpr) {}, + descend: func(table string, expr edgeExpr, indent string) { + if ref := memberOf(g.unit, table); ref != nil && g.hasExtent(ref) { + v.descend(table, expr.Src, indent) + } + }, + } + g.emitVariableUnionWalk(f, armed) + } + } +} + +// emitExtent emits `ExtentAt`: the running offset every array reachable by +// value from one record takes, in the order the pack lays them, and +// `Extent`, the whole extent from a fresh offset. +func (g *tableGen) emitExtent(st *ir.Struct) { + g.pf("// %sExtentAt: the node extent %s's lists and maps take, PRE-ORDER, advancing\n", st.Name, st.Name) + g.pf("// the running offset exactly as %sExtentPack advances it (§2.8, §2.9).\n", st.Name) + g.pf("template \ninline bool %sExtentAt( const Ctx & ctx, const %s & value, int64_t & at )\n{\n", st.Name, st.Name) + if !g.hasExtent(st) { + g.pf(" (void) ctx; (void) value; (void) at; // no list or map below this record\n") + g.pf(" return true;\n}\n\n") + return + } + g.emitExtentWalk(st, "value", extentVisitor{ + mapField: func(f *ir.Field, expr, ind string) { + entry := mapEntryOf(f) + g.pf("%s{\n", ind) + g.pf("%s TableMapCursor<%s> cursor = TableMapOrder( ctx, %s );\n", ind, entry.Name, expr) + g.pf("%s if ( !cursor.ok ) { return false; }\n", ind) + g.pf("%s at = ( at + %d ) & ~(int64_t) %d; // at alignof( %s )\n", ind, alignOfEntry(g.unit, entry)-1, alignOfEntry(g.unit, entry)-1, entry.Name) + g.pf("%s at += (int64_t) cursor.count * (int64_t) sizeof( %s ); // the whole array FIRST\n", ind, entry.Name) + if g.isVar(entry.Name) { + g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order\n%s {\n", ind, ind) + g.pf("%s if ( !%sExtentAt( ctx, *cursor[i], at ) ) { TableMapRelease( cursor ); return false; }\n", ind, entry.Name) + g.pf("%s }\n", ind) + } + g.pf("%s TableMapRelease( cursor );\n", ind) + g.pf("%s}\n", ind) + }, + listField: func(f *ir.Field, expr, ind string) { + elem := g.listElementType(f) + g.pf("%s{\n", ind) + g.pf("%s TableListCursor<%s> cursor = TableListElements( ctx, %s );\n", ind, elem, expr) + g.pf("%s if ( !cursor.ok ) { return false; }\n", ind) + g.pf("%s at = ( at + (int64_t) alignof( %s ) - 1 ) & ~( (int64_t) alignof( %s ) - 1 );\n", ind, elem, elem) + g.pf("%s at += (int64_t) cursor.count * (int64_t) sizeof( %s ); // the whole array FIRST\n", ind, elem) + if ref := listElementStruct(f); ref != nil && g.hasExtent(ref) { + g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ ) // then, element by element in index order\n%s {\n", ind, ind) + g.pf("%s if ( !%sExtentAt( ctx, cursor[i], at ) ) { return false; }\n", ind, ref.Name) + g.pf("%s }\n", ind) + } + g.pf("%s}\n", ind) + }, + descend: func(table, expr, ind string) { + g.pf("%sif ( !%sExtentAt( ctx, %s, at ) ) { return false; }\n", ind, table, expr) + }, + }) + g.pf(" return true;\n}\n\n") + g.pf("// the whole extent of one node, from a fresh offset: what a pack reserves\n") + g.pf("// for it beside the record's own storage.\n") + g.pf("template \ninline int64_t %sExtent( const Ctx & ctx, const %s & value )\n{\n", st.Name, st.Name) + g.pf(" int64_t at = 0;\n") + g.pf(" if ( !%sExtentAt( ctx, value, at ) ) { return -1; }\n", st.Name) + g.pf(" return at;\n}\n\n") +} + +// emitExtentPack emits `ExtentPack`: the same walk, copying each map's +// entries in key order and each list's elements in index order into the +// node's extent and pointing the record's slot at them. +func (g *tableGen) emitExtentPack(st *ir.Struct) { + g.pf("// %sExtentPack: carve %s's arrays out of the node's extent and copy the\n", st.Name, st.Name) + g.pf("// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER,\n") + g.pf("// advancing the same running offset %sExtentAt advances (§2.8, §2.9).\n", st.Name) + g.pf("template \ninline bool %sExtentPack( const Ctx & ctx, const %s & src, %s & dst, uint8_t * extent, int64_t & at, int64_t capacity )\n{\n", st.Name, st.Name, st.Name) + if !g.hasExtent(st) { + g.pf(" (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no list or map below this record\n") + g.pf(" return true;\n}\n\n") + return + } + dstOf := func(expr string) string { return "dst" + expr[len("src"):] } + g.emitExtentWalk(st, "src", extentVisitor{ + mapField: func(f *ir.Field, expr, ind string) { + entry := mapEntryOf(f) + slot := dstOf(expr) + g.pf("%s{\n", ind) + g.pf("%s TableMapCursor<%s> cursor = TableMapOrder( ctx, %s );\n", ind, entry.Name, expr) + g.pf("%s if ( !cursor.ok ) { return false; }\n", ind) + g.pf("%s at = ( at + %d ) & ~(int64_t) %d;\n", ind, alignOfEntry(g.unit, entry)-1, alignOfEntry(g.unit, entry)-1) + g.pf("%s const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( %s );\n", ind, entry.Name) + g.pf("%s if ( at + bytes > capacity ) { TableMapRelease( cursor ); return false; }\n", ind) + g.pf("%s %s * placed = (%s *) ( extent + at );\n", ind, entry.Name, entry.Name) + g.pf("%s at += bytes;\n", ind) + g.pf("%s %s.count = cursor.count;\n", ind, slot) + g.pf("%s %s.padding = 0;\n", ind, slot) + g.pf("%s %s.entries.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &%s.entries ) : 0;\n", ind, slot, slot) + g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ )\n%s {\n", ind, ind) + g.pf("%s memcpy( (void *) ( placed + i ), (const void *) cursor[i], sizeof( %s ) ); // trivially copyable, by construction\n", ind, entry.Name) + g.pf("%s }\n", ind) + if g.isVar(entry.Name) { + g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ )\n%s {\n", ind, ind) + g.pf("%s if ( !%sExtentPack( ctx, *cursor[i], placed[i], extent, at, capacity ) ) { TableMapRelease( cursor ); return false; }\n", ind, entry.Name) + g.pf("%s }\n", ind) + } + g.pf("%s TableMapRelease( cursor );\n", ind) + g.pf("%s}\n", ind) + }, + listField: func(f *ir.Field, expr, ind string) { + elem := g.listElementType(f) + slot := dstOf(expr) + g.pf("%s{\n", ind) + g.pf("%s TableListCursor<%s> cursor = TableListElements( ctx, %s );\n", ind, elem, expr) + g.pf("%s if ( !cursor.ok ) { return false; }\n", ind) + g.pf("%s at = ( at + (int64_t) alignof( %s ) - 1 ) & ~( (int64_t) alignof( %s ) - 1 );\n", ind, elem, elem) + g.pf("%s const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( %s );\n", ind, elem) + g.pf("%s if ( at + bytes > capacity ) { return false; }\n", ind) + g.pf("%s %s * placed = (%s *) ( extent + at );\n", ind, elem, elem) + g.pf("%s at += bytes;\n", ind) + g.pf("%s %s.count = cursor.count;\n", ind, slot) + g.pf("%s %s.padding = 0;\n", ind, slot) + g.pf("%s %s.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &%s.elements ) : 0;\n", ind, slot, slot) + g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only\n%s {\n", ind, ind) + g.pf("%s memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( %s ) ); // trivially copyable, by construction\n", ind, elem) + g.pf("%s }\n", ind) + if ref := listElementStruct(f); ref != nil && g.hasExtent(ref) { + g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ )\n%s {\n", ind, ind) + g.pf("%s if ( !%sExtentPack( ctx, cursor[i], placed[i], extent, at, capacity ) ) { return false; }\n", ind, ref.Name) + g.pf("%s }\n", ind) + } + g.pf("%s}\n", ind) + }, + descend: func(table, expr, ind string) { + g.pf("%sif ( !%sExtentPack( ctx, %s, %s, extent, at, capacity ) ) { return false; }\n", ind, table, expr, dstOf(expr)) + }, + }) + g.pf(" return true;\n}\n\n") +} + +// emitExtentWalkSurface emits the framing walk and the two extent walks for +// every variable member of a unit that declares a list or a map. They are +// emitted for EVERY such member, because a walk that descends a by-value +// nesting has to be able to name the nested one's. +func (g *tableGen) emitExtentWalkSurface(members []*ir.Struct) { + if !g.anyExtent { + return + } + for _, st := range g.varMembers(members) { + g.emitWireExtent(st) + g.emitExtent(st) + g.emitExtentPack(st) + } +} + +// emitNodeBytes emits the bytes ONE NODE takes in a packed region: the +// record's own storage rounded to the arena's alignment, plus the extent its +// lists and maps take (docs/SPEC-TABLES.md §2.8, §2.9, §6.3), the sum rounded +// again so the next node starts aligned. A unit with neither construct emits +// exactly the term it always emitted. +func (g *tableGen) emitNodeBytes(table, expr, ind, onBad string, plain func(term string), extent func(term string)) { + target := memberOf(g.unit, table) + if !g.anyExtent || target == nil || !g.hasExtent(target) { + // a node with no container below it takes exactly the term it always + // took, so a unit without one emits what it emitted before either + // construct existed + plain(fmt.Sprintf("TableAlignUp64( (int64_t) sizeof( %s ) )", table)) + return + } + g.pf("%sint64_t node_extent = %sExtent( ctx, %s );\n", ind, table, expr) + g.pf("%sif ( node_extent < 0 ) { %s }\n", ind, onBad) + extent(fmt.Sprintf("TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + node_extent )", table)) +} + +// emitWireExtent emits `WireExtent`: the region bytes one record's lists +// and maps command, read from the wire FRAMING alone at every depth +// (docs/SPEC-TABLES.md §2.8, §2.9, §6.5). False is the refusal, carrying its +// reason, and it is what makes LoadMeasure answer -1. +func (g *tableGen) emitWireExtent(st *ir.Struct) { + g.pf("// %sWireExtent: the extent %s's lists and maps command, from the FRAMING alone.\n", st.Name, st.Name) + g.pf("// It reads no field value, so a caller can refuse a number it did not\n") + g.pf("// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5).\n") + g.pf("inline bool %sWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason )\n{\n", st.Name) + if !g.hasExtent(st) { + g.pf(" (void) body; (void) length; (void) at; (void) ids; (void) reason; // no list or map below this record\n") + g.pf(" return true;\n}\n\n") + return + } + g.pf(" TableReport scratch; // the scan's framing damage is the LOAD's to report\n") + g.pf(" TableReader r( body, length, &scratch, ids );\n") + g.pf(" for ( ;; )\n {\n") + g.pf(" uint64_t field_ref = 0;\n") + g.pf(" if ( !r.getleb( field_ref ) ) { return true; }\n") + g.pf(" if ( field_ref == 0 ) { return true; }\n") + g.pf(" if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; }\n") + g.pf(" const uint64_t field_id = ids->at( field_ref );\n") + g.pf(" if ( !r.has( 1 ) ) { return true; }\n") + g.pf(" uint8_t field_kind = r.get8();\n") + g.emitWireExtentCases(st) + g.pf(" if ( !r.skip( field_kind ) ) { return true; }\n") + g.pf(" }\n}\n\n") +} + +// emitWireExtentCases emits one arm per list field, one per map field and one +// per by-value nesting that holds either, in DECLARATION ORDER, so the framing +// scan advances the running offset in the same order the pack and the load +// carve it. +func (g *tableGen) emitWireExtentCases(st *ir.Struct) { + for _, f := range st.Fields { + if f.IsMap() { + entry := mapEntryOf(f) + inner := "NULL" + if g.hasExtent(entry) { + inner = "&" + entry.Name + "WireExtent" + } + g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s\n {\n", ir.TableFieldWireId(f), tkArray, f.Name) + g.pf(" uint64_t map_len = 0;\n") + g.pf(" if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; }\n") + g.pf(" const uint8_t * map_body = r.buffer + r.offset;\n") + g.pf(" r.offset += (int64_t) map_len;\n") + g.pf(" if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( %s ), (int64_t) alignof( %s ), %s, ids, reason ) ) { return false; }\n", + entry.Name, entry.Name, inner) + g.pf(" continue;\n }\n") + continue + } + if f.IsList() { + elem := g.listElementType(f) + inner := "NULL" + if ref := listElementStruct(f); ref != nil && g.hasExtent(ref) { + inner = "&" + ref.Name + "WireExtent" + } + g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s: an unbounded array\n {\n", ir.TableFieldWireId(f), tkArray, f.Name) + g.pf(" uint64_t list_len = 0;\n") + g.pf(" if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; }\n") + g.pf(" const uint8_t * list_body = r.buffer + r.offset;\n") + g.pf(" r.offset += (int64_t) list_len;\n") + g.pf(" if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( %s ), (int64_t) alignof( %s ), %d, %d, %s, ids, reason ) ) { return false; }\n", + elem, elem, listElementWireKind(f), listElementFloor(f), inner) + g.pf(" continue;\n }\n") + continue + } + switch g.edgeOf(f) { + case edgeNested: + ref, _ := f.Type.Ref.(*ir.Struct) + if ref == nil || !g.hasExtent(ref) { + continue + } + // a nested table's arrays are part of THIS node's extent, so its + // own scan runs over the nested body at the running offset + kind, walk := tkTable, "" + switch { + case f.KeyEnum != "": + kind, walk = tkKeyed, "TableWireExtentKeyed" + case f.Array != ir.ArrayNone: + kind, walk = tkArray, "TableWireExtentElements" + } + g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s: a nesting that holds a list or a map\n {\n", ir.TableFieldWireId(f), kind, f.Name) + g.pf(" uint64_t nested_len = 0;\n") + g.pf(" if ( !r.getleb( nested_len ) || !r.room( nested_len ) ) { return true; }\n") + g.pf(" const uint8_t * nested_body = r.buffer + r.offset;\n") + g.pf(" r.offset += (int64_t) nested_len;\n") + if walk == "" { + g.pf(" if ( !%sWireExtent( nested_body, (int64_t) nested_len, at, ids, reason ) ) { return false; }\n", f.Type.Name) + } else { + g.pf(" if ( !%s( nested_body, (int64_t) nested_len, at, &%sWireExtent, ids, reason ) ) { return false; }\n", walk, f.Type.Name) + } + g.pf(" continue;\n }\n") + case edgeArm: + un := f.Type.Ref.(*ir.Union) + any := false + for _, v := range un.Variants { + if ref := memberOf(g.unit, v.Type); ref != nil && g.hasExtent(ref) { + any = true + } + } + if !any { + continue + } + g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s: a union arm that holds a list or a map\n {\n", ir.TableFieldWireId(f), tkUnion, f.Name) + g.pf(" uint64_t arm_ref = 0;\n") + g.pf(" if ( !r.getleb( arm_ref ) ) { return true; }\n") + g.pf(" if ( arm_ref == 0 ) { continue; } // None: the reference is the whole payload\n") + g.pf(" if ( arm_ref > (uint64_t) ids->count ) { return true; }\n") + g.pf(" const uint64_t arm_id = ids->at( arm_ref );\n") + g.pf(" if ( !r.has( 1 ) ) { return true; }\n") + g.pf(" r.offset += 1; // the arm's kind byte\n") + g.pf(" uint64_t arm_len = 0;\n") + g.pf(" if ( !r.getleb( arm_len ) || !r.room( arm_len ) ) { return true; }\n") + g.pf(" const uint8_t * arm_body = r.buffer + r.offset;\n") + g.pf(" r.offset += (int64_t) arm_len;\n") + g.pf(" switch ( arm_id )\n {\n") + for _, v := range un.Variants { + ref := memberOf(g.unit, v.Type) + if ref == nil || !g.hasExtent(ref) { + continue + } + g.pf(" case 0x%016xull: if ( !%sWireExtent( arm_body, (int64_t) arm_len, at, ids, reason ) ) { return false; } break; // %s\n", + ir.TableWireId(v.Name), v.Type, v.Name) + } + g.pf(" default: break; // an arm this reader cannot name reads None\n") + g.pf(" }\n") + g.pf(" continue;\n }\n") + } + } +} + +// emitRootDataBytes emits a load's DATA term for the root itself: its record, +// plus the extent its own lists and maps take, read from the wire framing +// (§2.8, §2.9, §6.5). `reason` is the refusal's carrier, declared by the +// caller where a unit has an extent. +func (g *tableGen) emitRootDataBytes(st *ir.Struct, ind, onBad string) { + if !g.anyExtent { + g.pf("%sint64_t data = TableAlignUp64( (int64_t) sizeof( %s ) );\n", ind, st.Name) + return + } + g.pf("%sint64_t root_extent = 0;\n", ind) + g.pf("%sif ( !%sWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { %s }\n", ind, st.Name, onBad) + g.pf("%sint64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + root_extent );\n", ind, st.Name) +} + +// ---- the COOK's write side at the extent (docs/SPEC-TABLES.md §2.8, §2.9, §7.6) ---- +// +// A cook is a region written verbatim, so a cooked map is its SORTED entry +// array and a cooked list its element array in INDEX order, where the cook +// put them: the node's extent, laid after the record's own storage by the +// same PRE-ORDER rule the pack lays it by. A map's Find is then a binary +// search over the mapped bytes and a list's indexing one multiply, in place, +// with nothing to parse. + +// cookExtentSignature is one record's extent writer. +func (g *tableGen) cookExtentSignature(st *ir.Struct) string { + return fmt.Sprintf("template inline bool %sCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const %s & value, TableByteOrder order )", st.Name, st.Name) +} + +// emitCookExtent emits one record's extent writer: every list and map +// reachable by value, PRE-ORDER, each element or entry through its own cook +// writer. +func (g *tableGen) emitCookExtent(st *ir.Struct) { + g.pf("// %sCookExtent: %s's arrays into the node's extent, PRE-ORDER, a map's entries\n", st.Name, st.Name) + g.pf("// in ASCENDING key order and a list's elements in INDEX order, each through its\n") + g.pf("// own cook writer (§2.8, §2.9, §7.6).\n") + g.pf("%s\n{\n", g.cookExtentSignature(st)) + if !g.hasExtent(st) { + g.pf(" (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order;\n") + g.pf(" return true; // no list or map below this record\n}\n\n") + return + } + ml := ir.RecordLayout(g.unit, st) + offsetOf := func(name string) int64 { + for i := range ml.Fields { + if ml.Fields[i].Field.Name == name { + return ml.Fields[i].Offset + } + } + return 0 + } + usesRegion := false + for _, f := range st.Fields { + if f.IsList() && listElementIsPointer(f) { + usesRegion = true + } + } + if !usesRegion { + g.pf(" (void) region; // a table element's and an entry's references resolve through their own bodies\n") + } + for _, f := range st.Fields { + if f.IsMap() { + entry := mapEntryOf(f) + el := ir.RecordLayout(g.unit, entry) + slot := offsetOf(f.Name) + g.pf(" { // %s\n", f.Name) + g.pf(" TableMapCursor<%s> cursor = TableMapOrder( ctx, value.%s );\n", entry.Name, f.Name) + g.pf(" if ( !cursor.ok ) { return false; }\n") + g.pf(" at = ( at + %d ) & ~(int64_t) %d; // at alignof( %s )\n", el.Align-1, el.Align-1, entry.Name) + g.pf(" uint8_t * array = extent + at;\n") + g.pf(" at += (int64_t) cursor.count * %d; // the whole array FIRST\n", el.Size) + g.pf(" // the SIXTEEN BYTES of the slot: the self-relative delta, then the count\n") + g.pf(" table_cook_put( record + %d, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + %d ) ) : 0, 8, order );\n", slot, slot) + g.pf(" table_cook_put( record + %d, (uint64_t) (uint32_t) cursor.count, 4, order );\n", slot+8) + g.pf(" for ( int32_t i = 0; i < cursor.count; i++ )\n {\n") + g.pf(" %s\n", g.cookBodyCall(entry, fmt.Sprintf("array + i * %d", el.Size), "*cursor[i]")) + g.pf(" }\n") + if g.hasExtent(entry) { + g.pf(" for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order\n {\n") + g.pf(" if ( !%sCookExtent( ctx, region, extent, at, array + i * %d, *cursor[i], order ) ) { TableMapRelease( cursor ); return false; }\n", entry.Name, el.Size) + g.pf(" }\n") + } + g.pf(" TableMapRelease( cursor );\n }\n") + continue + } + if f.IsList() { + elem := g.listElementType(f) + size, align := ir.ListElementLayout(g.unit, f) + slot := offsetOf(f.Name) + g.pf(" { // %s: an unbounded array\n", f.Name) + g.pf(" TableListCursor<%s> cursor = TableListElements( ctx, value.%s );\n", elem, f.Name) + g.pf(" if ( !cursor.ok ) { return false; }\n") + g.pf(" at = ( at + %d ) & ~(int64_t) %d; // at alignof( %s )\n", align-1, align-1, elem) + g.pf(" uint8_t * array = extent + at;\n") + g.pf(" at += (int64_t) cursor.count * %d; // the whole array FIRST\n", size) + g.pf(" // the SIXTEEN BYTES of the slot: the self-relative delta, then the count\n") + g.pf(" table_cook_put( record + %d, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + %d ) ) : 0, 8, order );\n", slot, slot) + g.pf(" table_cook_put( record + %d, (uint64_t) (uint32_t) cursor.count, 4, order );\n", slot+8) + g.pf(" for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only\n {\n") + if listElementIsPointer(f) { + // the self-relative delta of §6.3 to the node the numbering reached, + // or a refusal for one it did not, exactly as a pointer field's slot + g.pf(" if ( !table_cook_ref( region, array + i * %d, (const void *) %sAt( ctx, cursor[i] ), order ) ) { return false; }\n", size, f.Type.Name) + } else { + g.emitCookWriteElement(f, fmt.Sprintf("array + i * %d", size), "cursor[i]", " ", "_"+f.Name) + } + g.pf(" }\n") + if ref := listElementStruct(f); ref != nil && g.hasExtent(ref) { + g.pf(" for ( int32_t i = 0; i < cursor.count; i++ ) // then, element by element in index order\n {\n") + g.pf(" if ( !%sCookExtent( ctx, region, extent, at, array + i * %d, cursor[i], order ) ) { return false; }\n", ref.Name, size) + g.pf(" }\n") + } + g.pf(" }\n") + continue + } + if g.edgeOf(f) != edgeNested { + continue + } + ref, _ := f.Type.Ref.(*ir.Struct) + if ref == nil || !g.hasExtent(ref) { + continue + } + nested := offsetOf(f.Name) + if f.Array == ir.ArrayNone { + g.pf(" if ( !%sCookExtent( ctx, region, extent, at, record + %d, value.%s, order ) ) { return false; } // %s\n", ref.Name, nested, f.Name, f.Name) + continue + } + stride := cookElementBytes(g.unit, f) + base := "value." + f.Name + bound := fmt.Sprintf("%d", f.ArrayBound) + if f.KeyEnum != "" && st.IsTable { + base += ".slots" + } + if f.Array == ir.ArrayCounted { + // THE LIVE COUNT, as the extent walk counts it: a slot past the + // count is storage the walk does not reach, and a non-empty + // container in one was already refused there (§7.6) + bound = fmt.Sprintf("( value.%s_count < %d ? value.%s_count : %d )", f.Name, f.ArrayBound, f.Name, f.ArrayBound) + } + g.pf(" for ( int32_t i = 0; i < %s; i++ ) // %s\n {\n", bound, f.Name) + g.pf(" if ( !%sCookExtent( ctx, region, extent, at, record + %d + i * %d, %s[i], order ) ) { return false; }\n", ref.Name, nested, stride, base) + g.pf(" }\n") + } + g.pf(" return true;\n}\n\n") +} + +// emitCookNode emits `CookNode`: one NODE's record and then its own extent. +// A nested record's writer is the body alone, because a nesting's arrays are +// part of the HOLDER's extent and this walk already reached them. +func (g *tableGen) emitCookNode(st *ir.Struct) { + ml := ir.RecordLayout(g.unit, st) + record := cookAlignUp(ml.Size, ir.RegionAlignFloor) + g.pf("// %sCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9).\n", st.Name) + g.pf("template inline bool %sCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const %s & value, TableByteOrder order )\n{\n", st.Name, st.Name) + if g.isVar(st.Name) { + g.pf(" if ( !%sCookBody( ctx, region, at, value, order ) ) { return false; }\n", st.Name) + } else { + g.pf(" %sCookBody( at, value, order );\n", st.Name) + } + g.pf(" int64_t extent_at = 0;\n") + g.pf(" return %sCookExtent( ctx, region, at + %d, extent_at, at, value, order );\n}\n\n", st.Name, record) +} + +// emitCookExtentSurface emits the extent writer and the node writer for every +// closure member of a unit that declares a list or a map. +func (g *tableGen) emitCookExtentSurface(members []*ir.Struct) { + if !g.anyExtent { + return + } + var bodies []*ir.Struct + for _, st := range members { + if ir.RecordLayout(g.unit, st) != nil { + bodies = append(bodies, st) + } + } + for _, st := range bodies { + g.pf("%s;\n", g.cookExtentSignature(st)) + } + g.pf("\n") + for _, st := range bodies { + g.emitCookExtent(st) + } + for _, st := range bodies { + g.emitCookNode(st) + } +} + +// emitCookNodeBytes emits one node's whole span in a cooked region: its +// record at the region's alignment floor, plus the extent its lists and maps +// take (docs/SPEC-TABLES.md §2.8, §2.9, §7.2). +func (g *tableGen) emitCookNodeBytes(st *ir.Struct, ind, expr, onBad string) { + ml := ir.RecordLayout(g.unit, st) + if !g.anyExtent || !g.hasExtent(st) { + g.pf("%ssize = %d; node_align = %d;\n", ind, ml.Size, ml.Align) + return + } + g.pf("%s{\n", ind) + g.pf("%s const int64_t extent = %sExtent( ctx, %s );\n", ind, st.Name, expr) + g.pf("%s if ( extent < 0 ) { %s }\n", ind, onBad) + g.pf("%s size = %d + extent; node_align = %d;\n", ind, cookAlignUp(ml.Size, ir.RegionAlignFloor), ml.Align) + g.pf("%s}\n", ind) +} + +// onlyExtentFields reports a record whose every field is a list or a map, a +// cook body that writes the empty slots and reads nothing off the value, +// because the extent writer fills them. +func onlyExtentFields(st *ir.Struct) bool { + for _, f := range st.Fields { + if !f.IsMap() && !f.IsList() { + return false + } + } + return len(st.Fields) > 0 +} + +// placeColumn is the ONE function column an out-of-line array carries +// (docs/SPEC-TABLES.md §8.1, §16): the resolver the ONE text walk cannot spell +// for itself, because TableMap and TableList are types it has no +// name for. A map places one entry BY KEY and hands it back at its defaults, and a +// list ignores the key and appends. Empty in a unit that declares neither, so +// such a unit's descriptors are what they always were. +func (g *tableGen) placeColumn(f *ir.Field) string { + if !g.anyExtent { + return "" + } + if f.IsList() { + return g.listPlaceThunk(f) + ", " + } + if !f.IsMap() { + return "NULL, " + } + entry := mapEntryOf(f) + n := entry.Name + hold := fmt.Sprintf("TableMap<%s>", n) + var insert string + if mapKeyIsString(f) { + insert = fmt.Sprintf("[]( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * "+ + "{ if ( key == NULL || key_length > k%sKeyBound ) { return NULL; } "+ // KEYS NEVER CLAMP + "%s * placed = TableMapPlace( worker, *(%s *) slot, key ); "+ + "if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }", n, n, hold) + } else { + typ, _ := g.cppFieldType(ir.MapKeyField(f).Type) + insert = fmt.Sprintf("[]( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * "+ + "{ %s * placed = TableMapPlace( worker, *(%s *) slot, (%s) key_value ); "+ + "if ( placed != NULL ) { TableEntrySetKey( *placed, (%s) key_value ); } return (void *) placed; }", n, hold, typ, typ) + } + return insert + ", " +} + +// nodeStorageBody, nodeStorageArg and nodeStorageReader are the EXTRA +// parameters a node's storage takes where a list or a map rides in an extent +// (docs/SPEC-TABLES.md §2.8, §2.9): the record's body and the id table, from +// which the framing scan sums the arrays, and the refusal's reason. A root +// that can name no such record does not take them, so a unit without either +// construct emits the dispatch it always emitted. +func (g *tableGen) nodeStorageBody(anyExtent bool) string { + if anyExtent { + return "const uint8_t * body, " + } + return "" +} + +func (g *tableGen) nodeStorageTail(anyExtent bool) string { + if anyExtent { + return ", const TableIdTable * ids, TableRefuseReason & reason" + } + return "" +} + +func (g *tableGen) nodeStorageArg(root *ir.Struct) string { + if g.rootHasExtent(root) { + return "body, " + } + return "" +} + +func (g *tableGen) nodeStorageArgTail(root *ir.Struct) string { + if g.rootHasExtent(root) { + return ", &ids_table, reason" + } + return "" +} + +func (g *tableGen) nodeStorageReader(root *ir.Struct) string { + if g.rootHasExtent(root) { + return "r.buffer, " + } + return "" +} + +func (g *tableGen) nodeStorageReaderTail(root *ir.Struct) string { + if g.rootHasExtent(root) { + return ", r.ids, reason" + } + return "" +} + +// rootHasExtent reports whether any record one root's numbering can name holds +// a list or a map by value, which is what decides the signature above. +func (g *tableGen) rootHasExtent(root *ir.Struct) bool { + if !g.anyExtent { + return false + } + return slices.ContainsFunc(g.pointerReachable(root), g.hasExtent) +} + +// emitUnreachedExtentRefusal refuses an UNREACHED NON-EMPTY SLOT, the same +// refusal §7.6 gives a pointer in that position (docs/SPEC-TABLES.md §2.8, +// §2.9): a COUNTED array's slots past its live count are storage the walk does +// not reach, so a non-empty list or map in one names elements the region will +// not hold, and the write answers false with nothing partial written. +// +// The test is the extent itself: an empty container takes no bytes and +// advances the running offset by none, so a record whose extent measures ZERO +// is a record whose every by-value list and map is empty. A measure that +// refuses answers non-zero here too, and refusing on it is the same answer one +// level up. +func (g *tableGen) emitUnreachedExtentRefusal(f *ir.Field, ref *ir.Struct, subject string) { + if f.Array != ir.ArrayCounted { + return // every other array shape is reached whole + } + g.pf(" for ( int32_t i = %s.%s_count; i < %d; i++ ) // %s: the slots the walk does not reach (§7.6)\n {\n", + subject, f.Name, f.ArrayBound, f.Name) + g.pf(" if ( !TableExtentUnreachedEmpty( %sExtent( ctx, %s.%s[i] ) ) ) { return false; }\n", ref.Name, subject, f.Name) + g.pf(" }\n") +} diff --git a/internal/codegen/cpptable/json.go b/internal/codegen/cpptable/json.go index ef2dcd035..6cb8afcbf 100644 --- a/internal/codegen/cpptable/json.go +++ b/internal/codegen/cpptable/json.go @@ -26,19 +26,29 @@ import ( // with three stubs no field ever reaches, and a pointered unit answers with the // graph half — the builder's reader, the region's writer and the `&node` map — // which carries its own gate across the pointered units of the corpus. -func tableJsonWalk(pkg string, variable bool, anyMap bool) string { +func tableJsonWalk(pkg string, variable bool, anyMap bool, anyList bool) string { guard := strings.ToUpper(pkg) + "_SCHEMA_TABLE_JSON" adapters := tableJsonFixedAdapters if variable { adapters = tableJsonGraphSource } - // the MAP half, on the pointer half's own terms (docs/SPEC-TABLES.md §2.8): - // a map makes its holder variable-length, so the real half always follows - // the graph half it names, and a map-free unit carries the stub. + // the EXTENT accessors both out-of-line halves read, then the MAP half and + // the LIST half, each on the pointer half's own terms (docs/SPEC-TABLES.md + // §2.8, §2.9): either construct makes its holder variable-length, so a real + // half always follows the graph half it names, and a unit without the + // construct carries the stub. + extentAdapters := "" + if anyMap || anyList { + extentAdapters = tableJsonExtentAdapters + } mapAdapters := tableJsonNoMapAdapters if anyMap { mapAdapters = tableJsonMapAdapters } + listAdapters := tableJsonNoListAdapters + if anyList { + listAdapters = tableJsonListAdapters + } // The include guard is LOAD-BEARING in a .cpp, which is not where a reader // expects to find one — hence the comment riding with it. It is what lets // several same-package .cpp files be concatenated into one translation @@ -56,7 +66,9 @@ func tableJsonWalk(pkg string, variable bool, anyMap bool) string { tableJsonAdapterDeclarations + tableJsonWalkSource + "\n" + adapters + + extentAdapters + "\n" + mapAdapters + + "\n" + listAdapters + "\n} // namespace " + pkg + "\n\n#endif // " + guard + "\n" } @@ -86,21 +98,52 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); + +` + +// tableJsonExtentAdapters is what both out-of-line halves read off a slot +// (docs/SPEC-TABLES.md §7.2, §8.1): the sixteen bytes are an int64 +// self-relative reference to the array and the int32 count, the same two +// facts for a map and a list, so one pair of accessors serves both. Emitted +// only into a unit that declares either. +const tableJsonExtentAdapters = ` +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} ` // tableJsonFixedAdapters answers for a unit that declares no pointer: no field @@ -139,12 +182,15 @@ inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char *, int32_t d // zero-cost property (§2.2) holding for the text form. const tableJsonMapAdapters = `// ---- json map walk: begin ---- -inline bool TableJsonIsMap( const TableFieldInfo * f ) { return f->entry != NULL; } +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} // the entry's two rows: fields[0] IS the key and fields[1] IS the value, which // is what makes a user's own table of pairs the same bytes (§2.8) -inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->entry->fields[0]; } -inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->entry->fields[1]; } +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } @@ -204,16 +250,17 @@ inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const // A region holds them in that order already, so this is the array in place. inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) { - const int32_t count = f->map_count( slot ); + const int32_t count = TableJsonExtentCount( slot ); if ( count == 0 ) { out.raw( "{}", 2 ); return true; } const TableFieldInfo * key = TableJsonMapKeyField( f ); const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); out.put( '{' ); for ( int32_t i = 0; i < count; i++ ) { if ( i > 0 ) { out.put( ',' ); } out.line( depth + 1 ); - const void * entry = f->map_at( slot, i ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); TableJsonWriteMapKey( out, entry, key ); out.raw( ": ", 2 ); if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } @@ -302,15 +349,15 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf } if ( !fits ) { in.report->kind_mismatch++; place = false; } } - const int32_t before = f->map_count( (const void *) slot ); - void * entry = place ? f->map_insert( *graph->worker, slot, token, token_length, key_value ) : NULL; + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; if ( place && entry == NULL ) { // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the // wire's rule, because a clamped key is a merged entry (§2.8). in.report->clamped++; } - else if ( entry != NULL && f->map_count( (const void *) slot ) == before ) + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) { in.report->duplicate++; // last-wins, the object rule inside the map } @@ -385,7 +432,139 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, } ` +// tableJsonListAdapters is the list half (docs/SPEC-TABLES.md §2.9, §16): the +// JSON array a bounded array already takes, with every element the text +// carries read, because there is no bound to drop a tail against. It is +// emitted ONLY into a unit that declares an unbounded array, and it is one +// half, the same bytes in every list-bearing .cpp, which is the generic-walk +// gate's property holding for the construct. +const tableJsonListAdapters = `// ---- json list walk: begin ---- + +// an unbounded array is the out-of-line array that is not a map (§8.1) +inline bool TableJsonIsList( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && !TableJsonIsMap( f ); +} + +// ToJson WRITES THE ELEMENTS IN INDEX ORDER, which is the only order there is, +// so unpack then pack is byte-stable without a rule of its own (§2.9, §17.2). +// A region holds the array in place, so this steps it at the descriptor's pitch. +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + const uint8_t * elements = TableJsonExtentElements( slot ); + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const uint8_t * element = elements + (int64_t) i * f->elem_size; + if ( f->kind == 17 ) + { + // a []*T's elements take the pointer row (§16.7): the pointee's + // object in place, null, or ` + "`&node`" + ` for a shared one + if ( !TableJsonWritePointer( out, element, f, depth + 1 ) ) { return false; } + } + else if ( !TableJsonWriteScalar( out, element, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; +} + +// FromJson READS EVERY ELEMENT THE TEXT CARRIES, appending each through the +// descriptor's place resolver: ` + "`[]`" + ` is an empty list, and null is +// kind_mismatch, the array row's own rule (§16.2). LAST WINS holds for a +// repeated key: the list goes back to EMPTY before this occurrence's elements +// land, the builder's storage being reclaimed at reset (§2.9). +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) slot, 8, 0 ); + TableJsonSetRaw( (uint8_t *) slot + 8, 4, 0 ); + const char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + void * element = f->place( *graph->worker, slot, NULL, 0, 0 ); + if ( element == NULL ) + { + // NOT ADDED: the arena could not carve another segment, or the + // count met the int32 cap. The text cannot be placed whole, and + // the read stops where §16.1's rule stops it. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( f->kind == 17 ) + { + // an element of a []*T (§2.9): null is a null slot, an object is the + // pointee in place or an ` + "`&node`" + ` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) element, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, element, f, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + // the wrong shape for the element kind: the slot keeps its + // defaults and the event counts, the array row's rule (§16.2) + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, element, f, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json list walk: end ---- +` + +// tableJsonNoListAdapters answers for a unit that declares no UNBOUNDED ARRAY: +// no field is one, so the two slot adapters are never reached, and a unit +// with no list carries no list machinery (§2.2, §2.9). +const tableJsonNoListAdapters = `// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} +` + // emitJsonDeclarations puts one closure member's text-form surface in the + // HEADER: three declarations and nothing else. The definitions, and the walker // they call, live in the generated Table.cpp — so a translation unit // that includes the header to use the wire codecs or the descriptors pays @@ -1211,6 +1390,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -2218,6 +2401,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; diff --git a/internal/codegen/cpptable/lists.go b/internal/codegen/cpptable/lists.go new file mode 100644 index 000000000..fc8bf25d8 --- /dev/null +++ b/internal/codegen/cpptable/lists.go @@ -0,0 +1,852 @@ +// Unbounded arrays in the C++ reference (docs/SPEC-TABLES.md §2.9): the +// runtime a `[]T` field's storage, its builder and its const surface are +// spelled in, and the per-field codecs the generated code hangs off it. +// +// An unbounded array is §2.8's map with the KEY and the SORT taken out: a +// counted array whose count the data decides, its elements by-value records +// inside the holder's node extent. So nothing here is a new wire construct, +// and most of what a map needed is not needed: no entry table, no order, no +// key compare, no ascending check and no duplicate rule. What the runtime +// adds is the reference-and-count slot, a builder that appends into segments +// that never move, and a cursor the four writing walks read in INDEX order +// without allocating. +package cpptable + +import ( + "fmt" + "strings" + + "github.com/mas-bandwidth/schema/v2/ir" +) + +// unitHasList reports whether any closure member declares a `[]T`. It gates +// the list runtime: not one symbol of it appears in a list-free unit's +// generated header (docs/SPEC-TABLES.md §2.2, §2.9). +func unitHasList(u *ir.Unit, closure map[string]bool) bool { + for name := range closure { + st := memberOf(u, name) + if st == nil { + continue + } + for _, f := range st.Fields { + if f.IsList() { + return true + } + } + } + return false +} + +// listFieldsOf lists one member's unbounded arrays in declaration order. +func listFieldsOf(st *ir.Struct) []*ir.Field { + var out []*ir.Field + for _, f := range st.Fields { + if f.IsList() { + out = append(out, f) + } + } + return out +} + +// listVerb spells one of a list field's claimed surface names on its holder: +// `` followed by the verb (docs/SPEC-TABLES.md §2.9, §11). +func listVerb(owner string, f *ir.Field, verb string) string { + return owner + ir.GoExportName(f.Name) + verb +} + +// listElementIsPointer reports a `[]*T`: the elements are pointer SLOTS, so +// the builder's Add hands back the slot at null and the const form answers the +// resolved `const T *` (docs/SPEC-TABLES.md §2.9). +func listElementIsPointer(f *ir.Field) bool { return f.Type.Pointer } + +// listTypeArg is the type argument the storage names: the element type for a +// `[]T`, and `T *` for a `[]*T`, which selects the pointer specialization. +func (g *tableGen) listTypeArg(f *ir.Field) string { + if listElementIsPointer(f) { + return f.Type.Name + " *" + } + typ, _ := g.cppFieldType(f.Type) + return typ +} + +// listStorageType is the C++ spelling of a list field's storage. +func (g *tableGen) listStorageType(f *ir.Field) string { + return fmt.Sprintf("TableList<%s>", g.listTypeArg(f)) +} + +// listElementType is the C++ type ONE ELEMENT is stored as: a TableRef for a +// `[]*T`, the element's own type otherwise. +func (g *tableGen) listElementType(f *ir.Field) string { + if listElementIsPointer(f) { + return "TableRef" + } + typ, _ := g.cppFieldType(f.Type) + return typ +} + +// listElementStruct is the element's table when the element is one, else nil. +func listElementStruct(f *ir.Field) *ir.Struct { + if listElementIsPointer(f) || f.Type.Kind != ir.TNamed { + return nil + } + ref, _ := f.Type.Ref.(*ir.Struct) + return ref +} + +// listElementWireKind is the element kind the list's kind 14 body carries: +// kind 17 for a pointer element, the element's own kind otherwise (§2.9, §3). +func listElementWireKind(f *ir.Field) int { + if listElementIsPointer(f) { + return tkNodeIndex + } + return ir.TableWireElemKind(f) +} + +// listElementFloor is the smallest wire footprint ONE element commands +// (docs/SPEC-TABLES.md §4.2, §6.5): a scalar its own width, a reference-shaped +// element (a pointer's node index, an enum's variant reference, a union's arm +// reference) one byte, a table element its own `L` and its terminator. It is +// what bounds the N a list's `L` can carry, and therefore what a LoadMeasure +// may be asked for. +func listElementFloor(f *ir.Field) int { + if listElementIsPointer(f) { + return 1 + } + switch kind := ir.TableWireElemKind(f); kind { + case tkTable: + return 2 + case tkEnum, tkUnion: + return 1 + default: + return tableKindWidth(kind) + } +} + +// ---- the runtime (docs/SPEC-TABLES.md §2.9) ---- + +// tableListRuntime is the list half of the variable-length runtime: the +// storage type and its const surface, the builder's head and segments, the +// index-order cursor the four writing walks read, and the load side's fill. +// It is emitted only into a unit that declares an unbounded array. +func tableListRuntime(pkg string) string { + guard := strings.ToUpper(pkg) + "_SCHEMA_TABLE_LIST" + return `#ifndef ` + guard + ` +#define ` + guard + ` + +namespace ` + pkg + ` { + +// ---- an UNBOUNDED ARRAY: a counted array whose count the data decides (§2.9) ---- +// +// On the wire, in a region and in a cook a list is the kind 14 body a [..N]T +// writes, its elements by-value records inside the holder's node extent. What +// this adds is the slot, a builder that appends into segments that never +// move, and a const surface that indexes and iterates in place. There is no +// sort, no key and no lookup: the order is INSERTION order, and it is +// identity the way position is identity in a fixed array. + +// elements carved from ONE call to the allocator pair. A new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableListSegmentElements = 32; + +// THE ELEMENT STORAGE: T itself, and a TableRef slot for a []*T, whose +// elements are references exactly as a pointer field's slot is (§2.1) +template struct TableListStorage { typedef T Element; }; +template struct TableListStorage { typedef TableRef Element; }; + +// WHAT THE CONST FORM ANSWERS: the element by reference, and on a []*T the +// RESOLVED pointer, one add on the self-relative delta, NULL for a null slot, +// exactly as At answers it (§6.2, §6.3) +template struct TableListConst +{ + typedef const T & Result; + static Result At( const T * element ) { return *element; } +}; +template struct TableListConst +{ + typedef const T * Result; + static Result At( const TableRef * element ) + { + return element->value != 0 ? (const T *) ( (const uint8_t *) element + element->value ) : NULL; + } +}; + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.9, §7.2) ---- +// +// An int64 self-relative reference to the element array and an int32 count, +// then padding to eight. The reference is a TableRef like a pointer's: in the +// arena it names the builder's HEAD, in a region it is the delta from the slot +// to the first element, and 0 is the empty list in both. It is the map's slot +// exactly, because it is the same two facts. +template struct TableList +{ + typedef typename TableListStorage::Element Element; + + TableRef elements; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Element * Elements() const + { + return elements.value != 0 ? (const Element *) ( (const uint8_t *) &elements + elements.value ) : NULL; + } + int32_t size() const { return count; } + + // INDEXING IS BOUNDS-CHECKED IN EVERY BUILD (§2.4, §2.9): the extent is a + // number that CAME FROM A FILE, so an index past it is not a mistake a + // release build gets to make cheaply. There is no undefined-behavior path + // here in any configuration. The assert carries the message where a + // debugger can read it and NDEBUG removes that. The fatal is what stands + // after it. Both go through the hooks: define schema_assert and + // schema_fatal and this refusal lands in your own handler. + void RefuseIndex( int32_t index ) const + { + if ( (uint32_t) index >= (uint32_t) count ) + { + schema_assert( false && "an unbounded array is indexed inside its count, which came from a file" ); + schema_fatal(); + } + } + typename TableListConst::Result operator[]( int32_t index ) const + { + RefuseIndex( index ); + return TableListConst::At( Elements() + index ); + } + + // ---- iteration: INDEX order, the element and no key ---- + // + // It carries no iterator_traits, for the reason TableKeyed's does not + // (§13.9). + struct ConstIterator + { + const Element * at; + typename TableListConst::Result operator*() const { return TableListConst::At( at ); } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Elements() }; } + ConstIterator end() const { return ConstIterator{ Elements() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.9, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first element is added. Each +// segment is a fixed number of elements carved from one call to the allocator +// pair. An element's address is stable for the arena's life, so a T * handed +// back by Add stays valid while other elements arrive. +struct TableListHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an Add appends into + int32_t live; + int32_t dead; +}; + +template struct TableListSegment +{ + TableRef next; + int32_t used; // elements carved from this segment + int32_t padding; + uint32_t dead[ ( kTableListSegmentElements + 31 ) / 32 ]; // Erase marks one bit, never the element + Element elements[ kTableListSegmentElements ]; +}; + +inline bool TableListSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// the head, allocated when the first element is added +template +inline TableListHead * TableListReach( TableWorker & worker, TableList & list ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( list.elements.value != 0 ) { return (TableListHead *) TableArenaAt( *worker.arena, (uint32_t) list.elements.value ); } + uint32_t at = 0; + TableListHead * head = (TableListHead *) worker.AllocRaw( (int64_t) sizeof( TableListHead ), (int64_t) alignof( TableListHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + list.elements.value = (int64_t) at; + return head; +} + +// one element's storage, appended: the current segment when it has room, a +// new one carved from one call to the pair when it does not. NULL means NOT +// ADDED: an arena that cannot carve another segment, or a count at the int32 +// cap (§2.2, §2.9). +template +inline typename TableList::Element * TableListAppend( TableWorker & worker, TableListHead * head, TableList & list ) +{ + typedef typename TableList::Element Element; + if ( list.count >= INT32_MAX ) { return NULL; } // the int32 storage cap + TableListSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableListSegmentElements ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableListSegment *) worker.AllocRaw( (int64_t) sizeof( TableListSegment ), (int64_t) alignof( TableListSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableListSegment * previous = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Element * element = segment->elements + segment->used; + segment->used++; + head->live++; + list.count++; + return element; +} + +// ADD, whole: the head, the append, and the element at its declared defaults +// (§2.9). The text form's placement is this same call, because a list has no +// key to place under (§16). +template +inline typename TableList::Element * TableListPlace( TableWorker & worker, TableList & list ) +{ + typedef typename TableList::Element Element; + TableListHead * head = TableListReach( worker, list ); + if ( head == NULL ) { return NULL; } + Element * element = TableListAppend( worker, head, list ); + if ( element == NULL ) { return NULL; } + new ( element ) Element(); // value-init: the declared defaults, and null for a slot + return element; +} + +// ERASE, ADDRESSED BY THE POINTER (§2.9): the element Add handed back is the +// handle, because a list has no key and the address is the one thing the +// builder promises never moves (§6.4). It marks the element DEAD, one bit in +// the segment's slot and not in the element storage, and decrements the live +// count. False when the pointer is not this list's. Its storage is reclaimed +// at RESET and never reused mid-build, the map's rule for the map's reason. +template +inline bool TableListErase( TableArena & arena, TableList & list, const typename TableList::Element * element ) +{ + typedef typename TableList::Element Element; + if ( list.elements.value == 0 || element == NULL ) { return false; } + TableListHead * head = (TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableListSegment * segment = (TableListSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + if ( element >= segment->elements && element < segment->elements + segment->used ) + { + const int32_t i = (int32_t) ( element - segment->elements ); + if ( TableListSegmentDead( segment->dead, i ) ) { return false; } // already erased + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + list.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INDEX order, live elements only (§2.9) ---- +template struct TableListEach +{ + typedef typename TableList::Element Element; + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableListSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableListSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + Element * operator*() const { return segment->elements + index; } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableListEach TableListEachOf( const TableArena & arena, const TableList & list ) +{ + TableListEach each = { &arena, TableRef() }; + if ( list.elements.value != 0 ) + { + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + each.first = head->first; + } + return each; +} + +// ---- the INDEX-ORDER CURSOR the four writing walks read (§2.9) ---- +// +// Measure, Save, Lock and Cook each visit a list's live elements in the order +// they were added, and they allocate nothing to do it: a region's cursor is +// the array in place, and the builder's walks the segment chain. Indexing the +// builder's form is SEQUENTIAL by construction, every walk steps i, i + 1, +// i + 2, so the cursor remembers where the last access landed and moves one +// live slot per step. An access behind the memo restarts from the first +// segment, which no walk here does. +template struct TableListCursor +{ + const Element * elements = NULL; // the region's form: the array in place + const TableArena * arena = NULL; // the builder's form: the segments + TableRef first; + int32_t count = 0; + bool ok = false; + // the memo: the segment and slot the last access landed on, and the live + // index that slot holds + mutable const TableListSegment * segment = NULL; + mutable int32_t within = -1; + mutable int32_t logical = -1; + + const Element * At( int32_t index ) const + { + if ( elements != NULL ) { return elements + index; } + if ( segment == NULL || index < logical ) + { + segment = first.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL; + within = -1; + logical = -1; + } + while ( logical < index ) + { + for ( ;; ) + { + within++; + while ( segment != NULL && within >= segment->used ) + { + segment = segment->next.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + within = 0; + } + if ( segment == NULL ) { return NULL; } // the slot and the head disagree + if ( !TableListSegmentDead( segment->dead, within ) ) { break; } + } + logical++; + } + return segment->elements + within; + } + const Element & operator[]( int32_t index ) const { return *At( index ); } +}; + +// the REGION form: the array is the cursor +template +inline TableListCursor::Element> TableListElements( const TableRegionCtx &, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.elements = list.Elements(); + cursor.count = list.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: the live elements out of the segment chain, in the +// order they were added. A dead element costs nothing on any wire (§2.9). +template +inline TableListCursor::Element> TableListElements( const TableArena & arena, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.arena = &arena; + cursor.count = list.count; + if ( list.elements.value == 0 || list.count <= 0 ) { cursor.ok = list.count == 0; cursor.count = 0; return cursor; } + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + if ( head->live != list.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + cursor.first = head->first; + cursor.ok = true; + return cursor; +} + +template +inline TableListCursor::Element> TableListElements( const TableArenaCtx & ctx, const TableList & list ) +{ + return TableListElements( *ctx.arena, list ); +} + +// ---- the LOAD side: where a decoded element lands (§2.9) ---- +// +// The same two shapes the map's fill takes, because the decoder above them +// cannot tell which it has: a REGION carves the element array out of the +// holder node's own extent, PRE-ORDER, and the TOOL's path appends into the +// builder's arena. +template struct TableListFill +{ + typedef typename TableList::Element Element; + TableList * list = NULL; + Element * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; + bool refused = false; // a count above the int32 cap on the tool's path: LoadBuilder answers NULL +}; + +template +inline TableListFill TableListFillBegin( const TableNodeMap & nodes, TableList & list, uint64_t n ) +{ + typedef typename TableList::Element Element; + TableListFill fill; + fill.list = &list; + list.elements.value = 0; + list.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( n > (uint64_t) INT32_MAX ) + { + // A COUNT ABOVE THE int32 STORAGE CAP (§2.2, §2.9): into a region it was + // refused by LoadMeasure before this ran, and into a builder it is the + // refusal LoadBuilder answers NULL for, moving no counter + fill.refused = nodes.carve->worker != NULL; + return fill; + } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Element ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Element ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Element *) base; + fill.capacity = (int32_t) n; + list.elements.value = (int64_t) ( base - (const uint8_t *) &list.elements ); + fill.ok = true; + return fill; +} + +// the next slot, at the element's declared defaults. NULL when the arena +// could not carve, which the decoder reports as framing damage +template inline typename TableList::Element * TableListFillNext( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count >= fill.capacity ) { return NULL; } + Element * element = fill.array + fill.list->count; + new ( element ) Element(); + fill.list->count++; + return element; + } + return TableListPlace( *fill.worker, *fill.list ); +} + +// A SLOT WHOSE ELEMENT NEVER LANDED is given back (§2.9, §4): the array keeps +// what it decoded, and an element whose own framing gave out before one byte +// of it decoded was not decoded. The region's form uncounts it, and the builder's +// marks it dead, which is what the storage rule allows mid-build. +template inline void TableListFillDrop( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count > 0 ) { fill.list->count--; } + return; + } + if ( fill.list->elements.value == 0 ) { return; } + TableListHead * head = (TableListHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.list->elements.value ); + if ( head->last.value == 0 ) { return; } + TableListSegment * segment = (TableListSegment *) TableArenaAt( *fill.worker->arena, (uint32_t) head->last.value ); + if ( segment->used <= 0 ) { return; } + const int32_t i = segment->used - 1; + if ( TableListSegmentDead( segment->dead, i ) ) { return; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + fill.list->count--; +} + +// an EMPTY list's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableListFillEnd( TableListFill & fill ) +{ + if ( fill.array != NULL && fill.list->count == 0 ) { fill.list->elements.value = 0; } +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.9, §6.5) ---- +// +// N x sizeof( T ) rounded to alignof( T ), AT EVERY DEPTH. N is framing and +// not a value, so this reads no field: it walks the list's own header and, +// where a table element holds a list or a map of its own, the elements' +// headers under it. Every -1 carries its REASON (§6.5): the int32 cap first, +// because a count past it cannot fit any body, and then the body's own L. +inline bool TableListWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t elem_size, int64_t elem_align, uint8_t elem_kind, int64_t elem_floor, + TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != elem_kind ) { return true; } // another element kind: §4's ordinary kind mismatch, the field reads empty + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; } // an N the list's L cannot carry + at = ( at + elem_align - 1 ) & ~( elem_align - 1 ); + at += (int64_t) n * elem_size; + if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace ` + pkg + ` + +#endif // ` + guard + ` +` +} + +// ---- the list's wire codecs (docs/SPEC-TABLES.md §2.9, §3) ---- +// +// A list rides as the kind 14 ARRAY a bounded array rides as, over the +// element's own kind, so the element measure, write and read are the bounded +// array's own emitters over a cursor rather than over inline storage. What +// is the list's own is the cursor and the fill. + +// emitListMeasureField adds one list field to a body's measured size. +func (g *tableGen) emitListMeasureField(f *ir.Field) { + id := ir.TableFieldWireId(f) + cursor := "cursor_" + f.Name + g.pf(" {\n") + g.pf(" // %s: a kind %d array of kind %d elements, INDEX order (§2.9)\n", f.Name, tkArray, listElementWireKind(f)) + g.pf(" TableListCursor<%s> %s = TableListElements( ctx, value.%s );\n", g.listElementType(f), cursor, f.Name) + g.pf(" if ( !%s.ok ) { return -1; } // the slot and the head disagree\n", cursor) + g.pf(" if ( %s.count > 0 ) // an EMPTY list elides, the by-value rule (§3)\n {\n", cursor) + g.pf(" const uint64_t ref_%s = %s;\n", f.Name, g.wireRef(id)) + g.pf(" int64_t body_%s = 0;\n", f.Name) + g.emitArrayBodyMeasure(f, listElementWireKind(f), "body_"+f.Name, cursor+".count", cursor+"[%s]", " ", "return -1;", "_"+f.Name) + g.pf(" bytes += TableLebBytes( ref_%s ) + 1 + %s;\n", f.Name, framed("body_"+f.Name)) + g.pf(" }\n") + g.pf(" }\n") +} + +// emitListWriteField writes one list field: the array framing, then the live +// elements in INDEX order, dead elements dropped. +func (g *tableGen) emitListWriteField(f *ir.Field) { + id := ir.TableFieldWireId(f) + cursor := "cursor_" + f.Name + g.pf(" {\n") + g.pf(" TableListCursor<%s> %s = TableListElements( ctx, value.%s ); // %s\n", g.listElementType(f), cursor, f.Name, f.Name) + g.pf(" if ( !%s.ok ) { return false; }\n", cursor) + g.pf(" if ( %s.count > 0 ) // an EMPTY list elides, the by-value rule (§3)\n {\n", cursor) + g.pf(" const uint64_t ref_%s = %s;\n", f.Name, g.wireRef(id)) + g.pf(" int64_t body_%s = 0;\n", f.Name) + g.emitArrayBodyMeasure(f, listElementWireKind(f), "body_"+f.Name, cursor+".count", cursor+"[%s]", " ", "return false;", "_"+f.Name) + g.pf(" w.putleb( ref_%s ); w.put8( %d ); w.putleb( (uint64_t) body_%s ); // %s\n", f.Name, tkArray, f.Name, f.Name) + g.emitArrayBodyWrite(f, listElementWireKind(f), cursor+".count", cursor+"[%s]", " ", "_"+f.Name) + g.pf(" }\n") + g.pf(" }\n") +} + +// emitListReadField decodes one list field, and every reader rule §2.9 +// states lands here: the element kind checked against the reader's +// declaration, the count taken as the data's with nothing to clamp it +// against, the elements bounded by the body's own L, and a slot whose element +// never landed given back. +func (g *tableGen) emitListReadField(f *ir.Field) { + ind := " " + elemKind := listElementWireKind(f) + g.pf("%suint64_t body_len = 0;\n", ind) + g.pf("%sif ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; }\n", ind) + g.pf("%sint64_t body_end = r.offset + (int64_t) body_len;\n", ind) + g.pf("%s// A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps\n", ind) + g.pf("%s// the value it has, no counter is raised, and the walk continues past L.\n", ind) + g.pf("%sif ( body_len >= 2 )\n%s{\n", ind, ind) + g.pf("%s uint8_t elem_kind = r.get8();\n", ind) + g.pf("%s uint64_t count = 0;\n", ind) + g.pf("%s const bool counted_ok = r.getleb( count );\n", ind) + g.pf("%s if ( !counted_ok ) { r.report->malformed = true; }\n", ind) + g.pf("%s // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's\n", ind) + g.pf("%s // element-kind rule: the field reads EMPTY and one kind_mismatch counts\n", ind) + g.pf("%s else if ( elem_kind != %d ) { r.report->kind_mismatch++; r.offset = body_end; break; }\n", ind, elemKind) + g.pf("%s else\n%s {\n", ind, ind) + g.pf("%s // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped\n", ind) + g.pf("%s // cannot fire on it. A count above the int32 storage cap is the\n", ind) + g.pf("%s // fill's refusal, and it moves no counter.\n", ind) + g.pf("%s TableListFill<%s> fill = TableListFillBegin( nodes, value.%s, count );\n", ind, g.listTypeArg(f), f.Name) + g.pf("%s if ( fill.refused ) { nodes.refused = true; return false; }\n", ind) + g.pf("%s if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; }\n", ind) + g.pf("%s // elements are BOUNDED by the field body: a count the length cannot\n", ind) + g.pf("%s // cover keeps the decoded prefix, flags malformed, and the parent\n", ind) + g.pf("%s // continues at the next field\n", ind) + g.pf("%s TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids );\n", ind) + g.pf("%s for ( uint64_t i = 0; i < count; i++ )\n%s {\n", ind, ind) + g.pf("%s %s * slot = TableListFillNext( fill );\n", ind, g.listElementType(f)) + g.pf("%s if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve\n", ind) + g.pf("%s bool landed = false;\n", ind) + g.pf("%s do\n%s {\n", ind, ind) + g.emitTableReadElementInto(f, elemKind, "( *slot )", ind+" ", "sub", "_"+f.Name) + g.pf("%s landed = true;\n", ind) + g.pf("%s } while ( 0 );\n", ind) + g.pf("%s if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded\n", ind) + g.pf("%s }\n", ind) + g.pf("%s TableListFillEnd( fill );\n", ind) + g.pf("%s }\n", ind) + g.pf("%s}\n", ind) + g.pf("%sr.offset = body_end; // excess bytes and slack skip via the length\n", ind) +} + +// ---- the three walks at a list (docs/SPEC-TABLES.md §2.9, §3.1) ---- +// +// A list is a BY-VALUE EDGE of the ONE declaration-order walk: it is reached +// at its field's position, its elements are visited in INDEX ORDER, and each +// element is descended for the pointer slots inside it before the next +// element is reached. A []*T declared before a pointer field therefore +// reaches a shared node FIRST and numbers it first. The rule is the walk's, +// not the list's. + +// emitListEdge opens one list's cursor under the walk's subjects and visits +// each element through the emitter's own pointer or descend visitor: the +// element IS the pointer slot on a []*T, and a variable table to descend on a +// []T. The pack's twin is the array ExtentPack already placed in the node's +// extent, reached from the write subject's slot. +func (g *tableGen) emitListEdge(f *ir.Field, v edgeVisitor, onBad string) { + elem := g.listElementType(f) + cursor := "cursor_" + f.Name + g.pf(" { // %s: a by-value edge, elements in INDEX order (§2.9, §3.1)\n", f.Name) + g.pf(" TableListCursor<%s> %s = TableListElements( ctx, %s.%s );\n", elem, cursor, v.read, f.Name) + g.pf(" if ( !%s.ok ) { %s }\n", cursor, onBad) + placed := "" + if v.write != "" { + placed = "placed_" + f.Name + slot := v.write + "." + f.Name + g.pf(" %s * %s = (%s *) ( %s.elements.value != 0 ? ( (uint8_t *) &%s.elements + %s.elements.value ) : NULL );\n", + elem, placed, elem, slot, slot, slot) + } + g.pf(" for ( int32_t i = 0; i < %s.count; i++ )\n {\n", cursor) + expr := edgeExpr{Src: cursor + "[i]"} + if placed != "" { + expr.Dst = placed + "[i]" + } + saved := g.indent + g.indent = saved + " " + if listElementIsPointer(f) { + v.pointer(f, expr) + } else { + v.descend(f.Type.Name, expr, " ") + } + g.indent = saved + g.pf(" }\n }\n") +} + +// ---- the builder's three (docs/SPEC-TABLES.md §2.9) ---- +// +// FREE FUNCTIONS taking the worker or the arena, as Emplace and the map's +// five are. Add takes the WORKER because it may allocate a segment. Each and +// Erase take the arena because neither ever does. +func (g *tableGen) emitListBuilderSurface(owner *ir.Struct, f *ir.Field) { + hold := fmt.Sprintf("TableList<%s> & list", g.listTypeArg(f)) + elem := g.listElementType(f) + g.pf("// ---- %s.%s: the builder's three (§2.9) ----\n\n", owner.Name, f.Name) + if listElementIsPointer(f) { + g.pf("// ADD: the element is appended and handed back to fill. On a []*T that is\n") + g.pf("// the SLOT at null, which %sEmplace fills as it fills any pointer slot,\n", f.Type.Name) + g.pf("// and a second slot may hold the same reference: two slots, one node.\n") + } else { + g.pf("// ADD: the element is appended at its declared defaults and handed back to\n") + g.pf("// fill. Nothing ever moves (§6.4), so the pointer stays valid while other\n") + g.pf("// elements arrive.\n") + } + g.pf("// NULL means NOT ADDED: an arena that cannot carve another segment, or a\n") + g.pf("// count at the int32 cap. A caller that needs the reason checks size().\n") + g.pf("inline %s * %s( TableWorker & worker, %s )\n{\n", elem, listVerb(owner.Name, f, "Add"), hold) + g.pf(" return TableListPlace( worker, list );\n}\n\n") + g.pf("// ERASE, by the element's own pointer: marks it DEAD, one bit in the\n") + g.pf("// segment's slot and not in the element storage. False when the pointer is\n") + g.pf("// not this list's. Storage is held until the builder resets. INDICES ARE\n") + g.pf("// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save.\n") + g.pf("inline bool %s( TableArena & arena, %s, const %s * element )\n{\n", listVerb(owner.Name, f, "Erase"), hold, elem) + g.pf(" return TableListErase( arena, list, element );\n}\n\n") + g.pf("// EACH on the builder: INDEX order, live elements only, yielding the\n") + g.pf("// element Add handed back.\n") + g.pf("inline TableListEach<%s> %s( const TableArena & arena, const TableList<%s> & list )\n{\n", g.listTypeArg(f), listVerb(owner.Name, f, "Each"), g.listTypeArg(f)) + g.pf(" return TableListEachOf( arena, list );\n}\n\n") +} + +// emitListBuilderSurfaces emits every list field's builder three, after the +// arena and list runtimes they are spelled in terms of. +func (g *tableGen) emitListBuilderSurfaces(members []*ir.Struct) { + if !g.anyList { + return + } + for _, st := range members { + for _, f := range listFieldsOf(st) { + g.emitListBuilderSurface(st, f) + } + } +} + +// emitListAlignAsserts holds every list element to the arena's alignment +// (docs/SPEC-TABLES.md §2.9): a node's extent begins at the arena's alignment +// and nothing inside it can ask for more. +func (g *tableGen) emitListAlignAsserts(members []*ir.Struct) { + if !g.anyList { + return + } + for _, st := range members { + for _, f := range listFieldsOf(st) { + g.pf("static_assert( alignof( %s ) <= kTableAlign, \"%s.%s: an unbounded array's element alignment must fit the arena's\" );\n", g.listElementType(f), st.Name, f.Name) + } + } + g.pf("\n") +} + +// listPlaceThunk is the descriptor's PLACE column for a list field +// (docs/SPEC-TABLES.md §8.1, §16): the one resolver the text walk cannot spell +// for itself, because TableList is a type it has no name for. The key +// arguments are the map's and a list ignores them: it appends. +func (g *tableGen) listPlaceThunk(f *ir.Field) string { + return fmt.Sprintf("[]( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(%s *) slot ); }", g.listStorageType(f)) +} diff --git a/internal/codegen/cpptable/maps.go b/internal/codegen/cpptable/maps.go index 4c258e8f9..f842edec8 100644 --- a/internal/codegen/cpptable/maps.go +++ b/internal/codegen/cpptable/maps.go @@ -10,7 +10,6 @@ package cpptable import ( "fmt" - "slices" "strings" "github.com/mas-bandwidth/schema/v2/ir" @@ -525,12 +524,6 @@ inline TableMapEach TableMapEachOf( const TableArena & arena, const Table return each; } -// AN UNREACHED SLOT MUST HOLD NO MAP WITH ENTRIES IN IT (§2.8, §7.6). An empty -// map takes no bytes, so a record whose extent measures ZERO is a record whose -// every by-value map is empty; a measure that REFUSED answers non-zero here -// too, and refusing on it is the same answer one level up. -inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } - // ---- the LOAD side: where a decoded entry lands (§2.8) ---- // // THE READER TRUSTS NOTHING and spends one compare per entry. Every load path @@ -540,16 +533,9 @@ inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } // out of the holder node's own extent, and the TOOL's path appends into the // builder's arena, and the decoder above them cannot tell which it has. -// TableMapCarve is a node's extent cursor, PRE-ORDER: a map's whole entry -// array first, then, entry by entry in key order, the arrays of any map an -// entry's value holds by value. The cursor is the node map's, because the -// generated decoder is threaded with that and not with a region. -struct TableMapCarve -{ - uint8_t * at = NULL; // the region path: the node's extent, unspent - int64_t left = 0; - TableWorker * worker = NULL; // the TOOL's path: entries come from the arena -}; +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. // TableMapFill is one map field being decoded: where the next entry lands, and // the entry that last LANDED, which is what the ascending check compares @@ -674,10 +660,9 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // LoadMeasure's term for a map is N x sizeof( Entry ) rounded to // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value -// holds a map of its own, the entries' headers under it. The caller owns the -// allocation precisely so it can refuse a number it did not expect. -typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ); - +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect, +// and a refusal carries its reason (§6.5). // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -685,8 +670,8 @@ typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, i static const int64_t kTableMapEntryFloor = 2; inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, - int64_t entry_size, int64_t entry_align, TableMapWireExtentFn inner, - const TableIdTable * ids ) + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; TableReader r( body, length, &scratch, ids ); @@ -695,7 +680,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } const int64_t rest = length - r.offset; - if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { return false; } // an N the map's L cannot carry + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); at += (int64_t) n * entry_size; if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term @@ -703,49 +688,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & { uint64_t elem = 0; if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// the same framing walk over an ARRAY OF TABLES that is not a map: its -// elements' own maps are part of this node's extent too -inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each -// length-prefixed element (docs/SPEC-TABLES.md §3.2) -inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t key = 0; - if ( !r.getleb( key ) ) { return true; } - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } r.offset += (int64_t) elem; } return true; @@ -1149,410 +1092,6 @@ func (g *tableGen) emitMapReadField(f *ir.Field) { g.pf("%sr.offset = body_end; // the remaining entries skip by the map's L\n", ind) } -// ---- the NODE EXTENT (docs/SPEC-TABLES.md §2.8, §6.3) ---- -// -// A map's entries are BY-VALUE RECORDS INSIDE THE HOLDER'S NODE EXTENT, laid -// after the record's own storage: count x sizeof( Entry ) at alignof( Entry ), -// zero slack, one array per map reachable BY VALUE from the record — which -// includes a map inside a nested table and a map inside an entry — in -// depth-first field order. The placement is PRE-ORDER: a map's whole entry -// array first, then, entry by entry in key order, the arrays of any map an -// entry's value holds by value. -// -// Two emitters walk that layout and they are ONE walk: the measure advances a -// running offset, and the pack advances the same one and copies. Nothing -// passes between them, which is what makes `used == total` a real check. - -// hasMapExtent reports a member with any map reachable by value — the members -// that carry an extent, and the ones whose two extent walks are emitted. -func (g *tableGen) hasMapExtent(st *ir.Struct) bool { - for _, f := range st.Fields { - if f.IsMap() { - return true - } - // A MAP REACHABLE BY VALUE IS THIS RECORD'S EXTENT (docs/SPEC-TABLES.md - // §2.8), whichever by-value edge reaches it: a nested table, an array - // of them, an enum-keyed array of them, or a union arm. - switch g.edgeOf(f) { - case edgeNested: - if ref, ok := f.Type.Ref.(*ir.Struct); ok && g.hasMapExtent(ref) { - return true - } - case edgeArm: - for _, v := range f.Type.Ref.(*ir.Union).Variants { - if ref := memberOf(g.unit, v.Type); ref != nil && g.hasMapExtent(ref) { - return true - } - } - } - // a nested table this walk does not call an edge can still hold a map, - // because a map makes its holder VARIABLE and every variable nesting is - // an edge — so there is nothing else to look at here - } - return false -} - -// memberOf resolves one closure member by name. -func memberOf(u *ir.Unit, name string) *ir.Struct { - if st := u.Tables[name]; st != nil { - return st - } - return u.Structs[name] -} - -// emitMapExtent emits `MapExtentAt`: the running offset every map array -// reachable by value from one record takes, in the order the pack lays them. -func (g *tableGen) emitMapExtent(st *ir.Struct) { - g.pf("// %sMapExtentAt: the node extent %s's maps take, PRE-ORDER, advancing the\n", st.Name, st.Name) - g.pf("// running offset exactly as %sMapPack advances it (docs/SPEC-TABLES.md §2.8).\n", st.Name) - g.pf("template \ninline bool %sMapExtentAt( const Ctx & ctx, const %s & value, int64_t & at )\n{\n", st.Name, st.Name) - if !g.hasMapExtent(st) { - g.pf(" (void) ctx; (void) value; (void) at; // no map below this record\n") - g.pf(" return true;\n}\n\n") - return - } - g.emitMapExtentWalk(st, "value", func(f *ir.Field, expr, ind string) { - entry := mapEntryOf(f) - g.pf("%s{\n", ind) - g.pf("%s TableMapCursor<%s> cursor = TableMapOrder( ctx, %s );\n", ind, entry.Name, expr) - g.pf("%s if ( !cursor.ok ) { return false; }\n", ind) - g.pf("%s at = ( at + %d ) & ~(int64_t) %d; // at alignof( %s )\n", ind, alignOfEntry(g.unit, entry)-1, alignOfEntry(g.unit, entry)-1, entry.Name) - g.pf("%s at += (int64_t) cursor.count * (int64_t) sizeof( %s ); // the whole array FIRST\n", ind, entry.Name) - if g.isVar(entry.Name) { - g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order\n%s {\n", ind, ind) - g.pf("%s if ( !%sMapExtentAt( ctx, *cursor[i], at ) ) { TableMapRelease( cursor ); return false; }\n", ind, entry.Name) - g.pf("%s }\n", ind) - } - g.pf("%s TableMapRelease( cursor );\n", ind) - g.pf("%s}\n", ind) - }, func(table, expr, ind string) { - g.pf("%sif ( !%sMapExtentAt( ctx, %s, at ) ) { return false; }\n", ind, table, expr) - }) - g.pf(" return true;\n}\n\n") - g.pf("// the whole extent of one node, from a fresh offset: what a pack reserves\n") - g.pf("// for it beside the record's own storage.\n") - g.pf("template \ninline int64_t %sMapExtent( const Ctx & ctx, const %s & value )\n{\n", st.Name, st.Name) - g.pf(" int64_t at = 0;\n") - g.pf(" if ( !%sMapExtentAt( ctx, value, at ) ) { return -1; }\n", st.Name) - g.pf(" return at;\n}\n\n") -} - -// emitMapPack emits `MapPack`: the same walk, copying each map's entries in -// key order into the node's extent and pointing the record's slot at them. -func (g *tableGen) emitMapPack(st *ir.Struct) { - g.pf("// %sMapPack: carve %s's map arrays out of the node's extent and copy the\n", st.Name, st.Name) - g.pf("// entries in ASCENDING key order, PRE-ORDER, advancing the same running\n") - g.pf("// offset %sMapExtentAt advances (docs/SPEC-TABLES.md §2.8).\n", st.Name) - g.pf("template \ninline bool %sMapPack( const Ctx & ctx, const %s & src, %s & dst, uint8_t * extent, int64_t & at, int64_t capacity )\n{\n", st.Name, st.Name, st.Name) - if !g.hasMapExtent(st) { - g.pf(" (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no map below this record\n") - g.pf(" return true;\n}\n\n") - return - } - dstOf := func(expr string) string { return "dst" + expr[len("src"):] } - g.emitMapExtentWalk(st, "src", func(f *ir.Field, expr, ind string) { - entry := mapEntryOf(f) - slot := dstOf(expr) - g.pf("%s{\n", ind) - g.pf("%s TableMapCursor<%s> cursor = TableMapOrder( ctx, %s );\n", ind, entry.Name, expr) - g.pf("%s if ( !cursor.ok ) { return false; }\n", ind) - g.pf("%s at = ( at + %d ) & ~(int64_t) %d;\n", ind, alignOfEntry(g.unit, entry)-1, alignOfEntry(g.unit, entry)-1) - g.pf("%s const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( %s );\n", ind, entry.Name) - g.pf("%s if ( at + bytes > capacity ) { TableMapRelease( cursor ); return false; }\n", ind) - g.pf("%s %s * placed = (%s *) ( extent + at );\n", ind, entry.Name, entry.Name) - g.pf("%s at += bytes;\n", ind) - g.pf("%s %s.count = cursor.count;\n", ind, slot) - g.pf("%s %s.padding = 0;\n", ind, slot) - g.pf("%s %s.entries.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &%s.entries ) : 0;\n", ind, slot, slot) - g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ )\n%s {\n", ind, ind) - g.pf("%s memcpy( (void *) ( placed + i ), (const void *) cursor[i], sizeof( %s ) ); // trivially copyable, by construction\n", ind, entry.Name) - g.pf("%s }\n", ind) - if g.isVar(entry.Name) { - g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ )\n%s {\n", ind, ind) - g.pf("%s if ( !%sMapPack( ctx, *cursor[i], placed[i], extent, at, capacity ) ) { TableMapRelease( cursor ); return false; }\n", ind, entry.Name) - g.pf("%s }\n", ind) - } - g.pf("%s TableMapRelease( cursor );\n", ind) - g.pf("%s}\n", ind) - }, func(table, expr, ind string) { - g.pf("%sif ( !%sMapPack( ctx, %s, %s, extent, at, capacity ) ) { return false; }\n", ind, table, expr, dstOf(expr)) - }) - g.pf(" return true;\n}\n\n") -} - -// emitMapExtentWalk is the ONE walk both extent emitters take: the record's -// fields in declaration order, every map at its own position, descending each -// by-value edge in place. A pointer is NOT an edge here — a pointee is its own -// node with its own extent — and neither is a union arm that is one, for the -// same reason. -func (g *tableGen) emitMapExtentWalk(st *ir.Struct, subject string, mapField func(f *ir.Field, expr, ind string), descend func(table, expr, ind string)) { - v := edgeVisitor{read: subject} - for _, f := range st.Fields { - if f.IsMap() { - mapField(f, subject+"."+f.Name, " ") - continue - } - switch g.edgeOf(f) { - case edgeNested: - ref, _ := f.Type.Ref.(*ir.Struct) - if ref == nil || !g.hasMapExtent(ref) { - continue - } - g.emitVariableByValueWalk(f, v, func(expr edgeExpr) { descend(f.Type.Name, expr.Src, " ") }) - g.emitUnreachedMapRefusal(f, ref, subject) - case edgeArm: - un := f.Type.Ref.(*ir.Union) - any := false - for _, arm := range un.Variants { - if ref := memberOf(g.unit, arm.Type); ref != nil && g.hasMapExtent(ref) { - any = true - } - } - if !any { - continue - } - // only the ARMS that hold a map are descended: a pointer arm and a - // byte buffer arm reach nodes, not this node's extent - armed := edgeVisitor{read: subject, - pointer: func(*ir.Field, edgeExpr) {}, - blob: func(*ir.Field, edgeExpr) {}, - descend: func(table string, expr edgeExpr, indent string) { - if ref := memberOf(g.unit, table); ref != nil && g.hasMapExtent(ref) { - descend(table, expr.Src, indent) - } - }, - } - g.emitVariableUnionWalk(f, armed) - } - } -} - -// alignOfEntry is the C ABI alignment of one generated entry record — the -// alignment its array is laid at, and the same model §20.3 commits the -// compiler to for every record in the closure. -func alignOfEntry(u *ir.Unit, entry *ir.Struct) int64 { - if ml := ir.RecordLayout(u, entry); ml != nil && ml.Align > 0 { - return ml.Align - } - return 8 -} - -// ---- the three walks at a map (docs/SPEC-TABLES.md §2.8, §3.1) ---- -// -// A map is a BY-VALUE EDGE of the ONE declaration-order walk: it is reached at -// its field's position, its entries are visited in ASCENDING KEY ORDER, and -// each entry's value is descended for the pointer slots inside it before the -// next entry is reached. A map declared before a pointer field therefore -// reaches a shared node FIRST and numbers it first, exactly as a union arm or -// a nested table declared there does. The rule is the walk's, not the map's. - -// mapNumberEdge descends one map's entries for the NUMBERING walk. -func (g *tableGen) mapNumberEdge(f *ir.Field) { - g.emitMapEntryLoop(f, "value", "return false;", func(entry, elem, ind string) { - g.pf("%sif ( !%sNumber( ctx, numbering, %s ) ) { TableMapRelease( cursor_%s ); return false; }\n", ind, entry, elem, f.Name) - }) -} - -// mapPackMeasureEdge descends one map's entries for the PACK MEASURE. -func (g *tableGen) mapPackMeasureEdge(f *ir.Field) { - g.emitMapEntryLoop(f, "value", "return -1;", func(entry, elem, ind string) { - g.pf("%sint64_t inner = %sPackMeasure( ctx, seen, %s );\n", ind, entry, elem) - g.pf("%sif ( inner < 0 ) { TableMapRelease( cursor_%s ); return -1; }\n", ind, f.Name) - g.pf("%sbytes += inner;\n", ind) - }) -} - -// mapPackEdge descends one map's entries for the PACK, against the array -// MapPack already placed in the node's extent. -func (g *tableGen) mapPackEdge(f *ir.Field) { - entry := mapEntryOf(f) - g.emitMapEntryLoopHead(f, "src", "return false;") - g.pf(" %s * placed_%s = (%s *) ( dst.%s.entries.value != 0 ? ( (uint8_t *) &dst.%s.entries + dst.%s.entries.value ) : NULL );\n", - entry.Name, f.Name, entry.Name, f.Name, f.Name, f.Name) - g.pf(" for ( int32_t i = 0; i < cursor_%s.count; i++ )\n {\n", f.Name) - g.pf(" if ( !%sPackEdges( ctx, seen, *cursor_%s[i], placed_%s[i], base, capacity, used ) ) { TableMapRelease( cursor_%s ); return false; }\n", - entry.Name, f.Name, f.Name, f.Name) - g.pf(" }\n") - g.pf(" TableMapRelease( cursor_%s );\n }\n", f.Name) -} - -// emitMapEntryLoopHead opens one map's sorted cursor over the given subject. -func (g *tableGen) emitMapEntryLoopHead(f *ir.Field, subject, onBad string) { - entry := mapEntryOf(f) - g.pf(" { // %s: a by-value edge, entries in ASCENDING key order (§2.8, §3.1)\n", f.Name) - g.pf(" TableMapCursor<%s> cursor_%s = TableMapOrder( ctx, %s.%s );\n", entry.Name, f.Name, subject, f.Name) - g.pf(" if ( !cursor_%s.ok ) { %s }\n", f.Name, onBad) -} - -// emitMapEntryLoop is the whole shape: the cursor, the loop, the release. -func (g *tableGen) emitMapEntryLoop(f *ir.Field, subject, onBad string, body func(entry, elem, ind string)) { - entry := mapEntryOf(f) - g.emitMapEntryLoopHead(f, subject, onBad) - g.pf(" for ( int32_t i = 0; i < cursor_%s.count; i++ )\n {\n", f.Name) - body(entry.Name, fmt.Sprintf("*cursor_%s[i]", f.Name), " ") - g.pf(" }\n") - g.pf(" TableMapRelease( cursor_%s );\n }\n", f.Name) -} - -// emitMapWalkSurface emits the two extent walks for every variable member of a -// map-bearing unit. They are emitted for EVERY such member, because a walk -// that descends a by-value nesting has to be able to name the nested one's. -func (g *tableGen) emitMapWalkSurface(members []*ir.Struct) { - if !g.anyMap { - return - } - for _, st := range g.varMembers(members) { - g.emitMapWireExtent(st) - g.emitMapExtent(st) - g.emitMapPack(st) - } -} - -// emitNodeBytes emits the bytes ONE NODE takes in a packed region: the -// record's own storage rounded to the arena's alignment, plus the extent its -// maps take (docs/SPEC-TABLES.md §2.8, §6.3), the sum rounded again so the -// next node starts aligned. A unit with no map emits exactly the term it -// always emitted. -func (g *tableGen) emitNodeBytes(table, expr, ind, onBad string, plain func(term string), extent func(term string)) { - target := memberOf(g.unit, table) - if !g.anyMap || target == nil || !g.hasMapExtent(target) { - // a node with no map below it takes exactly the term it always took, - // so a map-free unit emits what it emitted before the construct existed - plain(fmt.Sprintf("TableAlignUp64( (int64_t) sizeof( %s ) )", table)) - return - } - g.pf("%sint64_t node_extent = %sMapExtent( ctx, %s );\n", ind, table, expr) - g.pf("%sif ( node_extent < 0 ) { %s }\n", ind, onBad) - extent(fmt.Sprintf("TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + node_extent )", table)) -} - -// emitMapWireExtent emits `WireExtent`: the region bytes one record's maps -// command, read from the wire FRAMING alone at every depth -// (docs/SPEC-TABLES.md §2.8, §6.5). False is the refusal — an N the map's L -// cannot carry — and it is what makes LoadMeasure answer -1. -func (g *tableGen) emitMapWireExtent(st *ir.Struct) { - g.pf("// %sWireExtent: the extent %s's maps command, from the FRAMING alone.\n", st.Name, st.Name) - g.pf("// It reads no field value, so a caller can refuse a number it did not\n") - g.pf("// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5).\n") - g.pf("inline bool %sWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids )\n{\n", st.Name) - if !g.hasMapExtent(st) { - g.pf(" (void) body; (void) length; (void) at; (void) ids; // no map below this record\n") - g.pf(" return true;\n}\n\n") - return - } - g.pf(" TableReport scratch; // the scan's framing damage is the LOAD's to report\n") - g.pf(" TableReader r( body, length, &scratch, ids );\n") - g.pf(" for ( ;; )\n {\n") - g.pf(" uint64_t field_ref = 0;\n") - g.pf(" if ( !r.getleb( field_ref ) ) { return true; }\n") - g.pf(" if ( field_ref == 0 ) { return true; }\n") - g.pf(" if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; }\n") - g.pf(" const uint64_t field_id = ids->at( field_ref );\n") - g.pf(" if ( !r.has( 1 ) ) { return true; }\n") - g.pf(" uint8_t field_kind = r.get8();\n") - g.emitMapExtentWireCases(st) - g.pf(" if ( !r.skip( field_kind ) ) { return true; }\n") - g.pf(" }\n}\n\n") -} - -// emitMapExtentWireCases emits one arm per map field and one per by-value -// nesting that holds a map, in DECLARATION ORDER, so the framing scan advances -// the running offset in the same order the pack and the load carve it. -func (g *tableGen) emitMapExtentWireCases(st *ir.Struct) { - for _, f := range st.Fields { - if f.IsMap() { - entry := mapEntryOf(f) - inner := "NULL" - if g.hasMapExtent(entry) { - inner = "&" + entry.Name + "WireExtent" - } - g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s\n {\n", ir.TableFieldWireId(f), tkArray, f.Name) - g.pf(" uint64_t map_len = 0;\n") - g.pf(" if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; }\n") - g.pf(" const uint8_t * map_body = r.buffer + r.offset;\n") - g.pf(" r.offset += (int64_t) map_len;\n") - g.pf(" if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( %s ), (int64_t) alignof( %s ), %s, ids ) ) { return false; }\n", - entry.Name, entry.Name, inner) - g.pf(" continue;\n }\n") - continue - } - switch g.edgeOf(f) { - case edgeNested: - ref, _ := f.Type.Ref.(*ir.Struct) - if ref == nil || !g.hasMapExtent(ref) { - continue - } - // a nested table's maps are part of THIS node's extent, so its own - // scan runs over the nested body at the running offset - kind, walk := tkTable, "" - switch { - case f.KeyEnum != "": - kind, walk = tkKeyed, "TableWireExtentKeyed" - case f.Array != ir.ArrayNone: - kind, walk = tkArray, "TableWireExtentElements" - } - g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s: a nesting that holds a map\n {\n", ir.TableFieldWireId(f), kind, f.Name) - g.pf(" uint64_t nested_len = 0;\n") - g.pf(" if ( !r.getleb( nested_len ) || !r.room( nested_len ) ) { return true; }\n") - g.pf(" const uint8_t * nested_body = r.buffer + r.offset;\n") - g.pf(" r.offset += (int64_t) nested_len;\n") - if walk == "" { - g.pf(" if ( !%sWireExtent( nested_body, (int64_t) nested_len, at, ids ) ) { return false; }\n", f.Type.Name) - } else { - g.pf(" if ( !%s( nested_body, (int64_t) nested_len, at, &%sWireExtent, ids ) ) { return false; }\n", walk, f.Type.Name) - } - g.pf(" continue;\n }\n") - case edgeArm: - un := f.Type.Ref.(*ir.Union) - any := false - for _, v := range un.Variants { - if ref := memberOf(g.unit, v.Type); ref != nil && g.hasMapExtent(ref) { - any = true - } - } - if !any { - continue - } - g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s: a union arm that holds a map\n {\n", ir.TableFieldWireId(f), tkUnion, f.Name) - g.pf(" uint64_t arm_ref = 0;\n") - g.pf(" if ( !r.getleb( arm_ref ) ) { return true; }\n") - g.pf(" if ( arm_ref == 0 ) { continue; } // None: the reference is the whole payload\n") - g.pf(" if ( arm_ref > (uint64_t) ids->count ) { return true; }\n") - g.pf(" const uint64_t arm_id = ids->at( arm_ref );\n") - g.pf(" if ( !r.has( 1 ) ) { return true; }\n") - g.pf(" r.offset += 1; // the arm's kind byte\n") - g.pf(" uint64_t arm_len = 0;\n") - g.pf(" if ( !r.getleb( arm_len ) || !r.room( arm_len ) ) { return true; }\n") - g.pf(" const uint8_t * arm_body = r.buffer + r.offset;\n") - g.pf(" r.offset += (int64_t) arm_len;\n") - g.pf(" switch ( arm_id )\n {\n") - for _, v := range un.Variants { - ref := memberOf(g.unit, v.Type) - if ref == nil || !g.hasMapExtent(ref) { - continue - } - g.pf(" case 0x%016xull: if ( !%sWireExtent( arm_body, (int64_t) arm_len, at, ids ) ) { return false; } break; // %s\n", - ir.TableWireId(v.Name), v.Type, v.Name) - } - g.pf(" default: break; // an arm this reader cannot name reads None\n") - g.pf(" }\n") - g.pf(" continue;\n }\n") - } - } -} - -// emitRootDataBytes emits a load's DATA term for the root itself: its record, -// plus the extent its own maps take, read from the wire framing (§2.8, §6.5). -func (g *tableGen) emitRootDataBytes(st *ir.Struct, ind, onBad string) { - if !g.anyMap { - g.pf("%sint64_t data = TableAlignUp64( (int64_t) sizeof( %s ) );\n", ind, st.Name) - return - } - g.pf("%sint64_t root_extent = 0;\n", ind) - g.pf("%sif ( !%sWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { %s }\n", ind, st.Name, onBad) - g.pf("%sint64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + root_extent );\n", ind, st.Name) -} - // ---- the builder's five, and the optional index (docs/SPEC-TABLES.md §2.8) ---- // // FREE FUNCTIONS taking the worker or the arena, as Emplace and the arena At @@ -1707,246 +1246,59 @@ func (g *tableGen) emitMapBuilderSurfaces(members []*ir.Struct) { } } -// ---- the COOK's write side at a map (docs/SPEC-TABLES.md §2.8, §7.6) ---- +// ---- the three walks at a map (docs/SPEC-TABLES.md §2.8, §3.1) ---- // -// A cook is a region written verbatim, so a cooked map is its SORTED entry -// array where the cook put it: the node's extent, laid after the record's own -// storage by the same PRE-ORDER rule the pack lays it by. Find is then a -// binary search over the mapped bytes, in place, with nothing to parse. - -// cookMapsSignature is one record's extent writer. -func (g *tableGen) cookMapsSignature(st *ir.Struct) string { - return fmt.Sprintf("template inline bool %sCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const %s & value, TableByteOrder order )", st.Name, st.Name) -} - -// emitCookMaps emits one record's extent writer: every map reachable by value, -// PRE-ORDER, each entry through its own cook body. -func (g *tableGen) emitCookMaps(st *ir.Struct) { - g.pf("// %sCookMaps: %s's map arrays into the node's extent, PRE-ORDER, the entries\n", st.Name, st.Name) - g.pf("// in ASCENDING key order, each through its own cook body (§2.8, §7.6).\n") - g.pf("%s\n{\n", g.cookMapsSignature(st)) - if !g.hasMapExtent(st) { - g.pf(" (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order;\n") - g.pf(" return true; // no map below this record\n}\n\n") - return - } - g.pf(" (void) region; // a map's entries carry their own references through their own bodies\n") - ml := ir.RecordLayout(g.unit, st) - offsetOf := func(name string) int64 { - for i := range ml.Fields { - if ml.Fields[i].Field.Name == name { - return ml.Fields[i].Offset - } - } - return 0 - } - for _, f := range st.Fields { - if f.IsMap() { - entry := mapEntryOf(f) - el := ir.RecordLayout(g.unit, entry) - slot := offsetOf(f.Name) - g.pf(" { // %s\n", f.Name) - g.pf(" TableMapCursor<%s> cursor = TableMapOrder( ctx, value.%s );\n", entry.Name, f.Name) - g.pf(" if ( !cursor.ok ) { return false; }\n") - g.pf(" at = ( at + %d ) & ~(int64_t) %d; // at alignof( %s )\n", el.Align-1, el.Align-1, entry.Name) - g.pf(" uint8_t * array = extent + at;\n") - g.pf(" at += (int64_t) cursor.count * %d; // the whole array FIRST\n", el.Size) - g.pf(" // the SIXTEEN BYTES of the slot: the self-relative delta, then the count\n") - g.pf(" table_cook_put( record + %d, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + %d ) ) : 0, 8, order );\n", slot, slot) - g.pf(" table_cook_put( record + %d, (uint64_t) (uint32_t) cursor.count, 4, order );\n", slot+8) - g.pf(" for ( int32_t i = 0; i < cursor.count; i++ )\n {\n") - g.pf(" %s\n", g.cookBodyCall(entry, fmt.Sprintf("array + i * %d", el.Size), "*cursor[i]")) - g.pf(" }\n") - if g.hasMapExtent(entry) { - g.pf(" for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order\n {\n") - g.pf(" if ( !%sCookMaps( ctx, region, extent, at, array + i * %d, *cursor[i], order ) ) { TableMapRelease( cursor ); return false; }\n", entry.Name, el.Size) - g.pf(" }\n") - } - g.pf(" TableMapRelease( cursor );\n }\n") - continue - } - if g.edgeOf(f) != edgeNested { - continue - } - ref, _ := f.Type.Ref.(*ir.Struct) - if ref == nil || !g.hasMapExtent(ref) { - continue - } - nested := offsetOf(f.Name) - if f.Array == ir.ArrayNone { - g.pf(" if ( !%sCookMaps( ctx, region, extent, at, record + %d, value.%s, order ) ) { return false; } // %s\n", ref.Name, nested, f.Name, f.Name) - continue - } - stride := cookElementBytes(g.unit, f) - base := "value." + f.Name - bound := fmt.Sprintf("%d", f.ArrayBound) - if f.KeyEnum != "" && st.IsTable { - base += ".slots" - } - if f.Array == ir.ArrayCounted { - // THE LIVE COUNT, as the extent walk counts it: a slot past the - // count is storage the walk does not reach, and a non-empty map in - // one was already refused there (§7.6) - bound = fmt.Sprintf("( value.%s_count < %d ? value.%s_count : %d )", f.Name, f.ArrayBound, f.Name, f.ArrayBound) - } - g.pf(" for ( int32_t i = 0; i < %s; i++ ) // %s\n {\n", bound, f.Name) - g.pf(" if ( !%sCookMaps( ctx, region, extent, at, record + %d + i * %d, %s[i], order ) ) { return false; }\n", ref.Name, nested, stride, base) - g.pf(" }\n") - } - g.pf(" return true;\n}\n\n") -} - -// emitCookNode emits `CookNode`: one NODE's record and then its own extent. -// A nested record's writer is the body alone, because a nesting's maps are -// part of the HOLDER's extent and this walk already reached them. -func (g *tableGen) emitCookNode(st *ir.Struct) { - ml := ir.RecordLayout(g.unit, st) - record := cookAlignUp(ml.Size, ir.RegionAlignFloor) - g.pf("// %sCookNode: one node — the record, then the extent its maps take (§2.8).\n", st.Name) - g.pf("template inline bool %sCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const %s & value, TableByteOrder order )\n{\n", st.Name, st.Name) - if g.isVar(st.Name) { - g.pf(" if ( !%sCookBody( ctx, region, at, value, order ) ) { return false; }\n", st.Name) - } else { - g.pf(" %sCookBody( at, value, order );\n", st.Name) - } - g.pf(" int64_t extent_at = 0;\n") - g.pf(" return %sCookMaps( ctx, region, at + %d, extent_at, at, value, order );\n}\n\n", st.Name, record) -} - -// emitCookMapSurface emits the extent writer and the node writer for every -// closure member of a map-bearing unit. -func (g *tableGen) emitCookMapSurface(members []*ir.Struct) { - if !g.anyMap { - return - } - var bodies []*ir.Struct - for _, st := range members { - if ir.RecordLayout(g.unit, st) != nil { - bodies = append(bodies, st) - } - } - for _, st := range bodies { - g.pf("%s;\n", g.cookMapsSignature(st)) - } - g.pf("\n") - for _, st := range bodies { - g.emitCookMaps(st) - } - for _, st := range bodies { - g.emitCookNode(st) - } -} +// A map is a BY-VALUE EDGE of the ONE declaration-order walk: it is reached at +// its field's position, its entries are visited in ASCENDING KEY ORDER, and +// each entry's value is descended for the pointer slots inside it before the +// next entry is reached. A map declared before a pointer field therefore +// reaches a shared node FIRST and numbers it first, exactly as a union arm or +// a nested table declared there does. The rule is the walk's, not the map's. -// cookNodeBytes is one node's whole span in a cooked region: its record at the -// region's alignment floor, plus the extent its maps take (docs/SPEC-TABLES.md -// §2.8, §7.2). -func (g *tableGen) emitCookNodeBytes(st *ir.Struct, ind, expr, onBad string) { - ml := ir.RecordLayout(g.unit, st) - if !g.anyMap || !g.hasMapExtent(st) { - g.pf("%ssize = %d; node_align = %d;\n", ind, ml.Size, ml.Align) - return - } - g.pf("%s{\n", ind) - g.pf("%s const int64_t extent = %sMapExtent( ctx, %s );\n", ind, st.Name, expr) - g.pf("%s if ( extent < 0 ) { %s }\n", ind, onBad) - g.pf("%s size = %d + extent; node_align = %d;\n", ind, cookAlignUp(ml.Size, ir.RegionAlignFloor), ml.Align) - g.pf("%s}\n", ind) +// mapNumberEdge descends one map's entries for the NUMBERING walk. +func (g *tableGen) mapNumberEdge(f *ir.Field) { + g.emitMapEntryLoop(f, "value", "return false;", func(entry, elem, ind string) { + g.pf("%sif ( !%sNumber( ctx, numbering, %s ) ) { TableMapRelease( cursor_%s ); return false; }\n", ind, entry, elem, f.Name) + }) } -// onlyMapFields reports a record whose every field is a map — a cook body that -// writes the empty slots and reads nothing off the value, because the extent -// writer fills them. -func onlyMapFields(st *ir.Struct) bool { - for _, f := range st.Fields { - if !f.IsMap() { - return false - } - } - return len(st.Fields) > 0 +// mapPackMeasureEdge descends one map's entries for the PACK MEASURE. +func (g *tableGen) mapPackMeasureEdge(f *ir.Field) { + g.emitMapEntryLoop(f, "value", "return -1;", func(entry, elem, ind string) { + g.pf("%sint64_t inner = %sPackMeasure( ctx, seen, %s );\n", ind, entry, elem) + g.pf("%sif ( inner < 0 ) { TableMapRelease( cursor_%s ); return -1; }\n", ind, f.Name) + g.pf("%sbytes += inner;\n", ind) + }) } -// mapColumn is the four descriptor columns a MAP field carries -// (docs/SPEC-TABLES.md §2.8, §16): the generated entry's descriptor, and the -// three thunks the ONE text walk cannot spell for itself. Empty in a unit that -// declares no map, so a map-free unit's descriptors are what they always were. -func (g *tableGen) mapColumn(f *ir.Field) string { - if !g.anyMap { - return "" - } - if !f.IsMap() { - return "NULL, NULL, NULL, NULL, " - } +// mapPackEdge descends one map's entries for the PACK, against the array +// MapPack already placed in the node's extent. +func (g *tableGen) mapPackEdge(f *ir.Field) { entry := mapEntryOf(f) - n := entry.Name - hold := fmt.Sprintf("TableMap<%s>", n) - count := fmt.Sprintf("[]( const void * slot ) -> int32_t { return ( (const %s *) slot )->count; }", hold) - at := fmt.Sprintf("[]( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const %s *) slot )->Entries() + index ); }", hold) - var insert string - if mapKeyIsString(f) { - insert = fmt.Sprintf("[]( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * "+ - "{ if ( key == NULL || key_length > k%sKeyBound ) { return NULL; } "+ // KEYS NEVER CLAMP - "%s * placed = TableMapPlace( worker, *(%s *) slot, key ); "+ - "if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }", n, n, hold) - } else { - typ, _ := g.cppFieldType(ir.MapKeyField(f).Type) - insert = fmt.Sprintf("[]( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * "+ - "{ %s * placed = TableMapPlace( worker, *(%s *) slot, (%s) key_value ); "+ - "if ( placed != NULL ) { TableEntrySetKey( *placed, (%s) key_value ); } return (void *) placed; }", n, hold, typ, typ) - } - return fmt.Sprintf("&%sTableInfo, %s, %s, %s, ", n, count, at, insert) -} - -// nodeStorageBody, nodeStorageArg and nodeStorageReader are the ONE extra -// parameter a node's storage takes where a map rides in an extent -// (docs/SPEC-TABLES.md §2.8): the record's body, from which the framing scan -// sums the entry arrays. A root that can name no such record does not take it, -// so a map-free unit's dispatch is the one it always emitted. -func (g *tableGen) nodeStorageBody(anyExtent bool) string { - if anyExtent { - return "const uint8_t * body, " - } - return "" -} - -func (g *tableGen) nodeStorageArg(root *ir.Struct) string { - if g.rootHasExtent(root) { - return "body, " - } - return "" -} - -func (g *tableGen) nodeStorageReader(root *ir.Struct) string { - if g.rootHasExtent(root) { - return "r.buffer, " - } - return "" + g.emitMapEntryLoopHead(f, "src", "return false;") + g.pf(" %s * placed_%s = (%s *) ( dst.%s.entries.value != 0 ? ( (uint8_t *) &dst.%s.entries + dst.%s.entries.value ) : NULL );\n", + entry.Name, f.Name, entry.Name, f.Name, f.Name, f.Name) + g.pf(" for ( int32_t i = 0; i < cursor_%s.count; i++ )\n {\n", f.Name) + g.pf(" if ( !%sPackEdges( ctx, seen, *cursor_%s[i], placed_%s[i], base, capacity, used ) ) { TableMapRelease( cursor_%s ); return false; }\n", + entry.Name, f.Name, f.Name, f.Name) + g.pf(" }\n") + g.pf(" TableMapRelease( cursor_%s );\n }\n", f.Name) } -// rootHasExtent reports whether any record one root's numbering can name holds -// a map by value — which is what decides both halves of the signature above. -func (g *tableGen) rootHasExtent(root *ir.Struct) bool { - if !g.anyMap { - return false - } - return slices.ContainsFunc(g.pointerReachable(root), g.hasMapExtent) +// emitMapEntryLoopHead opens one map's sorted cursor over the given subject. +func (g *tableGen) emitMapEntryLoopHead(f *ir.Field, subject, onBad string) { + entry := mapEntryOf(f) + g.pf(" { // %s: a by-value edge, entries in ASCENDING key order (§2.8, §3.1)\n", f.Name) + g.pf(" TableMapCursor<%s> cursor_%s = TableMapOrder( ctx, %s.%s );\n", entry.Name, f.Name, subject, f.Name) + g.pf(" if ( !cursor_%s.ok ) { %s }\n", f.Name, onBad) } -// emitUnreachedMapRefusal refuses an UNREACHED NON-EMPTY MAP SLOT, the same -// refusal §7.6 gives a pointer in that position (docs/SPEC-TABLES.md §2.8): a -// COUNTED array's slots past its live count are storage the walk does not -// reach, so a non-empty map in one names entries the region will not hold, and -// the write answers false with nothing partial written. -// -// The test is the extent itself: an empty map takes no bytes and advances the -// running offset by none, so a record whose extent measures ZERO is a record -// whose every by-value map is empty. A measure that refuses answers non-zero -// here too, and refusing on it is the same answer one level up. -func (g *tableGen) emitUnreachedMapRefusal(f *ir.Field, ref *ir.Struct, subject string) { - if f.Array != ir.ArrayCounted { - return // every other array shape is reached whole - } - g.pf(" for ( int32_t i = %s.%s_count; i < %d; i++ ) // %s: the slots the walk does not reach (§7.6)\n {\n", - subject, f.Name, f.ArrayBound, f.Name) - g.pf(" if ( !TableMapUnreachedEmpty( %sMapExtent( ctx, %s.%s[i] ) ) ) { return false; }\n", ref.Name, subject, f.Name) - g.pf(" }\n") +// emitMapEntryLoop is the whole shape: the cursor, the loop, the release. +func (g *tableGen) emitMapEntryLoop(f *ir.Field, subject, onBad string, body func(entry, elem, ind string)) { + entry := mapEntryOf(f) + g.emitMapEntryLoopHead(f, subject, onBad) + g.pf(" for ( int32_t i = 0; i < cursor_%s.count; i++ )\n {\n", f.Name) + body(entry.Name, fmt.Sprintf("*cursor_%s[i]", f.Name), " ") + g.pf(" }\n") + g.pf(" TableMapRelease( cursor_%s );\n }\n", f.Name) } diff --git a/internal/codegen/cpptable/pointers.go b/internal/codegen/cpptable/pointers.go index 7954411f0..aa8086c50 100644 --- a/internal/codegen/cpptable/pointers.go +++ b/internal/codegen/cpptable/pointers.go @@ -163,6 +163,11 @@ type edgeVisitor struct { // and because a map's entries are not a path under the write subject: the // pack's twin is the array it already placed in the node's extent. mapField func(f *ir.Field) + // listField is one `[]T` whose elements the walk descends, in INDEX order + // (docs/SPEC-TABLES.md §2.9). It takes the visitor itself, because a list's + // element is the pointer slot or the nested table the visitor already + // knows how to reach. The list adds only the cursor around them. + listField func(f *ir.Field, v edgeVisitor) } // at spells one storage PATH — ".field", ".field[i]", ".body.chunk" — under @@ -190,9 +195,25 @@ const ( // A map whose entry reaches nothing is not an edge — its extent is the // extent walk's, not this one's. edgeMap + // edgeList is a `[]T` whose ELEMENTS reach a node: a `[]*T`, whose elements + // ARE the pointer slots, or a `[]T` over a variable table with pointers + // inside it. The walk descends the elements in INDEX ORDER, the order the + // wire carries and the order a region holds (docs/SPEC-TABLES.md §2.9). A + // list whose elements reach nothing is not an edge: its extent is the + // extent walk's, not this one's. + edgeList ) func (g *tableGen) edgeOf(f *ir.Field) edgeKind { + if f.IsList() { + if listElementIsPointer(f) { + return edgeList + } + if ref := listElementStruct(f); ref != nil && g.isVar(ref.Name) { + return edgeList + } + return edgeNone + } if f.IsMap() { if g.noVariableEdges(f.MapEntry) { return edgeNone @@ -304,6 +325,8 @@ func (g *tableGen) emitEdgeOf(f *ir.Field, v edgeVisitor) { g.emitVariableUnionWalk(f, v) case edgeMap: v.mapField(f) + case edgeList: + v.listField(f, v) } } @@ -456,8 +479,9 @@ func (g *tableGen) emitVariableSurface(members []*ir.Struct) { if !g.anyVariable { return } - g.emitMapWalkSurface(members) + g.emitExtentWalkSurface(members) g.emitMapBuilderSurfaces(members) + g.emitListBuilderSurfaces(members) for _, st := range g.varMembers(members) { g.owner = st g.emitNumber(st) @@ -551,7 +575,8 @@ func (g *tableGen) emitNumber(st *ir.Struct) { descend: func(table string, expr edgeExpr, indent string) { g.pf("%sif ( !%sNumber( ctx, numbering, %s ) ) { return false; }\n", indent, table, expr.Src) }, - mapField: g.mapNumberEdge, + mapField: g.mapNumberEdge, + listField: func(f *ir.Field, v edgeVisitor) { g.emitListEdge(f, v, "return false;") }, }) g.pf(" return true;\n}\n\n") } @@ -614,7 +639,8 @@ func (g *tableGen) emitPackMeasure(st *ir.Struct) { g.pf("%sif ( inner < 0 ) { return -1; }\n", indent) g.pf("%sbytes += inner;\n", indent) }, - mapField: g.mapPackMeasureEdge, + mapField: g.mapPackMeasureEdge, + listField: func(f *ir.Field, v edgeVisitor) { g.emitListEdge(f, v, "return -1;") }, }) g.pf(" return bytes;\n}\n\n") } @@ -632,7 +658,7 @@ func (g *tableGen) emitPack(st *ir.Struct) { g.pf("// required sign (§6.3), and sharing and a back-reference are one fact. A\n") g.pf("// reference to a node whose descent is still OPEN is a cycle, and this\n") g.pf("// refuses it rather than packing one.\n") - if g.anyMap { + if g.anyExtent { // THE NODE'S EXTENT IS CARVED BEFORE ANY POINTEE IS PLACED // (docs/SPEC-TABLES.md §2.8, §6.3): a node's extent runs to the next // directory entry, so a pointee laid between two of its map arrays @@ -645,7 +671,7 @@ func (g *tableGen) emitPack(st *ir.Struct) { g.pf(" int64_t at = 0;\n") g.pf(" uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( %s ) );\n", st.Name) g.pf(" const int64_t room = capacity - ( (int64_t) ( extent - base ) );\n") - g.pf(" if ( !%sMapPack( ctx, src, dst, extent, at, room ) ) { return false; }\n", st.Name) + g.pf(" if ( !%sExtentPack( ctx, src, dst, extent, at, room ) ) { return false; }\n", st.Name) g.pf(" return %sPackEdges( ctx, seen, src, dst, base, capacity, used );\n}\n\n", st.Name) g.pf("template \ninline bool %sPackEdges( const Ctx & ctx, TablePackMap & seen, const %s & src, %s & dst, uint8_t * base, int64_t capacity, int64_t & used )\n{\n", st.Name, st.Name, st.Name) if g.noVariableEdges(st) { @@ -697,12 +723,13 @@ func (g *tableGen) emitPack(st *ir.Struct) { blob: g.emitPackBlobField, descend: func(table string, expr edgeExpr, indent string) { call := "Pack" - if g.anyMap { + if g.anyExtent { call = "PackEdges" // the extent walk already reached this nesting } g.pf("%sif ( !%s%s( ctx, seen, %s, %s, base, capacity, used ) ) { return false; }\n", indent, table, call, expr.Src, expr.Dst) }, - mapField: g.mapPackEdge, + mapField: g.mapPackEdge, + listField: func(f *ir.Field, v edgeVisitor) { g.emitListEdge(f, v, "return false;") }, }) g.pf(" return true;\n}\n\n") } @@ -811,8 +838,8 @@ func (g *tableGen) emitBuilderAndPublicSurface(st *ir.Struct) { g.pf(" below = %sPackMeasure( ctx, seen, root );\n", n) g.pf(" }\n") g.pf(" if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it\n") - if g.anyMap { - g.pf(" int64_t root_extent = %sMapExtent( ctx, root );\n", n) + if g.anyExtent { + g.pf(" int64_t root_extent = %sExtent( ctx, root );\n", n) g.pf(" if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run\n") g.pf(" int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + root_extent ) + below;\n", n) } else { @@ -823,7 +850,7 @@ func (g *tableGen) emitBuilderAndPublicSurface(st *ir.Struct) { g.pf(" // allocator's contract: a packed region carries node padding.\n") g.pf(" uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total );\n") g.pf(" if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; }\n") - if g.anyMap { + if g.anyExtent { g.pf(" int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + root_extent );\n", n) } else { g.pf(" int64_t used = TableAlignUp64( (int64_t) sizeof( %s ) );\n", n) @@ -1007,8 +1034,8 @@ func (g *tableGen) emitBuilderAndPublicSurface(st *ir.Struct) { g.pf(" nodes.entries = directory;\n") g.pf(" nodes.count = records + 1;\n") g.pf(" nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here\n") - if g.anyMap { - g.pf(" nodes.worker = &builder.main; // and a map's entries are the arena's, not a node extent's (§2.8)\n") + if g.anyExtent { + g.pf(" nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9)\n") } g.pf(" {\n") g.pf(" TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table );\n") @@ -1040,12 +1067,18 @@ func (g *tableGen) emitBuilderAndPublicSurface(st *ir.Struct) { g.pf(" }\n }\n") g.pf(" TableReader r( wire, wire_bytes, out, &ids_table );\n") g.pf(" r.nested = false; // the ROOT body, the one that carries the node table\n") - if g.anyMap { - g.pf(" TableMapCarve root_carve;\n") + if g.anyExtent { + g.pf(" TableExtentCarve root_carve;\n") g.pf(" root_carve.worker = &builder.main;\n") g.pf(" nodes.carve = &root_carve;\n") } g.pf(" bool ok = %sLoadBody( r, nodes, *root );\n", n) + if g.anyExtent { + g.pf(" // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md\n") + g.pf(" // §2.9): the partial builder is the caller's to discard, and the report\n") + g.pf(" // holds what it held when the count was met\n") + g.pf(" ok = ok && !nodes.refused;\n") + } g.pf(" allocator.free( allocator.context, directory );\n") g.pf(" return ok;\n}\n\n") } @@ -1066,7 +1099,7 @@ func (g *tableGen) emitRootNodeDispatch(st *ir.Struct) { g.pf("// this build cannot name — which keeps its index and reads null. A BYTE\n") g.pf("// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5),\n") g.pf("// which is the one answer the record's LENGTH decides.\n") - if g.anyMap { + if g.anyExtent { g.pf("// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8),\n") g.pf("// so a record's storage is its type's PLUS N x sizeof( Entry ) at every\n") g.pf("// depth, summed from the FRAMING: N is framing and not a value, and this\n") @@ -1074,23 +1107,23 @@ func (g *tableGen) emitRootNodeDispatch(st *ir.Struct) { } anyExtent := false for _, t := range reachable { - if g.anyMap && g.hasMapExtent(t) { + if g.anyExtent && g.hasExtent(t) { anyExtent = true } } // THE BODY IS ONLY A PARAMETER WHERE A MAP RIDES IN AN EXTENT: a root that // can name no such record answers from the type id and the length, exactly // as it did before the construct existed. - g.pf("inline int64_t %sNodeStorage( uint64_t type_id, %sint64_t length )\n{\n", n, g.nodeStorageBody(anyExtent)) + g.pf("inline int64_t %sNodeStorage( uint64_t type_id, %sint64_t length%s )\n{\n", n, g.nodeStorageBody(anyExtent), g.nodeStorageTail(anyExtent)) if len(blobs) == 0 { g.pf(" (void) length; // no byte buffer below this root: every node's storage is its type's\n") } g.pf(" switch ( type_id )\n {\n") for _, t := range reachable { - if g.anyMap && g.hasMapExtent(t) { + if g.anyExtent && g.hasExtent(t) { g.pf(" case 0x%016xull: // %s\n {\n", ir.TableWireId(t.Name), t.Name) g.pf(" int64_t extent = 0;\n") - g.pf(" if ( !%sWireExtent( body, length, extent ) ) { return kTableNodeRefused; }\n", t.Name) + g.pf(" if ( !%sWireExtent( body, length, extent, ids, reason ) ) { return kTableNodeRefused; }\n", t.Name) g.pf(" return TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + extent );\n }\n", t.Name) continue } @@ -1124,7 +1157,7 @@ func (g *tableGen) emitRootNodeDispatch(st *ir.Struct) { } g.pf(" default: break;\n }\n}\n\n") - if g.anyMap { + if g.anyExtent { g.pf("// %sNodeRecordBytes: one record's OWN storage, before the extent its maps\n", n) g.pf("// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins.\n") g.pf("inline int64_t %sNodeRecordBytes( uint64_t type_id )\n{\n", n) @@ -1163,14 +1196,18 @@ func (g *tableGen) emitRootNodeDispatch(st *ir.Struct) { g.pf("// %sNodeBody: PASS TWO's half — decode one record's body into the storage it\n", n) g.pf("// already owns.\n") g.pf("inline void %sNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at )\n{\n", n) - if g.anyMap { - g.pf(" // the node's own EXTENT, where its maps' entry arrays are carved from,\n") - g.pf(" // PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8). The tool's\n") - g.pf(" // path carries a worker instead: there the entries are the arena's.\n") - g.pf(" TableMapCarve carve;\n") + if g.anyExtent { + g.pf(" // the node's own EXTENT, where its lists' and maps' arrays are carved\n") + g.pf(" // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9).\n") + g.pf(" // The tool's path carries a worker instead: there the arrays are the\n") + g.pf(" // arena's.\n") + g.pf(" TableExtentCarve carve;\n") g.pf(" carve.worker = nodes.worker;\n") g.pf(" if ( carve.worker == NULL )\n {\n") - g.pf(" const int64_t storage = %sNodeStorage( type_id, %sr.size );\n", n, g.nodeStorageReader(st)) + if g.rootHasExtent(st) { + g.pf(" TableRefuseReason reason = count_over_length; // pass one already refused what this could refuse\n") + } + g.pf(" const int64_t storage = %sNodeStorage( type_id, %sr.size%s );\n", n, g.nodeStorageReader(st), g.nodeStorageReaderTail(st)) g.pf(" const int64_t record = storage > 0 ? %sNodeRecordBytes( type_id ) : 0;\n", n) g.pf(" carve.at = at + record;\n") g.pf(" carve.left = storage > record ? storage - record : 0;\n") @@ -1199,7 +1236,7 @@ func (g *tableGen) emitRootNodeDispatch(st *ir.Struct) { g.pf(" case %s: if ( r.size > 0 ) { memcpy( at + kTableBlobHeader, r.buffer, (size_t) r.size ); } break; // *%s\n", b.constant, b.word) } g.pf(" default: break;\n }\n") - if g.anyMap { + if g.anyExtent { g.pf(" nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done\n") } g.pf("}\n\n") @@ -1452,14 +1489,14 @@ func (g *tableGen) emitVariableLoadMeasure(st *ir.Struct, message bool) { // the connection's rather than the wire's, so there is no trailer to // locate and no stray-byte rule between a terminator and a first // entry — the message's last byte IS the body's terminator. - g.pf("inline int64_t %sLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL )\n{\n", n) + g.pf("inline int64_t %sLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL%s )\n{\n", n, g.loadMeasureReasonParam()) g.pf(" TableReport ignored;\n") g.pf(" if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; }\n") g.pf(" const TableIdTable & ids_table = vocabulary.table;\n") g.pf(" const uint8_t * const wire = message + 1;\n") g.pf(" const int64_t wire_bytes = message_bytes - 1;\n") } else { - g.pf("inline int64_t %sLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL )\n{\n", n) + g.pf("inline int64_t %sLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL%s )\n{\n", n, g.loadMeasureReasonParam()) g.pf(" TableReport ignored;\n") g.pf(" TableIdTable ids_table;\n") g.pf(" int64_t body_bytes = 0;\n") @@ -1472,16 +1509,23 @@ func (g *tableGen) emitVariableLoadMeasure(st *ir.Struct, message bool) { g.pf(" const int64_t wire_bytes = body_bytes;\n") } g.pf(" TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table );\n") - g.emitRootDataBytes(st, " ", "return -1;") + refuse := "return -1;" + if g.anyExtent { + // A -1 CARRIES A REASON (docs/SPEC-TABLES.md §6.5), as an enum + // out-parameter, and a refusal moves no counter + g.pf(" TableRefuseReason reason = count_over_length;\n") + refuse = "if ( reason_out != NULL ) { *reason_out = reason; } return -1;" + } + g.emitRootDataBytes(st, " ", refuse) g.pf(" int64_t records = 0;\n") g.pf(" uint64_t type_id = 0;\n") g.pf(" const uint8_t * body = NULL;\n") g.pf(" int64_t length = 0;\n") g.pf(" while ( TableNodeScanNext( scan, type_id, body, length ) )\n {\n") g.pf(" records++;\n") - g.pf(" int64_t storage = %sNodeStorage( type_id, %slength );\n", n, g.nodeStorageArg(st)) - if g.anyMap { - g.pf(" if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8)\n") + g.pf(" int64_t storage = %sNodeStorage( type_id, %slength%s );\n", n, g.nodeStorageArg(st), g.nodeStorageArgTail(st)) + if g.anyExtent { + g.pf(" if ( storage == kTableNodeRefused ) { %s } // an N the record's framing cannot carry (§2.8, §2.9)\n", refuse) } g.pf(" if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none\n") g.pf(" }\n") @@ -1544,6 +1588,9 @@ func (g *tableGen) emitVariableLoad(st *ir.Struct, message bool) { g.pf(" const uint8_t * body = NULL;\n") g.pf(" int64_t length = 0;\n\n") g.pf(" // the record count and the data bytes, from the FRAMING alone\n") + if g.anyExtent { + g.pf(" TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed\n") + } g.emitRootDataBytes(st, " ", "out->malformed = true; return NULL;") g.pf(" int64_t records = 0;\n") g.pf(" {\n") @@ -1551,8 +1598,8 @@ func (g *tableGen) emitVariableLoad(st *ir.Struct, message bool) { g.pf(" TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table );\n") g.pf(" while ( TableNodeScanNext( scan, type_id, body, length ) )\n {\n") g.pf(" records++;\n") - g.pf(" int64_t storage = %sNodeStorage( type_id, %slength );\n", n, g.nodeStorageArg(st)) - if g.anyMap { + g.pf(" int64_t storage = %sNodeStorage( type_id, %slength%s );\n", n, g.nodeStorageArg(st), g.nodeStorageArgTail(st)) + if g.anyExtent { g.pf(" if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; }\n") } g.pf(" if ( storage > 0 ) { data += storage; }\n") @@ -1572,7 +1619,7 @@ func (g *tableGen) emitVariableLoad(st *ir.Struct, message bool) { g.pf(" // resolves whichever way it points. It reads no body.\n") g.pf(" {\n") g.pf(" TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table );\n") - if g.anyMap { + if g.anyExtent { g.pf(" int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + root_extent );\n", n) } else { g.pf(" int64_t used = TableAlignUp64( (int64_t) sizeof( %s ) );\n", n) @@ -1580,8 +1627,9 @@ func (g *tableGen) emitVariableLoad(st *ir.Struct, message bool) { g.pf(" int64_t k = 0;\n") g.pf(" int32_t unknown_records = 0; // counted once the scan is known whole\n") g.pf(" while ( TableNodeScanNext( scan, type_id, body, length ) )\n {\n") - g.pf(" int64_t storage = %sNodeStorage( type_id, %slength );\n", n, g.nodeStorageArg(st)) + g.pf(" int64_t storage = %sNodeStorage( type_id, %slength%s );\n", n, g.nodeStorageArg(st), g.nodeStorageArgTail(st)) g.pf(" if ( storage <= 0 )\n {\n") + g.pf(" // a record whose type id this build cannot name KEEPS ITS\n") g.pf(" // INDEX, is counted once here and not once per pointer, and\n") g.pf(" // every reference to it reads null (§3.1)\n") @@ -1618,8 +1666,8 @@ func (g *tableGen) emitVariableLoad(st *ir.Struct, message bool) { g.pf(" // against a numbering already known good or already known bad\n") g.pf(" TableReader r( wire, wire_bytes, out, &ids_table );\n") g.pf(" r.nested = false; // the ROOT body, the one that carries the node table\n") - if g.anyMap { - g.pf(" TableMapCarve root_carve;\n") + if g.anyExtent { + g.pf(" TableExtentCarve root_carve;\n") g.pf(" root_carve.at = region + TableAlignUp64( (int64_t) sizeof( %s ) );\n", n) g.pf(" root_carve.left = root_extent;\n") g.pf(" nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's\n") @@ -1627,3 +1675,13 @@ func (g *tableGen) emitVariableLoad(st *ir.Struct, message bool) { g.pf(" %sLoadBody( r, nodes, *root );\n", n) g.pf(" return root;\n}\n\n") } + +// loadMeasureReasonParam is the out-parameter a LoadMeasure takes where a unit +// has an extent (docs/SPEC-TABLES.md §6.5): the reason a -1 carries. A unit +// with neither a list nor a map keeps the signature it always had. +func (g *tableGen) loadMeasureReasonParam() string { + if g.anyExtent { + return ", TableRefuseReason * reason_out = NULL" + } + return "" +} diff --git a/internal/tablecook/check.go b/internal/tablecook/check.go index 82a6e982e..e69a4c7bb 100644 --- a/internal/tablecook/check.go +++ b/internal/tablecook/check.go @@ -142,6 +142,19 @@ type scan struct { buf []byte dir []DirectoryEntry pointers int + // THE NODE UNDER THE SCAN, for §7.4's element-array clause: an unbounded + // array's slot must point inside its holder's own extent, so the walk + // carries where that node begins and ends, and the arrays it has already + // placed there, so no two overlap. Both are reset per node. + base int64 + extent int64 + arrays []arrayRange +} + +// arrayRange is one element array a list slot placed inside the node under +// the scan, in region offsets. +type arrayRange struct { + start, end int64 } // node walks one directory entry. A BYTE BUFFER's node has no fields to walk @@ -160,6 +173,7 @@ func (s *scan) node(base, extent int64, typeId uint64, st *ir.Struct) error { } return nil } + s.base, s.extent, s.arrays = base, base+extent, s.arrays[:0] return s.record(base, st) } @@ -198,10 +212,55 @@ func (s *scan) field(at int64, f *ir.Field) error { return err } return s.companion(pieces[1].Offset, f.ArrayBound, "used count") + case f.Array == ir.ArrayList: + return s.list(value.Offset, f) } return s.element(value.Offset, f) } +// list is §7.4's ELEMENT-ARRAY clause (docs/SPEC-TABLES.md §2.9, §7.4): an +// unbounded array's sixteen-byte slot holds an int64 self-relative delta to +// its element array and an int32 count, and the array must sit INSIDE THE +// HOLDER'S OWN EXTENT, meaning containment, alignment, fit, and no overlap with any +// other array already placed in that node, before the elements' own slots, +// companions and tags are walked as a bounded array's are. There is no fifth +// clause, because there are no keys and no order. The check reads those four +// facts and not the offset the layout rule computes, so the layout rule stays +// independent of the check exactly as the pack order does. +func (s *scan) list(at int64, f *ir.Field) error { + delta := int64(s.ord.Uint64(s.buf[at:])) + count := int64(int32(s.ord.Uint32(s.buf[at+8:]))) + if count < 0 { + return fmt.Errorf("the count is %d, and an extent is never negative", count) + } + if delta == RefNull { + if count != 0 { + return fmt.Errorf("the reference is null and the count is %d: an empty list is the only list a null names", count) + } + return nil + } + if count == 0 { + return fmt.Errorf("the count is 0 and the reference is not null: an empty list's reference is null in every encoding") + } + size, align := ir.ListElementLayout(s.m.Unit, f) + start := at + delta + end := start + count*size + if start < s.base || end > s.extent { + return fmt.Errorf("the element array runs [%d, %d) and its holder's extent is [%d, %d): the array leaves the node", start, end, s.base, s.extent) + } + if start%align != 0 { + return fmt.Errorf("the element array starts at %d, which is not aligned to %d", start, align) + } + for _, other := range s.arrays { + if start < other.end && other.start < end { + return fmt.Errorf("the element array [%d, %d) overlaps another array [%d, %d) in the same node", start, end, other.start, other.end) + } + } + s.arrays = append(s.arrays, arrayRange{start, end}) + return s.slots(start, f, count) +} + + // companion checks one count companion against its DECLARED bound. A negative // one is refused too: a count is an extent and an extent is never negative, and // a walker handed one indexes backwards out of the region. diff --git a/internal/tablecook/list_test.go b/internal/tablecook/list_test.go new file mode 100644 index 000000000..17b391af3 --- /dev/null +++ b/internal/tablecook/list_test.go @@ -0,0 +1,90 @@ +package tablecook_test + +import ( + "encoding/binary" + "strings" + "testing" + + "github.com/mas-bandwidth/schema/v2/internal/tablecook" + "github.com/mas-bandwidth/schema/v2/internal/tabletext" + "github.com/mas-bandwidth/schema/v2/ir" +) + +// §7.4's ELEMENT-ARRAY CLAUSE (docs/SPEC-TABLES.md §2.9, §7.4): an unbounded +// array's slot must point its array inside the holder's own extent, aligned, +// fitting, and overlapping no other array in that node. The tool cannot cook a +// list yet, so the cook under test is assembled BY HAND from §7.1 and §7.2: +// one root node, `Ints` from tables/lists, whose sixteen-byte slot names a +// three-element array laid after the record. + +// intsCook writes a cook of one Ints root with the given slot and count. The +// record is 24 bytes (the slot, then `after` and its padding), the array of +// three int32 follows at 24, and the data part rounds to 40. +func intsCook(u *ir.Unit, delta int64, count int32) []byte { + const header, data, attrib = int64(64), int64(40), int64(16) + out := make([]byte, header+data+attrib) + le := binary.LittleEndian + le.PutUint64(out[0:], tablecook.Magic) + le.PutUint64(out[8:], ir.BuildVersion(u)) + le.PutUint64(out[16:], tablecook.ByteOrderLittle) + le.PutUint64(out[24:], uint64(data)) + le.PutUint64(out[32:], uint64(attrib)) + le.PutUint64(out[40:], 8) + record := out[header:] + le.PutUint64(record[0:], uint64(delta)) + le.PutUint32(record[8:], uint32(count)) + le.PutUint32(record[16:], 5) // after + for i := 0; i < 3; i++ { + le.PutUint32(record[24+i*4:], uint32(10*(i+1))) + } + dir := out[header+data:] + le.PutUint64(dir[0:], 0) + le.PutUint64(dir[8:], ir.TableTypeId("Ints")) + return out +} + +// TestCookCheckListSlot: the clause reads CONTAINMENT, ALIGNMENT, FIT and NO +// OVERLAP, and nothing else, and a null reference is an empty list and only +// that. The Makefile's negative control drops the containment test through an +// overlay and requires this test to go red. +func TestCookCheckListSlot(t *testing.T) { + u := unit(t, "../../tables/lists") + m := tabletext.NewModel(u) + + res, err := tablecook.Check(m, intsCook(u, 24, 3)) + if err != nil { + t.Fatalf("a cook whose list slot names its array inside the node was refused: %v", err) + } + if res.Root != "Ints" || res.Nodes != 1 { + t.Fatalf("checked the wrong shape: %+v", res) + } + if _, err := tablecook.Check(m, intsCook(u, 0, 0)); err != nil { + t.Fatalf("an empty list, null reference and zero count, was refused: %v", err) + } + + cases := []struct { + name string + delta int64 + count int32 + want string + }{ + {"the array leaves the node", 40, 3, "leaves the node"}, + {"the array leaves the region", 4096, 3, "leaves the node"}, + {"the array is not aligned", 26, 3, "not aligned"}, + {"the count does not fit", 24, 5, "leaves the node"}, + {"the count is negative", 24, -1, "never negative"}, + {"a null reference with a count", 0, 3, "empty list"}, + {"a reference with no count", 24, 0, "reference is not null"}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + _, err := tablecook.Check(m, intsCook(u, c.delta, c.count)) + if err == nil { + t.Fatalf("FAILED: cook-check accepted a list slot that %s", c.name) + } + if !strings.Contains(err.Error(), c.want) { + t.Fatalf("refused, but not on the element-array clause: %v", err) + } + }) + } +} diff --git a/ir/tablelist.go b/ir/tablelist.go index 243cad97c..8dbbcd856 100644 --- a/ir/tablelist.go +++ b/ir/tablelist.go @@ -52,3 +52,18 @@ func ListFields(u *Unit) []string { func (f *Field) CountedOnWire() bool { return f != nil && (f.Array == ArrayCounted || f.Array == ArrayList) } + +// ListElementLayout is the storage one element of a `[]T` takes and the +// alignment its array is laid at (docs/SPEC-TABLES.md §2.9): a TableRef for a +// `[]*T`, and the element type's own size and alignment otherwise. The element +// array in a holder's node extent is `count × size` at `align`, and this is +// the one place both the C++ writer and the tool's cook-check take the two +// numbers from. +func ListElementLayout(u *Unit, f *Field) (size, align int64) { + single := *f + single.Array = ArrayNone + single.ArrayBound = 0 + single.Type.Optional = false + p := elementPiece(u, &single) + return p.size, p.align +} diff --git a/tables/lists/Holders.schema b/tables/lists/Holders.schema new file mode 100644 index 000000000..2d35058af --- /dev/null +++ b/tables/lists/Holders.schema @@ -0,0 +1,44 @@ +package listdemo + +// WHERE ELSE A LIST RIDES IN A HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.9): a +// list of tables that hold lists, a list whose element holds a map, and a +// pointed-at node that holds a list of its own, so the node's storage is its +// record plus the extent its list commands. Lists and maps are ONE population +// in the extent, in declaration order, pre-order. + +table Sample { v int32 } + +table Row +{ + items []Sample + label int32 +} + +table Sheet +{ + rows []Row // a list of tables that hold lists: LoadMeasure sums at every depth + pinned *Row // a pointed-at holder: the node's extent is its list's +} + +table Item { count int32 } + +table Squad +{ + roster map[uint8]Item + name int32 +} + +table Army +{ + squads []Squad // an element that holds a MAP: the element array first, then each map + after int32 +} + +// AN UNREACHED NON-EMPTY LIST SLOT IS REFUSED by Cook and by Lock (§7.6): a +// counted array's slots past its live count are storage the walk does not +// reach, so a list with elements in one names storage the region will not hold +table Deck +{ + hands [..3]Row + after int32 +} diff --git a/tables/lists/Migrate.schema b/tables/lists/Migrate.schema new file mode 100644 index 000000000..c0255f356 --- /dev/null +++ b/tables/lists/Migrate.schema @@ -0,0 +1,21 @@ +package listdemo + +// THE MIGRATION ITSELF (docs/SPEC-TABLES.md §2.9): "the same bytes" is a claim +// about two schemas, so ONE content is declared twice, once at a bound and +// once unbounded, and one pinned wire is what both write byte for byte and +// both read into equal values. The bound is ABOVE the instance's count, so the +// row proves the framing and not the clamp. + +table Unit { v int32 } + +table Bounded +{ + items [..8]Unit + tag int32 +} + +table Unbounded +{ + items []Unit + tag int32 +} diff --git a/tables/lists/Report.schema b/tables/lists/Report.schema new file mode 100644 index 000000000..95f863e63 --- /dev/null +++ b/tables/lists/Report.schema @@ -0,0 +1,25 @@ +package listdemo + +// THE REPORT ROWS' ROOTS (docs/SPEC-TABLES.md §2.9, §4): a []uint8 that +// carries 100,000 elements, past 2^16 and so past any bound a schema on the +// page declares, so a reader that CLAMPS the count goes red; and two roots +// that differ only in the ELEMENT's kind, read one against the other, with a +// field AFTER the list so the parent can be seen to read on. + +table Bytes +{ + data []uint8 + after int32 +} + +table Ints +{ + values []int32 + after int32 +} + +table Floats +{ + values []float32 + after int32 +} diff --git a/tables/lists/Save.schema b/tables/lists/Save.schema new file mode 100644 index 000000000..c8cc3bd5b --- /dev/null +++ b/tables/lists/Save.schema @@ -0,0 +1,48 @@ +package listdemo + +// The UNBOUNDED ARRAY corpus (docs/SPEC-TABLES.md §2.9): §2.9's own example, +// the five element classes, and the report rows' roots. A list is a counted +// array whose count the DATA decides, and every instance the harness pins over +// these crosses the wire, the text and the cook. + +table Placement +{ + x float32 + y float32 + model uint32 +} + +table LogEntry { tick uint32 } + +table Save +{ + placements []Placement // as many as the world has + log []*LogEntry // pointer elements, two slots may name one node + scores []int32 // a scalar element is an element like any other +} + +enum Grade { A, B, C } + +flags Perm { Read, Write, Own } + +table Point +{ + x int32 + y int32 +} + +union Hit +{ + point Point + damage int32 +} + +// the element set is [..N]T's exactly: an enum, a flags mask and a union are +// elements as they are in a bounded array +table Mixed +{ + grades []Grade + perms []Perm + hits []Hit + bounds []int32 | min = 0, max = 100 // a bar attribute qualifies the ELEMENT, as on a [..N]T +} diff --git a/tables/lists/Shared.schema b/tables/lists/Shared.schema new file mode 100644 index 000000000..674812f46 --- /dev/null +++ b/tables/lists/Shared.schema @@ -0,0 +1,19 @@ +package listdemo + +// SHARING AND THE WALK ORDER (docs/SPEC-TABLES.md §2.9, §3.1): a []*T's +// elements are pointer slots, so two slots may name one node beside a null +// one, and a []*T DECLARED BEFORE a pointer field reaches a shared node first +// and numbers it first. A walk that grouped lists after the pointer fields +// would number the two the other way round and the pinned wire says so. + +table Photo +{ + width uint32 + height uint32 +} + +table Album +{ + photos []*Photo // BEFORE cover: the walk-order pin + cover *Photo // the same node, through a pointer field +} diff --git a/tables/lists/tables.baseline b/tables/lists/tables.baseline new file mode 100644 index 000000000..fa2a9eba5 --- /dev/null +++ b/tables/lists/tables.baseline @@ -0,0 +1,107 @@ +schema-tables-baseline 7 +package listdemo + +table Album + field photos id=0x40b1d94aff3ab130 kind=14 elem=17 type=Photo array=unbounded + field cover id=0xaa19a78e404dea20 kind=17 type=Photo + +table Army + field squads id=0x7848019b0c02a926 kind=14 elem=13 type=Squad array=unbounded + field after id=0xbf82010f6f71eae9 kind=4 + +table Bounded + field items id=0x3e7884bf4f412c6f kind=14 elem=13 type=Unit array=bounded bound=8 + field tag id=0x56d7ab194448a4f3 kind=4 + +table Bytes + field data id=0x855b556730a34a05 kind=14 elem=6 array=unbounded + field after id=0xbf82010f6f71eae9 kind=4 + +table Deck + field hands id=0x81b46a69304ee2c9 kind=14 elem=13 type=Row array=bounded bound=3 + field after id=0xbf82010f6f71eae9 kind=4 + +table Floats + field values id=0x21277bcf1a4d67fb kind=14 elem=10 array=unbounded + field after id=0xbf82010f6f71eae9 kind=4 + +table Ints + field values id=0x21277bcf1a4d67fb kind=14 elem=4 array=unbounded + field after id=0xbf82010f6f71eae9 kind=4 + +table Item + field count id=0xb1e5e28e4479a274 kind=4 + +table LogEntry + field tick id=0x1e7683ef2ebc7684 kind=8 + +table Mixed + field grades id=0xd90a4e7682f799c5 kind=14 elem=7 enum=Grade array=unbounded + field perms id=0x4af2ed8470862ea8 kind=14 elem=9 flags=Perm array=unbounded + field hits id=0x732dfbcc9b0cf0bb kind=14 elem=15 union=Hit array=unbounded + field bounds id=0x52f60c4caef0b768 kind=14 elem=4 array=unbounded min=0 max=100 + +table Photo + field width id=0xdbdacd932fd1e9bf kind=8 + field height id=0x17720bf67d347222 kind=8 + +table Placement + field x id=0xaf63f54c86021707 kind=10 + field y id=0xaf63f44c86021554 kind=10 + field model id=0x9de543933e6e703a kind=8 + +table Point + field x id=0xaf63f54c86021707 kind=4 + field y id=0xaf63f44c86021554 kind=4 + +table Row + field items id=0x3e7884bf4f412c6f kind=14 elem=13 type=Sample array=unbounded + field label id=0x39f7fcec8fcb623d kind=4 + +table Sample + field v id=0xaf63eb4c86020609 kind=4 + +table Save + field placements id=0xd24733aa574d4b09 kind=14 elem=13 type=Placement array=unbounded + field log id=0x125073191daf5431 kind=14 elem=17 type=LogEntry array=unbounded + field scores id=0x01986b0b27400fb2 kind=14 elem=4 array=unbounded + +table Sheet + field rows id=0xa3a7061ff10a8138 kind=14 elem=13 type=Row array=unbounded + field pinned id=0x5f82477707ad620f kind=17 type=Row + +table Squad + field roster id=0x1c84390d304f4f42 kind=14 elem=13 array=map keykind=6 + field name id=0xc4bcadba8e631b86 kind=4 + +table Unbounded + field items id=0x3e7884bf4f412c6f kind=14 elem=13 type=Unit array=unbounded + field tag id=0x56d7ab194448a4f3 kind=4 + +table Unit + field v id=0xaf63eb4c86020609 kind=4 + +table ec07a2f760550a91.1c84390d304f4f42 + field key id=0x3dc94a19365b10ec kind=6 + field value id=0x7ce4fd9430e80cea kind=13 type=Item + +enum Grade + variant A id=0xaf63fc4c860222ec + variant B id=0xaf63ff4c86022805 + variant C id=0xaf63fe4c86022652 + +flags Perm + variant Read bit=0 + variant Write bit=1 + variant Own bit=2 + +union Hit + arm point id=0x73feab3544c345b1 payload=Point + arm damage id=0x7f6308be8ab37fc0 kind=4 + +## history +### 2026-09-05 (UTC) — first baseline: the unbounded array corpus +- baseline created over 20 tables — data written BEFORE this point is not covered by it + +### 2026-09-05 (UTC) — Deck: the unreached-slot control +- no compatibility-affecting edits; the wire absorbs the rest diff --git a/test/tables/lists_main.cpp b/test/tables/lists_main.cpp new file mode 100644 index 000000000..a5cf015fd --- /dev/null +++ b/test/tables/lists_main.cpp @@ -0,0 +1,1403 @@ +// THE LIST GATE (docs/SPEC-TABLES.md §2.9). One binary over the `tables/lists` +// unit: the builder's three, the four writing walks in INDEX order, the node +// extent a region and a cook carry, every reader rule, the migration golden, +// and the negative controls §2.9 names. Each row here is one of them, and the +// comment says which sabotage it turns red. +// +// Compiled WITHOUT the serialize include path: the Table headers stand alone. +// +// schema_test_lists every battery +// schema_test_lists measure-refusals the four LoadMeasure refusals alone +// (make tables-list-measure-refusals) + +#include +#include +#include + +#include + +#include "SaveTable.h" +#include "SharedTable.h" +#include "HoldersTable.h" +#include "MigrateTable.h" +#include "ReportTable.h" +#include "wirebuilder.h" + +using namespace listdemo; + +static int failures = 0; + +// ---- THE ALLOCATION AUDIT (docs/SPEC-TABLES.md §2.9, §6.5) ---- +// +// The reading path allocates nothing of its own: LoadMeasure, Load into the +// caller's region, the const indexing and iteration, and Open. Every +// allocation the program makes through operator new is counted here, and the +// audit requires the count to stay where it was across all of them. CONTROL: +// an allocation is planted in Load or in the const indexing, and the audit +// goes red. +static long long allocations = 0; + +void * operator new( size_t bytes ) +{ + allocations++; + void * p = malloc( bytes != 0 ? bytes : 1 ); + if ( p == NULL ) { abort(); } + return p; +} +void operator delete( void * p ) noexcept { free( p ); } +void operator delete( void * p, size_t ) noexcept { free( p ); } + +#define CHECK( condition ) \ + do \ + { \ + if ( !( condition ) ) \ + { \ + printf( "FAIL %s:%d: %s\n", __FILE__, __LINE__, #condition ); \ + fflush( stdout ); \ + failures++; \ + } \ + } while ( 0 ) + +#define CHECK_EQ( actual, expected ) \ + do \ + { \ + const long long a_ = (long long) ( actual ); \ + const long long e_ = (long long) ( expected ); \ + if ( a_ != e_ ) \ + { \ + printf( "FAIL %s:%d: %s = %lld, want %lld\n", \ + __FILE__, __LINE__, #actual, a_, e_ ); \ + fflush( stdout ); \ + failures++; \ + } \ + } while ( 0 ) + +// ---- every allocation sized from a measure goes through here ---- +// +// A measure is an answer from the code under test, so a region sized from one +// is the single place a broken measure reaches the allocator. The ceiling is +// CHECKED first, and a measure past it is a red CHECK on every platform rather +// than a call to calloc (test/tables/maps_main.cpp says why). +// +// 256 MiB: the largest measure this corpus produces is the 100,000-element +// clamp control's, well under a megabyte. + +static const int64_t kMeasureCeiling = 256 * 1024 * 1024; + +static void * measured_calloc( int64_t measure, int64_t extra, const char * expr, const char * file, int line ) +{ + if ( measure < 0 || measure > kMeasureCeiling ) + { + printf( "FAIL %s:%d: %s = %lld, past the %lld byte measure ceiling\n", + file, line, expr, (long long) measure, (long long) kMeasureCeiling ); + failures++; + return NULL; + } + return calloc( 1, (size_t) ( measure + extra ) ); +} + +#define MEASURED_CALLOC( measure, extra ) \ + measured_calloc( ( measure ), ( extra ), #measure, __FILE__, __LINE__ ) + +// ---- the shared golden wire (docs/SPEC-TABLES.md §3) ---- +// +// The C++ reference is the writer: these instances' encodings are pinned into +// testdata/wire/tables/.bin. A break here under an unchanged schema is +// stop-the-line, never a quiet re-pin. SCHEMA_UPDATE_WIRE_GOLDENS=1 rewrites +// them deliberately (make update-goldens). It answers whether the bytes are +// the pinned ones, because a COOK the pin refused is a file no Open may trust: +// Open matches the header and points (§7), so a layout sabotage that reached +// it would crash rather than fail a CHECK. + +static bool pin_golden( const char * name, const uint8_t * data, int64_t bytes ) +{ + char path[256]; + snprintf( path, sizeof( path ), "testdata/wire/tables/%s.bin", name ); + if ( getenv( "SCHEMA_UPDATE_WIRE_GOLDENS" ) ) + { + FILE * f = fopen( path, "wb" ); + if ( f == NULL ) { printf( "FAIL cannot write %s\n", path ); fflush( stdout ); failures++; return false; } + fwrite( data, 1, (size_t) bytes, f ); + fclose( f ); + return true; + } + FILE * f = fopen( path, "rb" ); + if ( f == NULL ) + { + printf( "FAIL missing table wire golden %s (run: make update-goldens)\n", path ); + fflush( stdout ); + failures++; + return false; + } + static uint8_t expected[1u << 20]; + const size_t n = fread( expected, 1, sizeof( expected ), f ); + fclose( f ); + if ( (int64_t) n != bytes || memcmp( expected, data, n ) != 0 ) + { + printf( "FAIL table wire golden %s: %lld bytes written, %lld pinned\n", + name, (long long) bytes, (long long) n ); + fflush( stdout ); + failures++; + return false; + } + return true; +} + +// A COOK IS WRITTEN FOR THE BUILD THAT OPENS IT (docs/SPEC-TABLES.md §7): the +// host's own order, so the round trip below holds on the big-endian leg too. +static TableByteOrder host_byte_order() +{ + const uint16_t probe = 1; + return *(const uint8_t *) &probe == 1 ? TableByteOrder::Little : TableByteOrder::Big; +} + +// THE COOKS `schema cook-check` READS (docs/SPEC-TABLES.md §7.4): when the +// Makefile names a directory, the cooks this gate writes are saved there, and +// beside one of them a FORGERY whose list slot points its array past the +// holder's extent, which the tool must refuse. +static void save_cook( const char * name, const void * data, int64_t bytes ) +{ + const char * dir = getenv( "SCHEMA_LIST_COOK_DIR" ); + if ( dir == NULL ) { return; } + char path[512]; + snprintf( path, sizeof( path ), "%s/%s.cook", dir, name ); + FILE * f = fopen( path, "wb" ); + if ( f == NULL ) { printf( "FAIL cannot write %s\n", path ); failures++; return; } + fwrite( data, 1, (size_t) bytes, f ); + fclose( f ); +} + +static void report_silent( const TableReport & r, const char * where ) +{ + if ( r.unknown != 0 || r.kind_mismatch != 0 || r.clamped != 0 || r.duplicate != 0 || r.malformed ) + { + printf( "FAIL %s: the report is not silent (unknown %d, kind_mismatch %d, clamped %d, duplicate %d, malformed %d)\n", + where, r.unknown, r.kind_mismatch, r.clamped, r.duplicate, (int) r.malformed ); + failures++; + } +} + +static void reports_agree( const TableReport & a, const TableReport & b ) +{ + CHECK_EQ( a.unknown, b.unknown ); + CHECK_EQ( a.kind_mismatch, b.kind_mismatch ); + CHECK_EQ( a.clamped, b.clamped ); + CHECK_EQ( a.duplicate, b.duplicate ); + CHECK_EQ( (int) a.malformed, (int) b.malformed ); +} + +// ---- the instances (docs/SPEC-TABLES.md §2.9) ---- + +// §2.9's own example: three placements, a []*T whose two slots name one node +// beside a null slot, and three scalars: `list_tables`. +static void build_save( SaveBuilder & b ) +{ + Save * save = b.GetRoot(); + for ( int i = 0; i < 3; i++ ) + { + Placement * placement = SavePlacementsAdd( b.main, save->placements ); + CHECK( placement != NULL ); + placement->x = 1.0f + (float) i; + placement->y = 2.0f * (float) i; + placement->model = (uint32_t) ( 3 + i ); + } + // a pointer element: Add hands back the SLOT at null, Emplace fills it as it + // fills any pointer slot, and a second slot holds the same reference + TableRef * slot = SaveLogAdd( b.main, save->log ); + CHECK( slot != NULL ); + LogEntry * shared = LogEntryEmplace( b.main, *slot ); + CHECK( shared != NULL ); + shared->tick = 7; + *SaveLogAdd( b.main, save->log ) = *slot; // two slots, one node + SaveLogAdd( b.main, save->log ); // and a null slot + *SaveScoresAdd( b.main, save->scores ) = 10; + *SaveScoresAdd( b.main, save->scores ) = 20; + *SaveScoresAdd( b.main, save->scores ) = 30; +} + +static uint8_t wire_tables[1u << 16]; +static int64_t bytes_tables = 0; + +// ---- the writer (docs/SPEC-TABLES.md §2.9) ---- + +static void test_writer() +{ + { + SaveBuilder b; + build_save( b ); + const int64_t measured = SaveMeasure( b ); + bytes_tables = SaveSave( b, wire_tables, sizeof( wire_tables ) ); + CHECK_EQ( measured, bytes_tables ); // measure == save over a list is a check on the arithmetic alone (§2.9) + pin_golden( "list_tables", wire_tables, bytes_tables ); + // MEASURE EQUALS SAVE AT EXACT CAPACITY, and one short of it refuses + static uint8_t exact[1u << 16]; + CHECK_EQ( SaveSave( b, exact, measured ), measured ); + CHECK( memcmp( exact, wire_tables, (size_t) measured ) == 0 ); + CHECK_EQ( SaveSave( b, exact, measured - 1 ), -1 ); + } + { + // CONTROL: the writer emits the elements OUT OF ORDER. `list_scalars` + // meets it: the byte compare against its pinned wire goes red while + // measure == save still holds. + SaveBuilder b; + Save * save = b.GetRoot(); + *SaveScoresAdd( b.main, save->scores ) = 10; + *SaveScoresAdd( b.main, save->scores ) = 20; + *SaveScoresAdd( b.main, save->scores ) = 30; + static uint8_t wire[1u << 16]; + const int64_t measured = SaveMeasure( b ); + const int64_t n = SaveSave( b, wire, sizeof( wire ) ); + CHECK_EQ( measured, n ); + pin_golden( "list_scalars", wire, n ); + } + { + // `list_empty`: an EMPTY list beside a full one elides under §3's + // by-value rule, and a fresh Save is the empty wire + SaveBuilder b; + Save * save = b.GetRoot(); + SavePlacementsAdd( b.main, save->placements )->model = 1; + SavePlacementsAdd( b.main, save->placements )->model = 2; + static uint8_t wire[1u << 16]; + const int64_t n = SaveSave( b, wire, sizeof( wire ) ); + CHECK_EQ( SaveMeasure( b ), n ); + pin_golden( "list_empty", wire, n ); + SaveBuilder fresh; + static uint8_t empty[64]; + CHECK_EQ( SaveSave( fresh, empty, sizeof( empty ) ), empty_wire_bytes ); + } + { + // CONTROL: `Save` emits a DEAD element. `list_erased` meets it, an + // erase from the MIDDLE with an add after it, so a writer that merely + // truncates still goes red, and the byte compare against the same + // five elements added directly says the sabotage is the skip and not + // the arithmetic. + SaveBuilder b; + Save * save = b.GetRoot(); + Placement * held[5] = { NULL, NULL, NULL, NULL, NULL }; + for ( int i = 0; i < 5; i++ ) + { + held[i] = SavePlacementsAdd( b.main, save->placements ); + held[i]->model = (uint32_t) ( 100 + i ); + } + CHECK( SavePlacementsErase( b.arena, save->placements, held[2] ) ); + CHECK( !SavePlacementsErase( b.arena, save->placements, held[2] ) ); // already erased: false + Placement foreign; + CHECK( !SavePlacementsErase( b.arena, save->placements, &foreign ) ); // not this list's: false + SavePlacementsAdd( b.main, save->placements )->model = 105; + CHECK_EQ( save->placements.count, 5 ); + static uint8_t erased[1u << 16]; + const int64_t measured = SaveMeasure( b ); + const int64_t n = SaveSave( b, erased, sizeof( erased ) ); + CHECK_EQ( measured, n ); + pin_golden( "list_erased", erased, n ); + + SaveBuilder direct; + const uint32_t models[5] = { 100, 101, 103, 104, 105 }; + for ( int i = 0; i < 5; i++ ) { SavePlacementsAdd( direct.main, direct.GetRoot()->placements )->model = models[i]; } + static uint8_t straight[1u << 16]; + const int64_t m = SaveSave( direct, straight, sizeof( straight ) ); + CHECK_EQ( m, n ); + CHECK( memcmp( straight, erased, (size_t) n ) == 0 ); + } + { + // the five element classes are [..N]T's exactly: an ENUM, a FLAGS mask + // and a UNION are elements as they are in a bounded array, and a bar + // attribute qualifies the ELEMENT + MixedBuilder b; + Mixed * mixed = b.GetRoot(); + *MixedGradesAdd( b.main, mixed->grades ) = Grade::A; + *MixedGradesAdd( b.main, mixed->grades ) = Grade::C; + *MixedGradesAdd( b.main, mixed->grades ) = Grade::B; + *MixedPermsAdd( b.main, mixed->perms ) = Perm_Read | Perm_Write; + *MixedPermsAdd( b.main, mixed->perms ) = Perm_Own; + Hit * point = MixedHitsAdd( b.main, mixed->hits ); + point->type = HitType::Point; + PointReset( point->point ); + point->point.x = 1; + point->point.y = 2; + Hit * damage = MixedHitsAdd( b.main, mixed->hits ); + damage->type = HitType::Damage; + damage->damage = 7; + MixedHitsAdd( b.main, mixed->hits ); // a None element in its place + *MixedBoundsAdd( b.main, mixed->bounds ) = 0; + *MixedBoundsAdd( b.main, mixed->bounds ) = 50; + *MixedBoundsAdd( b.main, mixed->bounds ) = 100; + static uint8_t wire[1u << 16]; + const int64_t measured = MixedMeasure( b ); + const int64_t n = MixedSave( b, wire, sizeof( wire ) ); + CHECK_EQ( measured, n ); + pin_golden( "list_mixed", wire, n ); + + const int64_t need = MixedLoadMeasure( wire, n ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport report; + const Mixed * loaded = MixedLoad( region, need, wire, n, &report ); + CHECK( loaded != NULL ); + report_silent( report, "list_mixed" ); + if ( loaded != NULL ) + { + CHECK_EQ( loaded->grades.size(), 3 ); + CHECK( loaded->grades[1] == Grade::C ); + CHECK_EQ( loaded->perms.size(), 2 ); + CHECK_EQ( loaded->perms[0], Perm_Read | Perm_Write ); + CHECK_EQ( loaded->hits.size(), 3 ); + CHECK( loaded->hits[0].type == HitType::Point && loaded->hits[0].point.y == 2 ); + CHECK( loaded->hits[1].type == HitType::Damage && loaded->hits[1].damage == 7 ); + CHECK( loaded->hits[2].type == HitType::None ); + CHECK_EQ( loaded->bounds.size(), 3 ); + CHECK_EQ( loaded->bounds[2], 100 ); + } + free( region ); + } +} + +// ---- the builder's three (docs/SPEC-TABLES.md §2.9) ---- + +static void test_builder() +{ + SaveBuilder b; + Save * save = b.GetRoot(); + + // ADD hands the element back at its declared defaults + Placement * first = SavePlacementsAdd( b.main, save->placements ); + CHECK( first != NULL ); + CHECK( first->x == 0.0f && first->model == 0 ); + CHECK_EQ( save->placements.count, 1 ); + first->model = 11; + + // MORE THAN ONE SEGMENT: an element's address is stable for the arena's + // life, so a pointer handed back by an early Add survives every later one + for ( int i = 1; i < 200; i++ ) + { + Placement * p = SavePlacementsAdd( b.main, save->placements ); + CHECK( p != NULL ); + p->model = (uint32_t) ( 11 + i ); + } + CHECK_EQ( save->placements.count, 200 ); + CHECK_EQ( first->model, 11 ); // NOTHING EVER MOVES (§6.4) + + // ERASE by the element's own pointer, from the MIDDLE. EACH on the builder + // is INDEX order, live elements only + int32_t seen = 0; + Placement * third = NULL; + for ( Placement * p : SavePlacementsEach( b.arena, save->placements ) ) + { + if ( seen == 2 ) { third = p; } + seen++; + } + CHECK_EQ( seen, 200 ); + CHECK( third != NULL && third->model == 13 ); + CHECK( SavePlacementsErase( b.arena, save->placements, third ) ); + CHECK_EQ( save->placements.count, 199 ); + seen = 0; + for ( Placement * p : SavePlacementsEach( b.arena, save->placements ) ) + { + CHECK( p->model != 13 ); // the dead element is skipped + if ( seen == 2 ) { CHECK_EQ( p->model, 14 ); } // INDICES ARE NOT STABLE ACROSS AN ERASE + seen++; + } + CHECK_EQ( seen, 199 ); + + // and the const form agrees once locked: what was index 3 is index 2 + CHECK( b.Lock() ); + const Save * locked = b.AsConst(); + CHECK( locked != NULL ); + if ( locked != NULL ) + { + CHECK_EQ( locked->placements.size(), 199 ); + CHECK_EQ( locked->placements[2].model, 14 ); + CHECK_EQ( locked->placements[198].model, 210 ); + } + + // a []*T: Add hands back the SLOT at null + SaveBuilder p; + TableRef * slot = SaveLogAdd( p.main, p.GetRoot()->log ); + CHECK( slot != NULL && slot->value == 0 ); + CHECK_EQ( p.GetRoot()->log.count, 1 ); + LogEntry * entry = LogEntryEmplace( p.main, *slot ); + CHECK( entry != NULL ); + entry->tick = 3; + int32_t slots = 0; + for ( TableRef * s : SaveLogEach( p.arena, p.GetRoot()->log ) ) { CHECK( LogEntryAt( p.arena, *s ) == entry ); slots++; } + CHECK_EQ( slots, 1 ); +} + +// ---- the const form: a locked region, a loaded one, an opened cook ---- + +static void check_const_form( const Save * s, const char * where ) +{ + if ( s == NULL ) { printf( "FAIL %s: no root\n", where ); failures++; return; } + CHECK_EQ( s->placements.size(), 3 ); + if ( s->placements.size() == 3 ) + { + CHECK( s->placements[0].x == 1.0f ); + CHECK_EQ( s->placements[2].model, 5 ); + int i = 0; + for ( const Placement & p : s->placements ) { CHECK_EQ( p.model, 3 + i ); i++; } + CHECK_EQ( i, 3 ); + } + // a []*T's const operator[] answers the RESOLVED pointer: TWO SLOTS, ONE + // NODE, and a null slot answers NULL + CHECK_EQ( s->log.size(), 3 ); + if ( s->log.size() == 3 ) + { + const LogEntry * a = s->log[0]; + const LogEntry * again = s->log[1]; + CHECK( a != NULL && a == again ); + if ( a != NULL ) { CHECK_EQ( a->tick, 7 ); } + CHECK( s->log[2] == NULL ); + int i = 0; + for ( const LogEntry * e : s->log ) { if ( i < 2 ) { CHECK( e == a ); } else { CHECK( e == NULL ); } i++; } + CHECK_EQ( i, 3 ); + } + CHECK_EQ( s->scores.size(), 3 ); + if ( s->scores.size() == 3 ) { CHECK_EQ( s->scores[1], 20 ); } +} + +static void test_const_forms() +{ + SaveBuilder b; + build_save( b ); + CHECK( b.Lock() ); + check_const_form( b.AsConst(), "the locked region" ); + + // a locked region re-saves the same bytes as the builder did + static uint8_t again[1u << 16]; + const int64_t n = SaveSave( b.AsConst(), again, sizeof( again ) ); + CHECK_EQ( n, bytes_tables ); + CHECK( memcmp( again, wire_tables, (size_t) bytes_tables ) == 0 ); + + // LOAD into the caller's exact-sized region + const int64_t need = SaveLoadMeasure( wire_tables, bytes_tables ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport report; + const Save * loaded = SaveLoad( region, need, wire_tables, bytes_tables, &report ); + check_const_form( loaded, "a loaded region" ); + report_silent( report, "a loaded region" ); + const int64_t back = SaveSave( loaded, again, sizeof( again ) ); + CHECK_EQ( back, bytes_tables ); + CHECK( memcmp( again, wire_tables, (size_t) bytes_tables ) == 0 ); + + // LoadBuilder is the TOOL's path, and it produces EXACTLY the same report + SaveBuilder into; + TableReport tool; + CHECK( SaveLoadBuilder( into, wire_tables, bytes_tables, &tool ) ); + reports_agree( tool, report ); + CHECK_EQ( into.GetRoot()->placements.count, 3 ); + CHECK_EQ( into.GetRoot()->log.count, 3 ); + const int64_t relocked = SaveSave( into, again, sizeof( again ) ); + CHECK_EQ( relocked, bytes_tables ); + CHECK( memcmp( again, wire_tables, (size_t) bytes_tables ) == 0 ); + + // the COOK: a region written verbatim, opened O(1) and indexed in place + const int64_t cook_bytes = SaveCookMeasure( loaded ); + CHECK( cook_bytes > 0 ); + void * cooked = MEASURED_CALLOC( cook_bytes, 0 ); + if ( cooked == NULL ) { free( region ); return; } + CHECK( SaveCook( loaded, cooked, (uint64_t) cook_bytes, host_byte_order() ) ); + check_const_form( SaveOpen( cooked, (uint64_t) cook_bytes ), "an opened cook" ); + save_cook( "save", cooked, cook_bytes ); + // and two cooks of one instance are ONE artifact, from the region and from the builder alike + void * twice = MEASURED_CALLOC( cook_bytes, 0 ); + if ( twice == NULL ) { free( cooked ); free( region ); return; } + CHECK( SaveCook( loaded, twice, (uint64_t) cook_bytes, host_byte_order() ) ); + CHECK( memcmp( cooked, twice, (size_t) cook_bytes ) == 0 ); + CHECK_EQ( SaveCookMeasure( into ), cook_bytes ); + CHECK( SaveCook( into, twice, (uint64_t) cook_bytes, host_byte_order() ) ); + CHECK( memcmp( cooked, twice, (size_t) cook_bytes ) == 0 ); + free( twice ); + free( cooked ); + // the region is EXACT: one byte short is refused. Load zeroes the region + // it is handed before it refuses, so this probe is the block's last act. + TableReport short_report; + CHECK( SaveLoad( region, need - 1, wire_tables, bytes_tables, &short_report ) == NULL ); + free( region ); +} + +// ---- the reader's rules, each on a hand-made body (§2.9) ---- + +// an `Ints` body written FROM THE GRAMMAR: a kind 14 array of kind 4 (int32) +// elements, `after` behind it, and one knob each for the controls +struct IntsSpec +{ + int32_t n; // elements written + int64_t declared_n; // -1: n itself + uint8_t element_kind; // 0: int32's own kind + int64_t body_len; // -1: the body's own length; else a FORGED array body length + int32_t after; +}; + +static IntsSpec ints_spec() +{ + IntsSpec spec = { 3, -1, 0, -1, 777 }; + return spec; +} + +struct Wire +{ + uint8_t bytes[4096]; + int64_t size; +}; + +static Wire build_ints( const IntsSpec & spec ) +{ + WireBuilder b; + b.field( "values", 14 ); + if ( spec.body_len < 0 ) + { + const int64_t body = b.open_len(); + b.u8( spec.element_kind != 0 ? spec.element_kind : 4 ); + b.leb( spec.declared_n >= 0 ? (uint64_t) spec.declared_n : (uint64_t) spec.n ); + for ( int32_t i = 0; i < spec.n; i++ ) { b.u32( (uint32_t) ( 10 * ( i + 1 ) ) ); } + b.close_len( body ); + } + else + { + // a body TOO SHORT for its own header, or short of its count + b.leb( (uint64_t) spec.body_len ); + for ( int64_t i = 0; i < spec.body_len; i++ ) { b.u8( i == 0 ? 4 : 0 ); } + } + b.field( "after", 4 ); + b.u32( (uint32_t) spec.after ); + b.end(); + Wire w; + w.size = b.finish( w.bytes ); + return w; +} + +struct Verdict +{ + int32_t count; + TableReport report; + int32_t after; + int32_t decoded[3]; +}; + +static Verdict read_ints( const Wire & w ) +{ + Verdict v = { -1, TableReport(), 0, { 0, 0, 0 } }; + const int64_t need = IntsLoadMeasure( w.bytes, w.size ); + if ( need < 0 ) { v.count = -2; return v; } // the measure REFUSED the framing + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return v; } + const Ints * ints = IntsLoad( region, need, w.bytes, w.size, &v.report ); + if ( ints != NULL ) + { + v.count = ints->values.size(); + v.after = ints->after; + for ( int32_t i = 0; i < v.count && i < 3; i++ ) { v.decoded[i] = ints->values[i]; } + } + free( region ); + + // EVERY LOAD PATH PRODUCES ONE REPORT (§2.9): the two paths agree on every + // wire either of them decodes + IntsBuilder into; + TableReport t; + IntsLoadBuilder( into, w.bytes, w.size, &t ); + reports_agree( t, v.report ); + CHECK_EQ( into.GetRoot()->values.count, v.count < 0 ? 0 : v.count ); + return v; +} + +static void test_reader() +{ + { + const Verdict v = read_ints( build_ints( ints_spec() ) ); + CHECK_EQ( v.count, 3 ); + CHECK_EQ( v.decoded[2], 30 ); + CHECK_EQ( v.after, 777 ); + report_silent( v.report, "a good Ints wire" ); + } + { + // CONTROL: the element-kind rule decodes anyway. An `Ints` wire read by + // `Floats` is §3's element-kind mismatch: the field reads EMPTY, one + // kind_mismatch counts, and the parent reads on. And the reverse. + Wire w = build_ints( ints_spec() ); + const int64_t need = FloatsLoadMeasure( w.bytes, w.size ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Floats * floats = FloatsLoad( region, need, w.bytes, w.size, &r ); + CHECK( floats != NULL ); + if ( floats != NULL ) + { + CHECK_EQ( floats->values.size(), 0 ); + CHECK_EQ( floats->after, 777 ); + } + CHECK_EQ( r.kind_mismatch, 1 ); + CHECK_EQ( r.unknown + r.clamped + r.duplicate + (int) r.malformed, 0 ); + free( region ); + + IntsSpec spec = ints_spec(); + spec.element_kind = 10; // float32's kind, under Ints' declaration + const Verdict v = read_ints( build_ints( spec ) ); + CHECK_EQ( v.count, 0 ); + CHECK_EQ( v.report.kind_mismatch, 1 ); + CHECK( !v.report.malformed ); + CHECK_EQ( v.after, 777 ); + } + { + // A COUNT THE BODY CANNOT COVER (§2.9): into a REGION, LoadMeasure + // answers -1 with the reason count_over_length and no Load runs. Into a + // BUILDER, the prefix the body covers lands, malformed counts, and the + // parent reads on past the field's L + IntsSpec spec = ints_spec(); + spec.declared_n = 1000; + Wire w = build_ints( spec ); + TableRefuseReason reason = count_over_extent_cap; + CHECK_EQ( IntsLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_length ); + IntsBuilder into; + TableReport t; + CHECK( IntsLoadBuilder( into, w.bytes, w.size, &t ) ); + CHECK( t.malformed ); + CHECK_EQ( t.kind_mismatch + t.clamped + t.unknown + t.duplicate, 0 ); + CHECK_EQ( into.GetRoot()->values.count, 3 ); // the prefix the body covers + CHECK_EQ( into.GetRoot()->after, 777 ); + } + { + // A BODY TOO SHORT TO CARRY ITS OWN HEADER IS INERT (§4): no element, + // no counter, the field keeps the value it has + IntsSpec spec = ints_spec(); + spec.body_len = 1; + const Verdict v = read_ints( build_ints( spec ) ); + CHECK_EQ( v.count, 0 ); + report_silent( v.report, "an inert list body" ); + CHECK_EQ( v.after, 777 ); + } + { + // an EMPTY list on the wire: a header and a zero count + IntsSpec spec = ints_spec(); + spec.n = 0; + const Verdict v = read_ints( build_ints( spec ) ); + CHECK_EQ( v.count, 0 ); + report_silent( v.report, "an empty list body" ); + CHECK_EQ( v.after, 777 ); + } + { + // A DAMAGED ELEMENT inside a good count: a list of TABLES whose third + // element's L runs past the body keeps the two it decoded, counts + // malformed, and the parent reads on past the field's L + WireBuilder b; + b.field( "items", 14 ); + const int64_t body = b.open_len(); + b.u8( 13 ); + b.leb( 3 ); + for ( int i = 0; i < 2; i++ ) + { + const int64_t elem = b.open_len(); + b.field( "v", 4 ); + b.u32( (uint32_t) ( 5 + i ) ); + b.end(); + b.close_len( elem ); + } + b.leb( 100 ); // the third element's L, past the body + b.close_len( body ); + b.field( "label", 4 ); + b.u32( 9 ); + b.end(); + Wire w; + w.size = b.finish( w.bytes ); + const int64_t need = RowLoadMeasure( w.bytes, w.size ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Row * row = RowLoad( region, need, w.bytes, w.size, &r ); + CHECK( row != NULL ); + CHECK( r.malformed ); + if ( row != NULL ) + { + CHECK_EQ( row->items.size(), 2 ); // the element that never landed is not counted + if ( row->items.size() == 2 ) { CHECK_EQ( row->items[1].v, 6 ); } + CHECK_EQ( row->label, 9 ); + } + RowBuilder into; + TableReport t; + RowLoadBuilder( into, w.bytes, w.size, &t ); + reports_agree( t, r ); + CHECK_EQ( into.GetRoot()->items.count, 2 ); + free( region ); + } +} + +// ---- THE CLAMP CONTROL AT 100,000 (docs/SPEC-TABLES.md §2.9) ---- +// +// CONTROL: the reader CLAMPS the count against something. A []uint8 carrying +// 100,000 elements is past 2^16 and so past any bound a schema on the page +// declares, so a clamp any control author happened to pick would show, and the +// decoded count goes red, and `clamped` stays at zero. +static void test_clamp_control() +{ + static const int32_t kElements = 100000; + BytesBuilder b; + Bytes * bytes = b.GetRoot(); + for ( int32_t i = 0; i < kElements; i++ ) + { + uint8_t * e = BytesDataAdd( b.main, bytes->data ); + if ( e == NULL ) { printf( "FAIL: Add answered NULL at %d\n", i ); failures++; return; } + *e = (uint8_t) ( i & 0xff ); + } + bytes->after = 4242; + CHECK_EQ( bytes->data.count, kElements ); + const int64_t measured = BytesMeasure( b ); + uint8_t * wire = (uint8_t *) MEASURED_CALLOC( measured, 0 ); + if ( wire == NULL ) { return; } + const int64_t n = BytesSave( b, wire, measured ); + CHECK_EQ( n, measured ); + + const int64_t need = BytesLoadMeasure( wire, n ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { free( wire ); return; } + TableReport r; + const Bytes * loaded = BytesLoad( region, need, wire, n, &r ); + CHECK( loaded != NULL ); + report_silent( r, "the clamp control" ); + if ( loaded != NULL ) + { + CHECK_EQ( loaded->data.size(), kElements ); + CHECK_EQ( r.clamped, 0 ); + bool intact = loaded->data.size() == kElements; + for ( int32_t i = 0; intact && i < kElements; i++ ) { if ( loaded->data[i] != (uint8_t) ( i & 0xff ) ) { intact = false; } } + CHECK( intact ); + CHECK_EQ( loaded->after, 4242 ); + } + BytesBuilder into; + TableReport t; + CHECK( BytesLoadBuilder( into, wire, n, &t ) ); + reports_agree( t, r ); + CHECK_EQ( into.GetRoot()->data.count, kElements ); + free( region ); + free( wire ); +} + +// ---- THE FOUR LoadMeasure REFUSALS (docs/SPEC-TABLES.md §2.9, §6.5) ---- +// +// A unit test and not a `report` row, because a refusal produces no counters. +// Each wire is built in memory with a SYNTHETIC count rather than a golden: +// a count above the int32 cap, which no golden could carry because the file +// would be two gigabytes, a count whose elements cannot fit the field's L, the +// same two at DEPTH, inside an element's own list, and a clean wire beside +// them, which must measure. Red if any of the four answers something other +// than -1 with its own reason, if the clean one refuses, or if any of them +// moves one of the report's counters. + +static Wire build_sheet( uint64_t rows, uint64_t items, int32_t real_items ) +{ + WireBuilder b; + b.field( "rows", 14 ); + const int64_t body = b.open_len(); + b.u8( 13 ); + b.leb( rows ); + { + const int64_t row = b.open_len(); + b.field( "items", 14 ); + const int64_t inner = b.open_len(); + b.u8( 13 ); + b.leb( items ); + for ( int32_t i = 0; i < real_items; i++ ) + { + const int64_t sample = b.open_len(); + b.field( "v", 4 ); + b.u32( (uint32_t) i ); + b.end(); + b.close_len( sample ); + } + b.close_len( inner ); + b.field( "label", 4 ); + b.u32( 1 ); + b.end(); + b.close_len( row ); + } + b.close_len( body ); + b.end(); + Wire w; + w.size = b.finish( w.bytes ); + return w; +} + +static void test_measure_refusals() +{ + // a count above the int32 cap, at the ROOT + { + IntsSpec spec = ints_spec(); + spec.declared_n = 0x80000000ll; + Wire w = build_ints( spec ); + TableRefuseReason reason = count_over_length; + CHECK_EQ( IntsLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_extent_cap ); + // and into a BUILDER it is the refusal LoadBuilder answers NULL for, + // the report holding what it held when the count was met: nothing + IntsBuilder into; + TableReport t; + CHECK( !IntsLoadBuilder( into, w.bytes, w.size, &t ) ); + report_silent( t, "the over-cap refusal into a builder" ); + } + // a count the field's L cannot carry, at the ROOT + { + IntsSpec spec = ints_spec(); + spec.declared_n = 100000; + Wire w = build_ints( spec ); + TableRefuseReason reason = count_over_extent_cap; + CHECK_EQ( IntsLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_length ); + } + // the same two at DEPTH, inside an element's own list + { + Wire w = build_sheet( 1, 0x80000000ull, 1 ); + TableRefuseReason reason = count_over_length; + CHECK_EQ( SheetLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_extent_cap ); + SheetBuilder into; + TableReport t; + CHECK( !SheetLoadBuilder( into, w.bytes, w.size, &t ) ); + report_silent( t, "the over-cap refusal at depth into a builder" ); + } + { + Wire w = build_sheet( 1, 100000, 1 ); + TableRefuseReason reason = count_over_extent_cap; + CHECK_EQ( SheetLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_length ); + } + // and a clean wire beside them, which must measure and load silently + { + Wire w = build_sheet( 1, 2, 2 ); + TableRefuseReason reason = count_over_length; + const int64_t need = SheetLoadMeasure( w.bytes, w.size, NULL, &reason ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Sheet * sheet = SheetLoad( region, need, w.bytes, w.size, &r ); + CHECK( sheet != NULL ); + report_silent( r, "the clean wire beside the refusals" ); + if ( sheet != NULL ) + { + CHECK_EQ( sheet->rows.size(), 1 ); + if ( sheet->rows.size() == 1 ) { CHECK_EQ( sheet->rows[0].items.size(), 2 ); } + } + free( region ); + } +} + +// ---- sharing and the walk order (docs/SPEC-TABLES.md §2.9, §3.1) ---- + +static void test_shared() +{ + { + // `list_shared`: two slots naming one node beside a null slot. CONTROL: + // a shared node is written TWICE: the region's byte count and the + // text round trip's &node resolution go red. + AlbumBuilder b; + Album * album = b.GetRoot(); + TableRef * slot = AlbumPhotosAdd( b.main, album->photos ); + Photo * photo = PhotoEmplace( b.main, *slot ); + photo->width = 640; + photo->height = 480; + *AlbumPhotosAdd( b.main, album->photos ) = *slot; + AlbumPhotosAdd( b.main, album->photos ); // null + static uint8_t wire[1u << 16]; + const int64_t n = AlbumSave( b, wire, sizeof( wire ) ); + CHECK_EQ( AlbumMeasure( b ), n ); + pin_golden( "list_shared", wire, n ); + + const int64_t need = AlbumLoadMeasure( wire, n ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Album * loaded = AlbumLoad( region, need, wire, n, &r ); + CHECK( loaded != NULL ); + report_silent( r, "list_shared" ); + if ( loaded != NULL ) + { + CHECK_EQ( loaded->photos.size(), 3 ); + if ( loaded->photos.size() == 3 ) + { + CHECK( loaded->photos[0] != NULL && loaded->photos[0] == loaded->photos[1] ); + CHECK( loaded->photos[2] == NULL ); + if ( loaded->photos[0] != NULL ) { CHECK_EQ( loaded->photos[0]->width, 640 ); } + } + CHECK( PhotoAt( loaded->cover ) == NULL ); + // ONE node in the region: the attribution names the root and one photo + int64_t attribution = 0; + CHECK( AlbumLoadMeasure( wire, n, &attribution ) == need ); + CHECK_EQ( attribution, 2 * (int64_t) sizeof( TableNodeDirEntry ) ); + } + // the text: one definition and one &node reference, a null in its place + CHECK( b.Lock() ); + const int64_t text_bytes = AlbumToJsonMeasure( b.AsConst() ); + char * text = (char *) MEASURED_CALLOC( text_bytes, 1 ); + if ( text == NULL ) { free( region ); return; } + CHECK_EQ( AlbumToJson( b.AsConst(), text, text_bytes ), text_bytes ); + CHECK( strstr( text, "&node" ) != NULL ); + CHECK( strstr( text, "null" ) != NULL ); + AlbumBuilder into; + TableReport t; + CHECK( AlbumFromJson( into, text, text_bytes, &t ) ); + report_silent( t, "list_shared from text" ); + static uint8_t from_text[1u << 16]; + CHECK_EQ( AlbumSave( into, from_text, sizeof( from_text ) ), n ); + CHECK( memcmp( from_text, wire, (size_t) n ) == 0 ); + free( text ); + free( region ); + } + { + // `list_before_pointer`: the []*T is DECLARED BEFORE `cover` and reaches + // the shared node first, so it numbers it first. CONTROL: the walk + // visits lists out of declaration order, grouped after the pointer + // fields, and the pinned wire goes red on the node numbering. + AlbumBuilder b; + Album * album = b.GetRoot(); + TableRef * first = AlbumPhotosAdd( b.main, album->photos ); + Photo * b_node = PhotoEmplace( b.main, *first ); + b_node->width = 2; + TableRef * second = AlbumPhotosAdd( b.main, album->photos ); + Photo * a_node = PhotoEmplace( b.main, *second ); + a_node->width = 1; + album->cover = *second; // the SAME node, through the pointer field declared after + static uint8_t wire[1u << 16]; + const int64_t n = AlbumSave( b, wire, sizeof( wire ) ); + CHECK_EQ( AlbumMeasure( b ), n ); + pin_golden( "list_before_pointer", wire, n ); + const int64_t need = AlbumLoadMeasure( wire, n ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Album * loaded = AlbumLoad( region, need, wire, n, &r ); + CHECK( loaded != NULL ); + report_silent( r, "list_before_pointer" ); + if ( loaded != NULL ) + { + CHECK_EQ( loaded->photos.size(), 2 ); + CHECK( PhotoAt( loaded->cover ) != NULL && PhotoAt( loaded->cover ) == loaded->photos[1] ); + CHECK( loaded->photos[0] != loaded->photos[1] ); + } + free( region ); + } +} + +// ---- where else a list rides in a holder's extent (§2.9) ---- + +static void test_nested() +{ + { + // `list_nested`: a list of tables that hold lists, and a pointed-at + // holder with a list of its own. CONTROL: LoadMeasure's term summed at + // ONE DEPTH only, and the measure goes red against the region Load fills. + SheetBuilder b; + Sheet * sheet = b.GetRoot(); + for ( int i = 0; i < 3; i++ ) + { + Row * row = SheetRowsAdd( b.main, sheet->rows ); + row->label = 10 + i; + for ( int k = 0; k <= i; k++ ) { RowItemsAdd( b.main, row->items )->v = 100 * i + k; } + } + Row * pinned = RowEmplace( b.main, sheet->pinned ); + pinned->label = 99; + RowItemsAdd( b.main, pinned->items )->v = 1; + RowItemsAdd( b.main, pinned->items )->v = 2; + static uint8_t wire[1u << 16]; + const int64_t n = SheetSave( b, wire, sizeof( wire ) ); + CHECK_EQ( SheetMeasure( b ), n ); + pin_golden( "list_nested", wire, n ); + + const int64_t need = SheetLoadMeasure( wire, n ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Sheet * loaded = SheetLoad( region, need, wire, n, &r ); + CHECK( loaded != NULL ); + report_silent( r, "list_nested" ); + if ( loaded != NULL ) + { + CHECK_EQ( loaded->rows.size(), 3 ); + for ( int32_t i = 0; i < loaded->rows.size(); i++ ) + { + const Row & row = loaded->rows[i]; + CHECK_EQ( row.label, 10 + i ); + CHECK_EQ( row.items.size(), i + 1 ); + for ( int32_t k = 0; k < row.items.size(); k++ ) { CHECK_EQ( row.items[k].v, 100 * i + k ); } + } + const Row * p = RowAt( loaded->pinned ); + CHECK( p != NULL ); + if ( p != NULL ) + { + CHECK_EQ( p->label, 99 ); + CHECK_EQ( p->items.size(), 2 ); + if ( p->items.size() == 2 ) { CHECK_EQ( p->items[1].v, 2 ); } + } + static uint8_t again[1u << 16]; + CHECK_EQ( SheetSave( loaded, again, sizeof( again ) ), n ); + CHECK( memcmp( again, wire, (size_t) n ) == 0 ); + + // the COOK, from the region and from the builder, one artifact + const int64_t cook_bytes = SheetCookMeasure( loaded ); + CHECK( cook_bytes > 0 ); + void * cooked = MEASURED_CALLOC( cook_bytes, 0 ); + if ( cooked != NULL ) + { + CHECK( SheetCook( loaded, cooked, (uint64_t) cook_bytes, host_byte_order() ) ); + // THE COOK IS PINNED on a little-endian host, so a LAYOUT + // sabotage (the element array laid after a nested container's, + // a wrong alignment, a dropped pre-order) goes red on a CHECK + // before any Open trusts the bytes. The big-endian leg writes + // the other order and skips the pin. + const bool trusted = host_byte_order() != TableByteOrder::Little || pin_golden( "list_nested_cook", (const uint8_t *) cooked, cook_bytes ); + const Sheet * opened = trusted ? SheetOpen( cooked, (uint64_t) cook_bytes ) : NULL; + CHECK( opened != NULL ); + if ( opened != NULL ) + { + CHECK_EQ( opened->rows.size(), 3 ); + if ( opened->rows.size() == 3 ) { CHECK_EQ( opened->rows[2].items[2].v, 202 ); } + const Row * op = RowAt( opened->pinned ); + CHECK( op != NULL && op->items.size() == 2 && op->items[0].v == 1 ); + } + CHECK_EQ( SheetCookMeasure( b ), cook_bytes ); + void * twice = MEASURED_CALLOC( cook_bytes, 0 ); + if ( twice != NULL ) + { + CHECK( SheetCook( b, twice, (uint64_t) cook_bytes, host_byte_order() ) ); + CHECK( memcmp( cooked, twice, (size_t) cook_bytes ) == 0 ); + free( twice ); + } + save_cook( "sheet", cooked, cook_bytes ); + // THE FORGERY for `schema cook-check`: the root's `rows` slot is the + // first sixteen bytes of the data part, and its delta is pointed + // past the region, which §7.4's containment clause must refuse + if ( getenv( "SCHEMA_LIST_COOK_DIR" ) != NULL ) + { + uint8_t * forged = (uint8_t *) cooked; + const int64_t delta = cook_bytes; // past every byte the file has + memcpy( forged + 64, &delta, sizeof( delta ) ); + save_cook( "sheet-forged", forged, cook_bytes ); + } + free( cooked ); + } + } + // and the TOOL's path reads every depth into a builder + SheetBuilder into; + TableReport t; + CHECK( SheetLoadBuilder( into, wire, n, &t ) ); + reports_agree( t, r ); + CHECK_EQ( into.GetRoot()->rows.count, 3 ); + static uint8_t relocked[1u << 16]; + CHECK_EQ( SheetSave( into, relocked, sizeof( relocked ) ), n ); + CHECK( memcmp( relocked, wire, (size_t) n ) == 0 ); + // the region is EXACT: one byte short is refused, so the extent scan + // counted every depth and every node and none twice. Load zeroes the + // region before it refuses, so this is the block's last act. + TableReport short_report; + CHECK( SheetLoad( region, need - 1, wire, n, &short_report ) == NULL ); + free( region ); + } + { + // `list_of_maps`: an element that holds a MAP. CONTROL: the element + // array is laid out AFTER a nested container's, breaking the pre-order + // rule, and the region's byte compare goes red. + ArmyBuilder b; + Army * army = b.GetRoot(); + for ( int i = 0; i < 2; i++ ) + { + Squad * squad = ArmySquadsAdd( b.main, army->squads ); + squad->name = 50 + i; + SquadRosterInsert( b.main, squad->roster, (uint8_t) ( 9 - i ) )->count = 90 + i; + SquadRosterInsert( b.main, squad->roster, (uint8_t) ( 2 + i ) )->count = 20 + i; + } + army->after = 8; + static uint8_t wire[1u << 16]; + const int64_t n = ArmySave( b, wire, sizeof( wire ) ); + CHECK_EQ( ArmyMeasure( b ), n ); + pin_golden( "list_of_maps", wire, n ); + const int64_t need = ArmyLoadMeasure( wire, n ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Army * loaded = ArmyLoad( region, need, wire, n, &r ); + CHECK( loaded != NULL ); + report_silent( r, "list_of_maps" ); + if ( loaded != NULL ) + { + CHECK_EQ( loaded->squads.size(), 2 ); + for ( int32_t i = 0; i < loaded->squads.size(); i++ ) + { + const Squad & squad = loaded->squads[i]; + CHECK_EQ( squad.name, 50 + i ); + CHECK_EQ( squad.roster.size(), 2 ); + const Item * low = squad.roster.Find( (uint8_t) ( 2 + i ) ); + CHECK( low != NULL && low->count == 20 + i ); + if ( squad.roster.size() == 2 ) { CHECK_EQ( squad.roster.begin().at->key, 2 + i ); } // sorted + } + CHECK_EQ( loaded->after, 8 ); + static uint8_t again[1u << 16]; + CHECK_EQ( ArmySave( loaded, again, sizeof( again ) ), n ); + CHECK( memcmp( again, wire, (size_t) n ) == 0 ); + // the cook lays the element array FIRST, then each element's map + const int64_t cook_bytes = ArmyCookMeasure( loaded ); + void * cooked = MEASURED_CALLOC( cook_bytes, 0 ); + if ( cooked != NULL ) + { + CHECK( ArmyCook( loaded, cooked, (uint64_t) cook_bytes, host_byte_order() ) ); + const bool trusted = host_byte_order() != TableByteOrder::Little || pin_golden( "list_of_maps_cook", (const uint8_t *) cooked, cook_bytes ); + const Army * opened = trusted ? ArmyOpen( cooked, (uint64_t) cook_bytes ) : NULL; + + CHECK( opened != NULL ); + + if ( opened != NULL && opened->squads.size() == 2 ) + { + const Item * item = opened->squads[1].roster.Find( (uint8_t) 8 ); + CHECK( item != NULL && item->count == 91 ); + } + save_cook( "army", cooked, cook_bytes ); + free( cooked ); + } + } + ArmyBuilder into; + TableReport t; + CHECK( ArmyLoadBuilder( into, wire, n, &t ) ); + reports_agree( t, r ); + static uint8_t relocked[1u << 16]; + CHECK_EQ( ArmySave( into, relocked, sizeof( relocked ) ), n ); + CHECK( memcmp( relocked, wire, (size_t) n ) == 0 ); + TableReport short_report; // one byte short is refused; last, because Load zeroes the region first + CHECK( ArmyLoad( region, need - 1, wire, n, &short_report ) == NULL ); + free( region ); + } + + { + // AN UNREACHED NON-EMPTY LIST SLOT IS REFUSED by Cook and by Lock, the + // same refusal §7.6 gives a pointer in that position. The WIRE is not + // refused, a counted array rides its live slots. + DeckBuilder past; + Deck * d = past.GetRoot(); + d->hands_count = 1; + RowItemsAdd( past.main, d->hands[0].items )->v = 1; + RowItemsAdd( past.main, d->hands[2].items )->v = 9; // past the count + static uint8_t rides[1u << 16]; + const int64_t measured = DeckMeasure( past ); + CHECK_EQ( DeckSave( past, rides, sizeof( rides ) ), measured ); + CHECK( !past.Lock() ); + CHECK( past.AsConst() == NULL ); // nothing partial + CHECK_EQ( DeckCookMeasure( past ), -1 ); + } +} + +// ---- the migration itself (docs/SPEC-TABLES.md §2.9) ---- + +static void test_migrates() +{ + // ONE content, TWO declarations of the holder, [..8]Unit and []Unit, ONE + // pinned wire both write byte for byte and both read into equal values, + // the report silent in both directions. The bound is above the count, so + // the row proves the framing and not the clamp. + static uint8_t bounded_wire[1u << 16]; + static uint8_t unbounded_wire[1u << 16]; + int64_t bounded_bytes = 0, unbounded_bytes = 0; + { + Bounded value; + BoundedReset( value ); + value.items_count = 3; + for ( int i = 0; i < 3; i++ ) { value.items[i].v = 7 * ( i + 1 ); } + value.tag = 42; + bounded_bytes = BoundedSave( value, bounded_wire, sizeof( bounded_wire ) ); + CHECK_EQ( BoundedMeasure( value ), bounded_bytes ); + } + { + UnboundedBuilder b; + Unbounded * value = b.GetRoot(); + for ( int i = 0; i < 3; i++ ) { UnboundedItemsAdd( b.main, value->items )->v = 7 * ( i + 1 ); } + value->tag = 42; + unbounded_bytes = UnboundedSave( b, unbounded_wire, sizeof( unbounded_wire ) ); + CHECK_EQ( UnboundedMeasure( b ), unbounded_bytes ); + } + CHECK_EQ( unbounded_bytes, bounded_bytes ); + CHECK( memcmp( bounded_wire, unbounded_wire, (size_t) bounded_bytes ) == 0 ); + pin_golden( "list_migrates", unbounded_wire, unbounded_bytes ); + + // each reads the other's wire silently, into equal values + { + Bounded back; + TableReport r; + CHECK( BoundedLoad( back, unbounded_wire, unbounded_bytes, &r ) ); + report_silent( r, "the bounded reader over the unbounded wire" ); + CHECK_EQ( back.items_count, 3 ); + CHECK_EQ( back.items[2].v, 21 ); + CHECK_EQ( back.tag, 42 ); + } + { + const int64_t need = UnboundedLoadMeasure( bounded_wire, bounded_bytes ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Unbounded * back = UnboundedLoad( region, need, bounded_wire, bounded_bytes, &r ); + CHECK( back != NULL ); + report_silent( r, "the unbounded reader over the bounded wire" ); + if ( back != NULL ) + { + CHECK_EQ( back->items.size(), 3 ); + if ( back->items.size() == 3 ) { CHECK_EQ( back->items[2].v, 21 ); } + CHECK_EQ( back->tag, 42 ); + } + free( region ); + } +} + +// ---- the TEXT form (docs/SPEC-TABLES.md §2.9, §16) ---- + +static void test_text() +{ + SaveBuilder b; + build_save( b ); + CHECK( b.Lock() ); + const Save * locked = b.AsConst(); + + const int64_t need = SaveToJsonMeasure( locked ); + CHECK( need > 0 ); + char * text = (char *) MEASURED_CALLOC( need, 1 ); + if ( text == NULL ) { return; } + const int64_t written = SaveToJson( locked, text, need ); + CHECK_EQ( written, need ); + + // A JSON ARRAY, in INDEX order, the pointer row per element of a []*T + CHECK( strstr( text, "\"placements\": [" ) != NULL ); + CHECK( strstr( text, "\"scores\": [" ) != NULL ); + CHECK( strstr( text, "&node" ) != NULL ); + CHECK( strstr( text, "null" ) != NULL ); + const char * ten = strstr( text, "10" ); + const char * thirty = strstr( text, "30" ); + CHECK( ten != NULL && thirty != NULL && ten < thirty ); + + // and the text reads back: one instance, one text, both ways + SaveBuilder into; + TableReport report; + CHECK( SaveFromJson( into, text, written, &report ) ); + report_silent( report, "list_tables from text" ); + CHECK_EQ( into.GetRoot()->placements.count, 3 ); + CHECK_EQ( into.GetRoot()->log.count, 3 ); + CHECK_EQ( into.GetRoot()->scores.count, 3 ); + CHECK( into.Lock() ); + const int64_t again_bytes = SaveToJsonMeasure( into.AsConst() ); + char * again = (char *) MEASURED_CALLOC( again_bytes, 1 ); + if ( again == NULL ) { free( text ); return; } + CHECK_EQ( SaveToJson( into.AsConst(), again, again_bytes ), again_bytes ); + CHECK_EQ( again_bytes, written ); + CHECK( memcmp( again, text, (size_t) written ) == 0 ); // byte-stable + static uint8_t from_text[1u << 16]; + CHECK_EQ( SaveSave( into.AsConst(), from_text, sizeof( from_text ) ), bytes_tables ); + CHECK( memcmp( from_text, wire_tables, (size_t) bytes_tables ) == 0 ); + free( again ); + free( text ); + + // `[]` is an empty list, null is kind_mismatch, a wrong-shaped element + // counts and keeps its slot at defaults, and EVERY element the text + // carries is read, because there is no bound to drop a tail against + struct TextRow { const char * text; int32_t count; int32_t mismatch; int32_t after; }; + const TextRow rows[] = { + { "{\"values\":[],\"after\":5}", 0, 0, 5 }, + { "{\"values\":null,\"after\":5}", 0, 1, 5 }, + { "{\"values\":[1,\"x\",3],\"after\":5}", 3, 1, 5 }, + { "{\"values\":[1,2],\"values\":[9],\"after\":5}", 1, 0, 5 }, // LAST WINS, whole + }; + for ( int i = 0; i < 4; i++ ) + { + IntsBuilder rb; + TableReport r; + CHECK( IntsFromJson( rb, rows[i].text, (int64_t) strlen( rows[i].text ), &r ) ); + CHECK_EQ( rb.GetRoot()->values.count, rows[i].count ); + CHECK_EQ( r.kind_mismatch, rows[i].mismatch ); + CHECK_EQ( r.clamped, 0 ); + CHECK_EQ( rb.GetRoot()->after, rows[i].after ); + } + { + static char many[8192]; + int at = snprintf( many, sizeof( many ), "{\"values\":[" ); + for ( int i = 0; i < 300; i++ ) { at += snprintf( many + at, sizeof( many ) - (size_t) at, "%s%d", i > 0 ? "," : "", i ); } + snprintf( many + at, sizeof( many ) - (size_t) at, "]}" ); + IntsBuilder rb; + TableReport r; + CHECK( IntsFromJson( rb, many, (int64_t) strlen( many ), &r ) ); + report_silent( r, "300 elements from text" ); + CHECK_EQ( rb.GetRoot()->values.count, 300 ); + int32_t i = 0; + for ( const int32_t * v : IntsValuesEach( rb.arena, rb.GetRoot()->values ) ) { if ( *v != i ) { failures++; printf( "FAIL element %d read as %d\n", i, *v ); break; } i++; } + } +} + +// ---- the reading path allocates nothing (§2.9, §6.5) ---- + +static void test_allocation_audit() +{ + const long long before = allocations; + const int64_t need = SaveLoadMeasure( wire_tables, bytes_tables ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Save * loaded = SaveLoad( region, need, wire_tables, bytes_tables, &r ); + CHECK( loaded != NULL ); + long long sum = 0; + if ( loaded != NULL ) + { + for ( int32_t i = 0; i < loaded->placements.size(); i++ ) { sum += loaded->placements[i].model; } + for ( const Placement & p : loaded->placements ) { sum += p.model; } + for ( const LogEntry * e : loaded->log ) { if ( e != NULL ) { sum += e->tick; } } + for ( int32_t i = 0; i < loaded->scores.size(); i++ ) { sum += loaded->scores[i]; } + } + CHECK( sum > 0 ); + const int64_t cook_bytes = SaveCookMeasure( loaded ); + void * cooked = MEASURED_CALLOC( cook_bytes, 0 ); + if ( cooked != NULL ) + { + CHECK( SaveCook( loaded, cooked, (uint64_t) cook_bytes, host_byte_order() ) ); + const Save * opened = SaveOpen( cooked, (uint64_t) cook_bytes ); + CHECK( opened != NULL ); + if ( opened != NULL ) { for ( const Placement & p : opened->placements ) { sum += p.model; } } + free( cooked ); + } + free( region ); + CHECK_EQ( allocations - before, 0 ); // not one operator new on the reading path +} + +int main( int argc, char ** argv ) +{ + + if ( argc > 1 && strcmp( argv[1], "measure-refusals" ) == 0 ) + { + test_measure_refusals(); + if ( failures != 0 ) + { + printf( "\n%d measure refusal check(s) failed\n", failures ); + return 1; + } + printf( "list measure refusals: four -1s with their reasons, one clean measure, no counter moved (docs/SPEC-TABLES.md §2.9, §6.5)\n" ); + return 0; + } + test_writer(); + test_builder(); + test_const_forms(); + test_reader(); + test_clamp_control(); + test_measure_refusals(); + test_shared(); + test_nested(); + test_migrates(); + test_text(); + test_allocation_audit(); + + if ( failures != 0 ) + { + printf( "\n%d list check(s) failed\n", failures ); + return 1; + } + printf( "lists: all checks passed (docs/SPEC-TABLES.md §2.9)\n" ); + return 0; +} diff --git a/testdata/wire/tables/list_before_pointer.bin b/testdata/wire/tables/list_before_pointer.bin new file mode 100644 index 0000000000000000000000000000000000000000..30c86c0abfd4c221f6ae0275a8ea724b1970f961 GIT binary patch literal 82 zcmZQ%kpx008rG8wdaZ literal 0 HcmV?d00001 diff --git a/testdata/wire/tables/list_empty.bin b/testdata/wire/tables/list_empty.bin new file mode 100644 index 0000000000000000000000000000000000000000..f795145aed903806044b93a21972a1eceaa76a3b GIT binary patch literal 47 tcmZQ%vn%><$uIK6$tR~frsvMR{4o9z5_E<}`p0RVEs2K4{{ literal 0 HcmV?d00001 diff --git a/testdata/wire/tables/list_erased.bin b/testdata/wire/tables/list_erased.bin new file mode 100644 index 0000000000000000000000000000000000000000..1ab58dfd24f09e435444c0acdfecf903045188fb GIT binary patch literal 71 zcmZQ%zh6&IHj6oZi0StBl<*Srz2jO?G}dmkBJ&zyJW{ CAqhbM literal 0 HcmV?d00001 diff --git a/testdata/wire/tables/list_migrates.bin b/testdata/wire/tables/list_migrates.bin new file mode 100644 index 0000000000000000000000000000000000000000..7703c1ee0f24142bf663366639636541bf10fb50 GIT binary patch literal 69 zcmZQ%>Ho>=SLn4Bw7mV~wfh6l>@&Z0 xi#NI+b}?Q3ub5pNDE~EieTXQK{*t`@K>g;feH_X48QVXs^Wpgx#0@eW2mrhTEcE~Y literal 0 HcmV?d00001 diff --git a/testdata/wire/tables/list_nested.bin b/testdata/wire/tables/list_nested.bin new file mode 100644 index 0000000000000000000000000000000000000000..c43c45b999e5349068d9d828836ea732cf3bd3df GIT binary patch literal 189 zcmZQ%zW@h05QLKVYY&`Oe>~ND9L83s#Ody(>B^g99STu5dlxJJMIA6!ne}7Ab uoo&+T{x^TVTXM27wfVeGUeBMjmc87)DgHkc{AN4(Px7Jg0(P)&1_l7^Fe8ot literal 0 HcmV?d00001 diff --git a/testdata/wire/tables/list_nested_cook.bin b/testdata/wire/tables/list_nested_cook.bin new file mode 100644 index 0000000000000000000000000000000000000000..4706884aec9ccfebad74168d528db700ffe13f6a GIT binary patch literal 248 zcmWG`_V9J~_xCNqVD4hYNJ)ktqJT55T1WGeO zX>KU30;OT*@Id(>Dg}sBf%pUvp9JDlKn$}NW==9t3|4BX+UH||brxtPm literal 0 HcmV?d00001 diff --git a/testdata/wire/tables/list_of_maps.bin b/testdata/wire/tables/list_of_maps.bin new file mode 100644 index 0000000000000000000000000000000000000000..db030336fcfd4a2d5402610361938b18e79d741c GIT binary patch literal 163 zcmZQ%>$lR008&|9V-9; literal 0 HcmV?d00001 diff --git a/testdata/wire/tables/list_of_maps_cook.bin b/testdata/wire/tables/list_of_maps_cook.bin new file mode 100644 index 0000000000000000000000000000000000000000..0b2c7c03e68500ceabb4bf63dbe481075f81b353 GIT binary patch literal 184 zcmWG`_V9J~_xCS15Ml*i7x_V da{_S`5Hka@C=i3}jRs;EXqjgf+HL-b4FE*L3_}0_ literal 0 HcmV?d00001 diff --git a/testdata/wire/tables/list_scalars.bin b/testdata/wire/tables/list_scalars.bin new file mode 100644 index 0000000000000000000000000000000000000000..57134b8152bcf70c7a4d4fb4d2686615450615b1 GIT binary patch literal 35 jcmZQ%w54Gzp4tROK~K2||yCME_p zK0X#^kYW)amIGpT9xg@>c1{j(di<}x`>uxSho0FbgIkpKVy literal 0 HcmV?d00001 From dc707884b8b2062be3dca80e954a9cb9e1af749c Mon Sep 17 00:00:00 2001 From: Glenn Fiedler Date: Sat, 5 Sep 2026 08:23:11 -0700 Subject: [PATCH 2/7] tables: cook-check refuses a map slot by name where its scan meets one The unit-level map refusal at cook-check refused every cook from a unit that declares a map anywhere, which is the whole tables/lists corpus. The scan now refuses the slot it cannot bound, naming the field, the reference that reads it and schema#380 as the PR that lands the clause, and a cook of a map-free root in the same unit checks as any other does. Co-Authored-By: Claude Fable 5.1 --- Makefile | 14 +++++++---- compiler/cook.go | 3 --- compiler/tablesmaps.go | 17 +++++++------- compiler/tablesmaps_test.go | 15 ++++++------ internal/tablecook/check.go | 10 +++++++- internal/tablecook/list_test.go | 41 +++++++++++++++++++++++++++++++++ 6 files changed, 77 insertions(+), 23 deletions(-) diff --git a/Makefile b/Makefile index 9ceed6c54..cbe22d586 100644 --- a/Makefile +++ b/Makefile @@ -2708,16 +2708,22 @@ tables-lists: build/schema_test_lists build/schema_test_lists_asan SCHEMA_LIST_COOK_DIR=build/lists-cooks ./build/schema_test_lists ./build/schema_test_lists_asan # `schema cook-check` reads what the runtime cooked (§7.4): the root's list - # slot, every element's own slots and companions, a pointed-at holder's - # list, and an element's map, and refuses the forgery beside them + # slot, every element's own slots and companions, and a pointed-at holder's + # list, and refuses the forgery beside them. The cook whose element holds a + # MAP is refused by name at the map slot, because the tool's map-slot + # clause is schema#380's next PR, and the refusal must be that one and not + # a list clause's ./bin/schema cook-check --root Save build/lists-cooks/save.cook tables/lists ./bin/schema cook-check --root Sheet build/lists-cooks/sheet.cook tables/lists - ./bin/schema cook-check --root Army build/lists-cooks/army.cook tables/lists + @if ./bin/schema cook-check --root Army build/lists-cooks/army.cook tables/lists > build/lists-cooks/army.log 2>&1; then \ + echo "LIST GATE FAILED: cook-check walked past an element's map slot, which it has no clause for"; exit 1; \ + fi + @grep -q "Squad.roster.*schema#380" build/lists-cooks/army.log || { echo "LIST GATE FAILED: the map-holding cook was refused, but not by name at the map slot"; cat build/lists-cooks/army.log; exit 1; } @if ./bin/schema cook-check --root Sheet build/lists-cooks/sheet-forged.cook tables/lists > build/lists-cooks/forged.log 2>&1; then \ echo "LIST GATE FAILED: cook-check accepted a list slot pointing past its holder's extent"; exit 1; \ fi @grep -q "leaves\|extent" build/lists-cooks/forged.log || { echo "LIST GATE FAILED: the forgery was refused, but not on the element-array clause"; cat build/lists-cooks/forged.log; exit 1; } - @echo "list gate: cook-check reads three cooks the runtime wrote and refuses the forged list slot" + @echo "list gate: cook-check reads two cooks the runtime wrote, refuses the forged list slot, and refuses the map-holding cook by name" # THE FOUR LoadMeasure REFUSALS are a unit test and not a `report` row (§2.9, # §6.5): each wire is built in memory with a SYNTHETIC count, and the answer diff --git a/compiler/cook.go b/compiler/cook.go index c2e67e36a..67012431c 100644 --- a/compiler/cook.go +++ b/compiler/cook.go @@ -117,9 +117,6 @@ func orderWord(big bool) string { // It is a person's decision to run it, not a parameter on a load: the runtime // keeps one `Open` that matches the header and points. func (c *Compiler) CookCheck(u *ir.Unit, root string, file []byte) (CookReport, error) { - if err := refuseToolMaps(u); err != nil { - return CookReport{}, err - } m := tabletext.NewModel(u) res, err := tablecook.Check(m, file) if err != nil { diff --git a/compiler/tablesmaps.go b/compiler/tablesmaps.go index 9b94e30a4..7a13a63f4 100644 --- a/compiler/tablesmaps.go +++ b/compiler/tablesmaps.go @@ -39,16 +39,17 @@ func refuseMaps(u *ir.Unit, target string) error { } // refuseToolMaps is the TOOL's COOK refusal (docs/SPEC-TABLES.md §2.8, §15): a unit -// whose table closure declares a map is refused by name at every table surface -// the tool has — pack, unpack, cook, cook-check and uncook — because -// internal/tablewire does not carry the construct yet. +// whose table closure declares a map is refused by name at the tool's COOK and +// UNCOOK surfaces, because internal/tablecook does not lay out the entry +// arrays yet. `cook-check` is not among them: its scan refuses a map SLOT by +// name where it meets one (internal/tablecook), so a cook of a map-free root +// in a unit that declares a map elsewhere is checked as any other is. // // It is here, at the surface, rather than in the engine, and it is NAMED rather -// than left to the decoder. Without it the engine meets a kind 14 field whose -// element kind is 13, decodes nothing into a slot it has no shape for, and -// reports FRAMING DAMAGE — an answer that sends its reader looking for a -// corrupt file when the file is fine and the reader is the one that is short. -// A refusal that says which is which is the whole difference. +// than left to the layout. Without it the engine lays out a region short of +// the entry arrays and a reader meets a slot pointing past its holder's +// extent, which is a corrupt file with nothing saying who wrote it. A refusal +// that says which is which is the whole difference. func refuseToolMaps(u *ir.Unit) error { fields := ir.MapFields(u) if len(fields) == 0 { diff --git a/compiler/tablesmaps_test.go b/compiler/tablesmaps_test.go index 67b620858..63b371fd4 100644 --- a/compiler/tablesmaps_test.go +++ b/compiler/tablesmaps_test.go @@ -271,17 +271,18 @@ func TestMapEntryIsNotARoot(t *testing.T) { } // TestToolRefusesMapsByName: the tool's WIRE and TEXT halves carry maps now -// (docs/SPEC-TABLES.md §2.8), and its COOK half does not — so the cook -// surfaces refuse a map-bearing unit BY NAME rather than laying out an entry -// array they have no placement for. Without the refusal a caller gets a cook -// whose region is short of the entries, which is worse than a diagnostic. +// (docs/SPEC-TABLES.md §2.8), and its COOK half does not — so the cook and +// uncook surfaces refuse a map-bearing unit BY NAME rather than laying out an +// entry array they have no placement for. Without the refusal a caller gets a +// cook whose region is short of the entries, which is worse than a diagnostic. +// `cook-check` refuses at the SLOT instead, where its scan meets one, and +// internal/tablecook's TestCookCheckMapSlotRefusedByName holds that. func TestToolRefusesMapsByName(t *testing.T) { u := unitFromSource(t, mapSrc) c := New() surfaces := map[string]func() error{ - "Cook": func() error { _, _, _, err := c.Cook(u, "Fleet", nil, CookOptions{}); return err }, - "Uncook": func() error { _, err := c.Uncook(u, "Fleet", nil); return err }, - "CookCheck": func() error { _, err := c.CookCheck(u, "Fleet", nil); return err }, + "Cook": func() error { _, _, _, err := c.Cook(u, "Fleet", nil, CookOptions{}); return err }, + "Uncook": func() error { _, err := c.Uncook(u, "Fleet", nil); return err }, } for name, call := range surfaces { t.Run(name, func(t *testing.T) { diff --git a/internal/tablecook/check.go b/internal/tablecook/check.go index e69a4c7bb..a32bc66a4 100644 --- a/internal/tablecook/check.go +++ b/internal/tablecook/check.go @@ -201,6 +201,15 @@ func (s *scan) field(at int64, f *ir.Field) error { } value := pieces[0] switch { + case f.IsMap(): + // §7.4's MAP-SLOT clause (docs/SPEC-TABLES.md §2.8, §7.4) is + // schema#380's next PR, beside the tool's cook half, so a cook that + // holds a map slot is refused HERE, by name, at the node that holds + // it, rather than walked past: a slot the scan cannot bound is a slot + // a forgery could steer through. The C++ reference reads it (--lang + // cpp). A cook of a unit that declares a map SOMEWHERE ELSE checks as + // any other does, because the scan meets no such slot. + return fmt.Errorf("a map slot, and `cook-check` carries no map-slot clause yet (docs/SPEC-TABLES.md §7.4): the C++ reference reads the cook (--lang cpp), and the tool's clause is schema#380's next PR") case f.Type.Pointer && f.Array == ir.ArrayNone: return s.ref(value.Offset, f) case f.Type.Kind == ir.TString, f.Type.Kind == ir.TBytes: @@ -260,7 +269,6 @@ func (s *scan) list(at int64, f *ir.Field) error { return s.slots(start, f, count) } - // companion checks one count companion against its DECLARED bound. A negative // one is refused too: a count is an extent and an extent is never negative, and // a walker handed one indexes backwards out of the region. diff --git a/internal/tablecook/list_test.go b/internal/tablecook/list_test.go index 17b391af3..0308df824 100644 --- a/internal/tablecook/list_test.go +++ b/internal/tablecook/list_test.go @@ -88,3 +88,44 @@ func TestCookCheckListSlot(t *testing.T) { }) } } + +// squadCook writes a cook of one Squad root from tables/lists, the holder of +// `roster map[uint8]Item`: a 24-byte record, the sixteen-byte map slot at +// null and `name` after it, and one directory entry. +func squadCook(u *ir.Unit) []byte { + const header, data, attrib = int64(64), int64(24), int64(16) + out := make([]byte, header+data+attrib) + le := binary.LittleEndian + le.PutUint64(out[0:], tablecook.Magic) + le.PutUint64(out[8:], ir.BuildVersion(u)) + le.PutUint64(out[16:], tablecook.ByteOrderLittle) + le.PutUint64(out[24:], uint64(data)) + le.PutUint64(out[32:], uint64(attrib)) + le.PutUint64(out[40:], 8) + le.PutUint32(out[header+16:], 7) // name + dir := out[header+data:] + le.PutUint64(dir[0:], 0) + le.PutUint64(dir[8:], ir.TableTypeId("Squad")) + return out +} + +// TestCookCheckMapSlotRefusedByName: §7.4's map-slot clause is schema#380's +// next PR, so the scan refuses a map slot BY NAME where it meets one, naming +// the field, the reference that reads it and the PR that lands the clause, +// and a cook of a map-free root in the same unit checks as any other does. +func TestCookCheckMapSlotRefusedByName(t *testing.T) { + u := unit(t, "../../tables/lists") + m := tabletext.NewModel(u) + _, err := tablecook.Check(m, squadCook(u)) + if err == nil { + t.Fatalf("FAILED: cook-check walked past a map slot it has no clause for") + } + for _, want := range []string{"Squad.roster", "schema#380", "cpp", "§7.4"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("the refusal does not name %q: %v", want, err) + } + } + if _, err := tablecook.Check(m, intsCook(u, 24, 3)); err != nil { + t.Fatalf("a map-free root in a unit that declares a map elsewhere was refused: %v", err) + } +} From 7d38cac9d62e19fdeebe04cb8c02a85166d83086 Mon Sep 17 00:00:00 2001 From: Glenn Fiedler Date: Sat, 5 Sep 2026 08:26:35 -0700 Subject: [PATCH 3/7] tables: the list controls run as one target, and the element-kind control's script survives make A blank line inside the aggregate target's continuation ended its dependency list, and the element-kind control's sed script carried commas and an unbalanced parenthesis that make's call split and cut. Co-Authored-By: Claude Fable 5.1 --- Makefile | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/Makefile b/Makefile index cbe22d586..612a12ccc 100644 --- a/Makefile +++ b/Makefile @@ -2837,7 +2837,7 @@ tables-lists-clamp-negative-control: bin/schema build/tables-generated/.stamp # and the decoded values go red. .PHONY: tables-lists-element-kind-negative-control tables-lists-element-kind-negative-control: bin/schema build/tables-generated/.stamp - $(call list_negative_control,elemkind,'s@else if ( elem_kind != %d ) { r.report->kind_mismatch++; r.offset = body_end; break; }\\n", ind, elemKind)@else if ( elem_kind != %d \&\& false ) { r.report->kind_mismatch++; r.offset = body_end; break; }\\n", ind, elemKind)@',internal/codegen/cpptable/lists.go,decoding under a changed element kind left the list gate GREEN) + $(call list_negative_control,elemkind,'s@else if ( elem_kind != %d ) { r.report->kind_mismatch++; r.offset = body_end; break; }@else if ( elem_kind != %d \&\& false ) { r.report->kind_mismatch++; r.offset = body_end; break; }@',internal/codegen/cpptable/lists.go,decoding under a changed element kind left the list gate GREEN) # LoadMeasure OVER A LIST OF TABLES HOLDING LISTS, summed at ONE DEPTH only: # `list_nested` meets it, and the measure goes red against the region Load @@ -2886,7 +2886,6 @@ tables-lists-allocation-negative-control: bin/schema build/tables-generated/.sta .PHONY: tables-lists-negative-controls tables-lists-negative-controls: tables-lists-allocation-negative-control \ tables-lists-order-negative-control \ - tables-lists-dead-element-negative-control \ tables-lists-preorder-negative-control \ tables-lists-walk-order-negative-control \ From b31db6110665a612a553439cc089dc62a679617c Mon Sep 17 00:00:00 2001 From: Glenn Fiedler Date: Sat, 5 Sep 2026 08:28:23 -0700 Subject: [PATCH 4/7] tables: pin the header goldens of the lists corpus Co-Authored-By: Claude Fable 5.1 --- testdata/golden/tables/lists/HoldersTable.cpp | 3217 +++++ testdata/golden/tables/lists/HoldersTable.h | 11518 ++++++++++++++++ testdata/golden/tables/lists/MigrateTable.cpp | 3149 +++++ testdata/golden/tables/lists/MigrateTable.h | 5606 ++++++++ testdata/golden/tables/lists/ReportTable.cpp | 3153 +++++ testdata/golden/tables/lists/ReportTable.h | 7562 ++++++++++ testdata/golden/tables/lists/SaveTable.cpp | 3181 +++++ testdata/golden/tables/lists/SaveTable.h | 8531 ++++++++++++ testdata/golden/tables/lists/SharedTable.cpp | 3134 +++++ testdata/golden/tables/lists/SharedTable.h | 5530 ++++++++ 10 files changed, 54581 insertions(+) create mode 100644 testdata/golden/tables/lists/HoldersTable.cpp create mode 100644 testdata/golden/tables/lists/HoldersTable.h create mode 100644 testdata/golden/tables/lists/MigrateTable.cpp create mode 100644 testdata/golden/tables/lists/MigrateTable.h create mode 100644 testdata/golden/tables/lists/ReportTable.cpp create mode 100644 testdata/golden/tables/lists/ReportTable.h create mode 100644 testdata/golden/tables/lists/SaveTable.cpp create mode 100644 testdata/golden/tables/lists/SaveTable.h create mode 100644 testdata/golden/tables/lists/SharedTable.cpp create mode 100644 testdata/golden/tables/lists/SharedTable.h diff --git a/testdata/golden/tables/lists/HoldersTable.cpp b/testdata/golden/tables/lists/HoldersTable.cpp new file mode 100644 index 000000000..127d117a7 --- /dev/null +++ b/testdata/golden/tables/lists/HoldersTable.cpp @@ -0,0 +1,3217 @@ +// Code generated by the schema compiler from Holders.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — the TABLE wire's text form (docs/SPEC-TABLES.md §16). +// Compile this file to use FromJson / ToJson; a project that +// never reads or writes a text does not compile it and pays nothing. + +#include "HoldersTable.h" + +#include // the text form: number formatting +#include // the text form: exact number conversion +#include // the text form: the runtime's decimal point + +// The guard is not vestigial. Several listdemo Table.cpp files may be +// concatenated into ONE translation unit — a unity build — and without it +// each would redefine the walk. It is also why the walk's functions may be +// weak (vague linkage) across separate objects: ODR requires their +// definitions to be token-identical, and the generic-walk gate is what +// proves that, byte for byte, across every generated .cpp. +#ifndef LISTDEMO_SCHEMA_TABLE_JSON +#define LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +// ---- the pointer adapters (docs/SPEC-TABLES.md §16.7) ---- +// +// The walk below is ONE walk, byte-identical in every generated .cpp, and a +// pointer is the one kind it cannot walk alone: reading one needs the +// builder's arena and writing one needs a region's deref, and neither exists +// in a unit that declares no pointer. So the walk calls these three and does +// not define them. A unit with no pointer defines them as stubs no field ever +// reaches; a pointered unit defines them in the graph half that follows the +// walk. + +struct TableJsonIn; +struct TableJsonOut; + +// a pointer field's object, or the `&node` reference standing in for it, into +// the slot; the cursor is on the opening brace +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// the node a pointer slot names, in place — or as `&node` when it is shared +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// the FIRST key of an object the walk is skipping begins with `&`: the cursor is +// on its value. A dropped definition still takes its label (§16.7); a fixed reader +// skips the value whole, as it skips everything else it does not place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); + +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- +// +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map +inline bool TableJsonIsMap( const TableFieldInfo * f ); +// the map as a plain JSON object keyed by the KEY, in ASCENDING key order +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that object back into the slot, in whatever order the text gives it +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); + +// ---- json walk: begin ---- +// +// The TEXT form (docs/SPEC-TABLES.md §16): one table, one text, one walk over the +// reflection descriptors (§8). Reading fills ONE caller-owned instance and +// allocates nothing beyond it; writing targets a caller buffer with the +// wire's measure/write symmetry. Everything AROUND this — which file goes +// with which instance, what key an instance is filed under, how instances +// link into a root table's collections — is a packer's opinion and stays +// with the tool that holds it. +// +// The dialect: trailing commas are accepted on read (the authoring files +// this exists for carry them) and never written; comments are not JSON and +// are refused; unknown keys are skipped and counted; a duplicate key is +// last-wins and counted; a key present with the wrong JSON type is skipped +// and counted, never coerced. + +static const int32_t kTableJsonMaxDepth = 128; + +// A key longer than this cannot name a field, so it is skipped as unknown. +static const int32_t kTableJsonMaxKey = 256; + +// The longest numeric token the walk will convert. Anything longer is a +// value no field can hold and counts as a kind mismatch. +static const int32_t kTableJsonMaxNumber = 512; + +// The decimal point the C runtime is CURRENTLY using. Number conversion is +// the one locale-sensitive corner of the grammar — JSON's point is always +// '.', the runtime's is whatever the program set — so every number crosses +// this one character on the way out and on the way back in. Nothing else in +// the walk consults the locale. +inline char TableJsonDecimalPoint() +{ + const struct lconv * conv = localeconv(); + if ( conv != NULL && conv->decimal_point != NULL && conv->decimal_point[0] != 0 ) + { + return conv->decimal_point[0]; + } + return '.'; +} + +// ---- storage access: the descriptors give an offset and a width, and the +// ---- storage is the HOST's, so every load and store goes through a width +// ---- switch rather than a memcpy into the low bytes of a wider word + +// finite: not a NaN, not an infinity. Written without — the walk's +// runtime surface stays the handful of functions it already names. +// A vocabulary entry the descriptor could not spell. The generated name +// functions answer "???" for a value outside the declared set, and that is +// not a name — writing it would put a spelling in the text that the reader +// then counts as unknown, turning a refusal into a silent loss. +inline bool TableJsonNamed( const char * name ) +{ + return name != NULL && strcmp( name, "???" ) != 0; +} + +inline bool TableJsonFinite( double v ) +{ + return v == v && v <= 1.7976931348623157e308 && v >= -1.7976931348623157e308; +} + +inline uint64_t TableJsonGetRaw( const void * storage, uint32_t width ) +{ + switch ( width ) + { + case 1: { uint8_t v = 0; memcpy( &v, storage, 1 ); return v; } + case 2: { uint16_t v = 0; memcpy( &v, storage, 2 ); return v; } + case 4: { uint32_t v = 0; memcpy( &v, storage, 4 ); return v; } + case 8: { uint64_t v = 0; memcpy( &v, storage, 8 ); return v; } + } + return 0; +} + +inline void TableJsonSetRaw( void * storage, uint32_t width, uint64_t value ) +{ + switch ( width ) + { + case 1: { uint8_t v = (uint8_t) value; memcpy( storage, &v, 1 ); break; } + case 2: { uint16_t v = (uint16_t) value; memcpy( storage, &v, 2 ); break; } + case 4: { uint32_t v = (uint32_t) value; memcpy( storage, &v, 4 ); break; } + case 8: { uint64_t v = value; memcpy( storage, &v, 8 ); break; } + } +} + +inline int64_t TableJsonGetSigned( const void * storage, uint32_t width ) +{ + uint64_t raw = TableJsonGetRaw( storage, width ); + if ( width < 8 ) + { + uint64_t sign = uint64_t( 1 ) << ( width * 8 - 1 ); + if ( ( raw & sign ) != 0 ) + { + raw |= ~( ( sign << 1 ) - 1 ); + } + } + return (int64_t) raw; +} + +// ---- the WIDE kinds (docs/SPEC-TABLES.md §3, §16.2) ---- +// +// The 128-bit integers and the fixed-point family convert EXACTLY, over two +// 64-bit lanes: a 128-bit integer is a decimal integer, a fixed value a +// decimal in WHOLE UNITS (1.0, -0.25, 3.0000152587890625) and nothing +// on either path passes through a double. Nothing here needs a 128-bit type +// either, which is what keeps this walk one text for every unit. +struct TableJsonWide +{ + uint64_t lo; + uint64_t hi; +}; + +inline bool TableJsonKindWide( uint8_t kind ) { return kind >= 18 && kind <= 29; } +inline bool TableJsonKindWideSigned( uint8_t kind ) { return kind == 18 || ( kind >= 20 && kind <= 24 ); } +inline bool TableJsonKindFixed( uint8_t kind ) { return kind >= 20 && kind <= 29; } + +inline bool TableJsonWideZero( TableJsonWide v ) { return v.lo == 0 && v.hi == 0; } +inline bool TableJsonWideNegative( TableJsonWide v ) { return ( v.hi >> 63 ) != 0; } + +inline int TableJsonWideCompare( TableJsonWide a, TableJsonWide b, bool is_signed ) +{ + if ( is_signed && TableJsonWideNegative( a ) != TableJsonWideNegative( b ) ) { return TableJsonWideNegative( a ) ? -1 : 1; } + if ( a.hi != b.hi ) { return a.hi < b.hi ? -1 : 1; } + if ( a.lo != b.lo ) { return a.lo < b.lo ? -1 : 1; } + return 0; +} + +inline TableJsonWide TableJsonWideShl( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.hi = v.lo << ( n - 64 ); return r; } + r.hi = ( v.hi << n ) | ( v.lo >> ( 64 - n ) ); + r.lo = v.lo << n; + return r; +} + +inline TableJsonWide TableJsonWideShr( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.lo = v.hi >> ( n - 64 ); return r; } + r.lo = ( v.lo >> n ) | ( v.hi << ( 64 - n ) ); + r.hi = v.hi >> n; + return r; +} + +inline TableJsonWide TableJsonWideNeg( TableJsonWide v ) +{ + TableJsonWide r; + r.lo = ~v.lo + 1; + r.hi = ~v.hi + ( r.lo == 0 ? 1 : 0 ); + return r; +} + +// v = v * m + a; the return is the carry out of 128 bits +inline uint32_t TableJsonWideMulAdd( TableJsonWide * v, uint32_t m, uint32_t a ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t carry = a; + for ( int i = 0; i < 4; i++ ) + { + uint64_t p = limb[i] * m + carry; + limb[i] = p & 0xffffffffull; + carry = p >> 32; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) carry; +} + +// v = v / d; the return is the remainder +inline uint32_t TableJsonWideDiv( TableJsonWide * v, uint32_t d ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t rem = 0; + for ( int i = 3; i >= 0; i-- ) + { + uint64_t cur = ( rem << 32 ) | limb[i]; + limb[i] = cur / d; + rem = cur % d; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) rem; +} + +// The storage of a wide kind, as lanes. A sixteen-byte storage is serialize's +// pair — native __int128 in the host's byte order, or the emulated struct with +// its low lane first — so the lanes are read in the host's order; a narrower +// storage is one lane, sign-extended for a signed kind. +inline TableJsonWide TableJsonWideLoad( const void * storage, uint32_t width, bool is_signed ) +{ + TableJsonWide v = { 0, 0 }; + if ( width == 16 ) + { + uint64_t half[2]; + memcpy( half, storage, 16 ); + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + v.lo = little ? half[0] : half[1]; + v.hi = little ? half[1] : half[0]; + return v; + } + v.lo = is_signed ? (uint64_t) TableJsonGetSigned( storage, width ) : TableJsonGetRaw( storage, width ); + v.hi = ( is_signed && ( v.lo >> 63 ) != 0 ) ? ~uint64_t( 0 ) : 0; + return v; +} + +inline void TableJsonWideStore( void * storage, uint32_t width, TableJsonWide v ) +{ + if ( width == 16 ) + { + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + uint64_t half[2]; + half[0] = little ? v.lo : v.hi; + half[1] = little ? v.hi : v.lo; + memcpy( storage, half, 16 ); + return; + } + TableJsonSetRaw( storage, width, v.lo ); +} + +// a counted field's companion: a string's length, a bytes' length, a counted +// array's count. Bounded by the declared extent on the way out, so a storage +// invariant a caller broke cannot walk off the end of the array. +inline int32_t TableJsonCount( const void * base, const TableFieldInfo * f ) +{ + if ( !f->counted ) + { + return f->array_bound; + } + int32_t count = 0; + memcpy( &count, (const uint8_t *) base + f->count_offset, sizeof( count ) ); + if ( count < 0 ) { count = 0; } + if ( count > f->array_bound ) { count = f->array_bound; } + return count; +} + +inline void TableJsonSetCount( void * base, const TableFieldInfo * f, int32_t count ) +{ + if ( f->counted ) + { + memcpy( (uint8_t *) base + f->count_offset, &count, sizeof( count ) ); + } +} + +// ---- what a field's kind expects to see in the text ---- +// +// One classifier, consulted by both directions, so a reader and a writer can +// never disagree about a kind's JSON form. 'o' object, 'a' array, 's' +// string, 'n' number, 'b' boolean. +// +// A vocabulary field is spelled by NAME: an enum is one name, a flags mask +// is the array of the names of its set bits. The two are told apart by the +// id column — an enum variant rides under a wire id, a flags BIT never does +// (docs/SPEC-TABLES.md §4), so a name function with no id function is flags. +// +// bytes(N) is the one kind whose element kind does not decide its form: it +// shares u8 with a plain array of u8, and rides as base64. The schema type +// name settles it, and "bytes" is a keyword no declaration can claim. +inline bool TableJsonIsBytes( const TableFieldInfo * f ) +{ + return f->is_array && f->kind == 6 && strcmp( f->type_name, "bytes" ) == 0; +} + +// An ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): its JSON form is an OBJECT +// keyed by variant name, not a positional array, because that is what the +// storage is — one slot per variant, addressed by the variant. +inline bool TableJsonIsKeyed( const TableFieldInfo * f ) +{ + return f->key_name != NULL; +} + +// THE KEY A STORAGE SLOT HOLDS (§2.4, §8): the storage shifts left, so slot i +// holds the key i + 1 and nothing is stored for None. This is the ONE place +// the walker spells the shift. +inline uint64_t TableJsonKeyedSlotKey( int64_t slot ) +{ + return (uint64_t) ( slot + 1 ); +} + +// A slot whose key names a variant of the keying enum. Every slot in +// [0, array_bound) does, unless the enum carries max-headroom variants outside +// a table closure, where a reserved value names nothing and its key id is 0 — +// the reserved id no declared name can fold to (§5). +inline bool TableJsonKeyedSlotValid( const TableFieldInfo * f, int64_t slot ) +{ + return f->key_id( TableJsonKeyedSlotKey( slot ) ) != 0; +} + +inline bool TableJsonIsFlags( const TableFieldInfo * f ) +{ + return f->enum_name != NULL && f->variant_id == NULL; +} + +inline bool TableJsonIsEnum( const TableFieldInfo * f ) +{ + return f->variant_id != NULL && f->arms == NULL; +} + +inline char TableJsonShape( const TableFieldInfo * f ) +{ + if ( TableJsonIsMap( f ) ) return 'o'; // a MAP: an object keyed by the KEY (§2.8) + if ( f->kind == 12 ) return 's'; // string + if ( TableJsonIsBytes( f ) ) return 's'; // bytes: base64 + if ( TableJsonIsKeyed( f ) ) return 'o'; // an object keyed by variant NAME + if ( f->is_array ) return 'a'; + if ( f->arms != NULL ) return 'o'; // union: an object with ONE key + if ( f->kind == 13 ) return 'o'; // nested table or type + if ( f->kind == 17 ) return f->table != NULL ? 'o' : 's'; // a pointer: the pointee's object in place, or null (§16.7); a byte buffer's string (§2.5) + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// the ELEMENT shape of an array field — the same classifier one level down +inline char TableJsonElementShape( const TableFieldInfo * f ) +{ + if ( f->arms != NULL ) return 'o'; // an element of an array of unions: one key, the arm (§2.6) + if ( f->kind == 13 ) return 'o'; + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// A guarded group rides only when its guard reads true — the wire's own +// elision (§4), carried into the text so a text and a wire written from one +// instance say the same thing. The guard is spelled as its branch condition +// over bool fields of the SAME type ("at_rest", "!at_rest", +// "active && has_target"), so evaluating it is a walk of the same +// descriptor. Nothing is inferred in the other direction: reading places +// every key it can name, and the guard is a plain bool key (§16.2). +inline bool TableJsonGuardHolds( const void * base, const TableTypeInfo * info, const char * guard ) +{ + const char * p = guard; + for ( ;; ) + { + while ( *p == ' ' || *p == '&' ) { p++; } + if ( *p == 0 ) { return true; } + bool want = true; + if ( *p == '!' ) { want = false; p++; } + const char * start = p; + while ( *p != 0 && *p != ' ' && *p != '&' ) { p++; } + size_t length = (size_t) ( p - start ); + bool value = false; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( strlen( f->name ) == length && strncmp( f->name, start, length ) == 0 ) + { + value = TableJsonGetRaw( (const uint8_t *) base + f->offset, f->elem_size ) != 0; + break; + } + } + if ( value != want ) { return false; } + } +} + +// ---- writing ---- + +// The writer sink MEASURES when the buffer is NULL and WRITES when it is +// not, over one code path — so measure and write agree byte for byte, the +// wire's invariant (§9) carried across. +struct TableJsonOut +{ + char * buffer; + int64_t capacity; + int64_t offset; + bool overflow; + void * graph; // the pointered write's identity map (§16.7); NULL for a fixed table + + void raw( const char * data, int64_t count ) + { + if ( buffer != NULL ) + { + if ( offset + count > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) count ); + } + offset += count; + } + void put( char c ) { raw( &c, 1 ); } + void text( const char * s ) { raw( s, (int64_t) strlen( s ) ); } + void line( int32_t depth ) + { + put( '\n' ); + for ( int32_t i = 0; i < depth; i++ ) { raw( " ", 2 ); } + } +}; + +inline const char * TableJsonBase64Alphabet() +{ + return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; +} + +inline void TableJsonWriteBase64( TableJsonOut & out, const uint8_t * data, int32_t length ) +{ + const char * alphabet = TableJsonBase64Alphabet(); + out.put( '"' ); + int32_t i = 0; + for ( ; i + 3 <= length; i += 3 ) + { + uint32_t triple = ( uint32_t( data[i] ) << 16 ) | ( uint32_t( data[i+1] ) << 8 ) | uint32_t( data[i+2] ); + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], + alphabet[ ( triple >> 6 ) & 0x3f ], alphabet[ triple & 0x3f ] }; + out.raw( quad, 4 ); + } + if ( i < length ) + { + int32_t left = length - i; + uint32_t triple = uint32_t( data[i] ) << 16; + if ( left == 2 ) { triple |= uint32_t( data[i+1] ) << 8; } + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], '=', '=' }; + if ( left == 2 ) { quad[2] = alphabet[ ( triple >> 6 ) & 0x3f ]; } + out.raw( quad, 4 ); + } + out.put( '"' ); +} + +// One UTF-8 sequence at s, or -1 when the bytes there are not one. Rejects +// the lot: a stray continuation, an overlong form, a surrogate half, and +// anything past U+10FFFF. +inline int32_t TableJsonUtf8( const char * s, int32_t remaining, int32_t * width ) +{ + unsigned char lead = (unsigned char) s[0]; + int32_t want = 0; + int32_t code = 0; + if ( lead < 0x80 ) { *width = 1; return lead; } + else if ( lead >= 0xc2 && lead <= 0xdf ) { want = 2; code = lead & 0x1f; } + else if ( lead >= 0xe0 && lead <= 0xef ) { want = 3; code = lead & 0x0f; } + else if ( lead >= 0xf0 && lead <= 0xf4 ) { want = 4; code = lead & 0x07; } + else { return -1; } + if ( remaining < want ) { return -1; } + for ( int32_t i = 1; i < want; i++ ) + { + unsigned char next = (unsigned char) s[i]; + if ( ( next & 0xc0 ) != 0x80 ) { return -1; } + code = ( code << 6 ) | ( next & 0x3f ); + } + if ( want == 3 && code < 0x800 ) { return -1; } // overlong + if ( want == 4 && code < 0x10000 ) { return -1; } // overlong + if ( code >= 0xd800 && code <= 0xdfff ) { return -1; } // a surrogate half + if ( code > 0x10ffff ) { return -1; } + *width = want; + return code; +} + +// A JSON text MUST be valid UTF-8 (RFC 8259 §8.1). The read path is +// byte-transparent — the wire imposes no encoding (§3) and a string may hold +// anything — so the WRITER is where that obligation is met: a byte that is +// not part of a well-formed sequence is written as U+FFFD, one per bad byte, +// and never raw. A text this walk writes is therefore readable by any +// conforming parser, which a raw byte would not be. The cost is stated +// plainly: for a string holding invalid UTF-8, the round trip is NOT +// byte-identical, because the alternative is emitting a text that is not +// JSON. +inline void TableJsonWriteString( TableJsonOut & out, const char * s, int32_t length ) +{ + static const char hex[] = "0123456789abcdef"; + out.put( '"' ); + for ( int32_t i = 0; i < length; i++ ) + { + unsigned char c = (unsigned char) s[i]; + switch ( c ) + { + case '"': out.raw( "\\\"", 2 ); break; + case '\\': out.raw( "\\\\", 2 ); break; + case '\b': out.raw( "\\b", 2 ); break; + case '\f': out.raw( "\\f", 2 ); break; + case '\n': out.raw( "\\n", 2 ); break; + case '\r': out.raw( "\\r", 2 ); break; + case '\t': out.raw( "\\t", 2 ); break; + default: + if ( c < 0x20 ) + { + char escape[6] = { '\\', 'u', '0', '0', hex[ c >> 4 ], hex[ c & 0xf ] }; + out.raw( escape, 6 ); + } + else if ( c < 0x80 ) + { + out.put( (char) c ); + } + else + { + int32_t width = 0; + if ( TableJsonUtf8( s + i, length - i, &width ) < 0 ) + { + out.raw( "\xef\xbf\xbd", 3 ); // U+FFFD, one per bad byte + } + else + { + out.raw( s + i, width ); + i += width - 1; + } + } + break; + } + } + out.put( '"' ); +} + +inline void TableJsonWriteUnsigned( TableJsonOut & out, uint64_t value ) +{ + char digits[24]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) ( value % 10 ) ); + value /= 10; + } while ( value != 0 ); + char text[24]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); +} + +inline void TableJsonWriteSigned( TableJsonOut & out, int64_t value ) +{ + if ( value < 0 ) + { + out.put( '-' ); + TableJsonWriteUnsigned( out, uint64_t( 0 ) - (uint64_t) value ); + return; + } + TableJsonWriteUnsigned( out, (uint64_t) value ); +} + +// A wide kind writes its raw storage as §16.2's text: a 128-bit integer as a +// decimal integer; a fixed value in WHOLE UNITS as the shortest exact decimal +// with at least one fractional digit (1.0, -0.25), the spelling the schema text +// gives a fixed default. The fraction terminates because a dyadic fraction has +// a finite decimal expansion — at most F digits. +inline void TableJsonWriteWide( TableJsonOut & out, const void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + TableJsonWide v = TableJsonWideLoad( storage, f->elem_size, is_signed ); + if ( is_signed && TableJsonWideNegative( v ) ) + { + out.put( '-' ); + v = TableJsonWideNeg( v ); + } + int frac = f->frac_bits; + TableJsonWide whole = TableJsonWideShr( v, frac ); + char digits[40]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) TableJsonWideDiv( &whole, 10 ) ); + } while ( !TableJsonWideZero( whole ) ); + char text[40]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); + if ( !TableJsonKindFixed( f->kind ) ) { return; } + out.put( '.' ); + // the fraction bits alone: v with everything at and above bit F cleared + TableJsonWide fraction = v; + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + if ( frac == 0 ) { fraction.lo = 0; } + if ( TableJsonWideZero( fraction ) ) + { + out.put( '0' ); + return; + } + while ( !TableJsonWideZero( fraction ) ) + { + // ×10: the digit is what lands at and above bit F, including the + // carry out of 128 bits when F leaves no room for it below + uint32_t carry = TableJsonWideMulAdd( &fraction, 10, 0 ); + uint64_t digit = TableJsonWideShr( fraction, frac ).lo; + if ( frac > 64 ) { digit |= uint64_t( carry ) << ( 128 - frac ); } + out.put( (char) ( '0' + (int) digit ) ); + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + } +} + +// A float writes at the SHORTEST precision that reads back as the same value +// at the field's own width, so a round trip is exact and a text stays +// readable. Non-finite values have no JSON spelling at all, and the writer +// REFUSES rather than losing one silently — the same rule measure and save +// already apply to an enum value no variant names (§5). +inline bool TableJsonWriteFloat( TableJsonOut & out, double value, bool single ) +{ + if ( !TableJsonFinite( value ) ) { return false; } + char text[64]; + int low = single ? 6 : 15; + int high = single ? 9 : 17; + int length = 0; + for ( int digits = low; ; digits++ ) + { + length = snprintf( text, sizeof( text ), "%.*g", digits, value ); + if ( length <= 0 || length >= (int) sizeof( text ) ) { return false; } + if ( digits >= high ) { break; } + // the round-trip check runs BEFORE the decimal point is normalised: + // the token still carries whatever point snprintf just produced + if ( single ) + { + if ( (double) strtof( text, NULL ) == value ) { break; } + } + else + { + if ( strtod( text, NULL ) == value ) { break; } + } + } + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int i = 0; i < length; i++ ) + { + if ( text[i] == point ) { text[i] = '.'; } + } + } + out.raw( text, length ); + return true; +} + +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration writes through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ); + +// one scalar, at one storage address: a nested object, a union, a +// vocabulary, or a number +inline bool TableJsonWriteScalar( TableJsonOut & out, const void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; None is {} + const TableUnionInfo * arms = f->arms(); + uint64_t tag = TableJsonGetRaw( (const uint8_t *) storage + arms->tag_offset, arms->tag_size ); + if ( tag == 0 ) + { + out.raw( "{}", 2 ); + return true; + } + if ( (int64_t) tag > f->enum_max ) + { + return false; // a tag no arm names, exactly as measure refuses it + } + const char * arm = f->enum_name( tag ); + // and refuse on the NAME, not merely on the bound: §16.2 says a value + // no variant NAMES is refused, so the check is the name. Writing + // whatever came back would emit "???", a spelling the reader counts + // as unknown — a silent round-trip loss in place of a refusal. + if ( !TableJsonNamed( arm ) ) { return false; } + out.put( '{' ); + out.line( depth + 1 ); + TableJsonWriteString( out, arm, (int32_t) strlen( arm ) ); + out.raw( ": ", 2 ); + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2): an arm that names + // no declaration carries the FIELD descriptor a field of its type + // would carry, offsets taken inside the union storage (§2.6), so the + // value walks through the field writer one key down. + if ( arms->arms[tag].field != NULL ) + { + if ( !TableJsonWriteField( out, storage, arms->arms[tag].field, depth + 1 ) ) + { + return false; + } + } + else if ( arms->arms[tag].table == NULL ) + { + out.raw( "null", 4 ); // a payload-free arm: the name selects it (§2.6) + } + else if ( !TableJsonWriteValue( out, (const uint8_t *) storage + arms->arms[tag].offset, arms->arms[tag].table, depth + 1 ) ) + { + return false; + } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->kind == 13 ) + { + return TableJsonWriteValue( out, storage, f->table, depth ); + } + if ( TableJsonIsEnum( f ) ) + { + uint64_t value = TableJsonGetRaw( storage, f->elem_size ); + // a value no variant names has no text spelling, exactly as it has no + // wire identity: the writer REFUSES rather than writing None over it, + // the rule measure and save already apply (docs/SPEC-TABLES.md §5) + if ( (int64_t) value > f->enum_max ) { return false; } + if ( value != 0 && f->variant_id( value ) == 0 ) { return false; } + const char * name = f->enum_name( value ); + if ( !TableJsonNamed( name ) ) { return false; } + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + return true; + } + if ( TableJsonIsFlags( f ) ) + { + uint64_t bits = TableJsonGetRaw( storage, f->elem_size ); + if ( bits == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + bool first = true; + for ( int64_t bit = 0; bit < 64; bit++ ) + { + if ( ( bits & ( uint64_t( 1 ) << bit ) ) == 0 ) { continue; } + if ( bit > f->enum_max ) + { + return false; // a bit no variant names has no text spelling + } + const char * name = f->enum_name( (uint64_t) bit ); + if ( !TableJsonNamed( name ) ) { return false; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + } + out.line( depth ); + out.put( ']' ); + return true; + } + switch ( f->kind ) + { + case 1: + out.text( TableJsonGetRaw( storage, f->elem_size ) != 0 ? "true" : "false" ); + return true; + case 10: + { + float v = 0.0f; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, (double) v, true ); + } + case 11: + { + double v = 0.0; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, v, false ); + } + case 2: case 3: case 4: case 5: + TableJsonWriteSigned( out, TableJsonGetSigned( storage, f->elem_size ) ); + return true; + default: + if ( TableJsonKindWide( f->kind ) ) + { + TableJsonWriteWide( out, storage, f ); + return true; + } + TableJsonWriteUnsigned( out, TableJsonGetRaw( storage, f->elem_size ) ); + return true; + } +} + +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ) +{ + const uint8_t * storage = (const uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonWriteMap( out, (const void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } + if ( f->kind == 17 && !f->is_array ) + { + return TableJsonWritePointer( out, storage, f, depth ); + } + if ( f->kind == 17 ) + { + // an ARRAY OF POINTERS (§2.1): the pointer row per element — the + // pointee's object in place, null, or `&node` for a shared one (§16.7) + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWritePointer( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; + } + if ( f->kind == 12 ) + { + TableJsonWriteString( out, (const char *) storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + TableJsonWriteBase64( out, storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + // one entry per SLOT, keyed by the variant that owns it, so inserting + // a variant next season moves nothing in the text either. Slot i holds + // the key i + 1: nothing is stored for None, so nothing is written for it. + out.put( '{' ); + bool first = true; + for ( int64_t slot = 0; slot < f->array_bound; slot++ ) + { + if ( !TableJsonKeyedSlotValid( f, slot ) ) { continue; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + const char * key = f->key_name( TableJsonKeyedSlotKey( slot ) ); + TableJsonWriteString( out, key, (int32_t) strlen( key ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteScalar( out, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + if ( first ) { out.raw( "}", 1 ); return true; } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->is_array ) + { + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWriteScalar( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + out.line( depth ); + out.put( ']' ); + return true; + } + return TableJsonWriteScalar( out, storage, f, depth ); +} + +// One instance's fields, in DECLARATION ORDER, defaults included — a text is +// for people and tools, and a text that elides is a text a reader has to know +// the schema to complete. `any` says whether the object is already open on +// entry — a shared node's `&node` opens it before the fields (§16.7) — and +// whether it is open on return. +inline bool TableJsonWriteFields( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth, bool & any ) +{ + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( f->guard[0] != 0 && !TableJsonGuardHolds( base, info, f->guard ) ) { continue; } + // an ABSENT optional writes no key: presence of the key IS the + // presence (§16.2), so an absent field is an absent key and nothing + // else would read back as absent + if ( f->optional && + TableJsonGetRaw( (const uint8_t *) base + f->present_offset, 1 ) == 0 ) + { + continue; + } + if ( !any ) { out.put( '{' ); } + else { out.put( ',' ); } + any = true; + out.line( depth + 1 ); + TableJsonWriteString( out, f->json, (int32_t) strlen( f->json ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, base, f, depth + 1 ) ) { return false; } + } + return true; +} + +// One instance as one object. The writer carries the reader's depth cap +// (§16.2): a pointer chain nests as deep as it is long (§16.7), and a text the +// writer produced past the cap would be a text the reader refuses. +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { return false; } + bool any = false; + if ( !TableJsonWriteFields( out, base, info, depth, any ) ) { return false; } + if ( !any ) + { + out.raw( "{}", 2 ); + return true; + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- reading ---- + +struct TableJsonIn +{ + const char * text; + int64_t size; + int64_t pos; + TableReport * report; + bool bad; // the text is not JSON: the walk stops and keeps what it placed + void * graph; // the pointered read's builder and label map (§16.7); NULL for a fixed table +}; + +inline void TableJsonSpace( TableJsonIn & in ) +{ + while ( in.pos < in.size ) + { + char c = in.text[in.pos]; + if ( c == ' ' || c == '\t' || c == '\n' || c == '\r' ) { in.pos++; continue; } + // comments are not JSON, and a walk that guessed at one would be + // reading a dialect nobody wrote down + if ( c == '/' ) { in.bad = true; } + return; + } +} + +inline char TableJsonPeek( TableJsonIn & in ) +{ + TableJsonSpace( in ); + return in.pos < in.size ? in.text[in.pos] : 0; +} + +// the shape of the value sitting at the cursor, without consuming it +inline char TableJsonValueShape( TableJsonIn & in ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return 'o'; + case '[': return 'a'; + case '"': return 's'; + case 't': case 'f': return 'b'; + case 'n': return 'z'; + case 0: return 0; + default: return 'n'; + } +} + +inline bool TableJsonLiteral( TableJsonIn & in, const char * word ) +{ + int64_t length = (int64_t) strlen( word ); + if ( in.pos + length > in.size || memcmp( in.text + in.pos, word, (size_t) length ) != 0 ) + { + in.bad = true; + return false; + } + in.pos += length; + return true; +} + +// one \uXXXX escape body; -1 when the four hex digits are not there +inline int TableJsonHex4( TableJsonIn & in ) +{ + if ( in.pos + 4 > in.size ) { return -1; } + int value = 0; + for ( int i = 0; i < 4; i++ ) + { + char c = in.text[in.pos + i]; + int digit; + if ( c >= '0' && c <= '9' ) { digit = c - '0'; } + else if ( c >= 'a' && c <= 'f' ) { digit = c - 'a' + 10; } + else if ( c >= 'A' && c <= 'F' ) { digit = c - 'A' + 10; } + else { return -1; } + value = ( value << 4 ) | digit; + } + in.pos += 4; + return value; +} + +inline int32_t TableJsonEncodeUtf8( uint32_t code, char * unit ) +{ + if ( code < 0x80 ) { unit[0] = (char) code; return 1; } + if ( code < 0x800 ) + { + unit[0] = (char) ( 0xc0 | ( code >> 6 ) ); + unit[1] = (char) ( 0x80 | ( code & 0x3f ) ); + return 2; + } + if ( code < 0x10000 ) + { + unit[0] = (char) ( 0xe0 | ( code >> 12 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( code & 0x3f ) ); + return 3; + } + unit[0] = (char) ( 0xf0 | ( code >> 18 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 12 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[3] = (char) ( 0x80 | ( code & 0x3f ) ); + return 4; +} + +// Scan one JSON string into a caller buffer. Bytes are appended ONE CODE +// POINT AT A TIME — an escape's encoding, or a UTF-8 sequence read whole — +// so a string longer than the field is clamped AT A CODE POINT BOUNDARY and +// never cut through a multi-byte character. Clamping is counted, never +// fatal, exactly as it is on the wire (§4). A NULL destination scans past a +// string without keeping it. +inline bool TableJsonScanString( TableJsonIn & in, char * out, int32_t capacity, int32_t * length ) +{ + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + int32_t placed = 0; + bool clamped = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos]; + if ( c == '"' ) { in.pos++; break; } + char unit[4]; + int32_t unit_length = 0; + if ( c == '\\' ) + { + in.pos++; + if ( in.pos >= in.size ) { in.bad = true; return false; } + char escape = in.text[in.pos++]; + switch ( escape ) + { + case '"': unit[0] = '"'; unit_length = 1; break; + case '\\': unit[0] = '\\'; unit_length = 1; break; + case '/': unit[0] = '/'; unit_length = 1; break; + case 'b': unit[0] = '\b'; unit_length = 1; break; + case 'f': unit[0] = '\f'; unit_length = 1; break; + case 'n': unit[0] = '\n'; unit_length = 1; break; + case 'r': unit[0] = '\r'; unit_length = 1; break; + case 't': unit[0] = '\t'; unit_length = 1; break; + case 'u': + { + int high = TableJsonHex4( in ); + if ( high < 0 ) { in.bad = true; return false; } + uint32_t code = (uint32_t) high; + if ( high >= 0xd800 && high <= 0xdbff && in.pos + 2 <= in.size && + in.text[in.pos] == '\\' && in.text[in.pos + 1] == 'u' ) + { + int64_t mark = in.pos; + in.pos += 2; + int low = TableJsonHex4( in ); + if ( low >= 0xdc00 && low <= 0xdfff ) + { + code = 0x10000 + ( ( (uint32_t) high - 0xd800 ) << 10 ) + ( (uint32_t) low - 0xdc00 ); + } + else + { + in.pos = mark; // a lone lead surrogate rides as itself + } + } + // a surrogate half that never found its partner has no + // UTF-8 encoding: encoding it anyway would manufacture + // CESU-8 — invalid UTF-8 — out of input that was valid + // JSON, so it reads as the replacement character + if ( code >= 0xd800 && code <= 0xdfff ) { code = 0xfffd; } + unit_length = TableJsonEncodeUtf8( code, unit ); + break; + } + default: in.bad = true; return false; + } + } + else if ( (unsigned char) c < 0x20 ) + { + in.bad = true; // a raw control character is not a JSON string body + return false; + } + else + { + // a UTF-8 sequence read WHOLE, so the clamp below can only land + // between code points. Only bytes that ACTUALLY look like + // continuations are taken: the wire imposes no encoding (§3), so + // a string may legitimately hold a stray lead byte, and one at + // the end of a text must not swallow the closing quote. + unsigned char lead = (unsigned char) c; + int32_t want = 1; + if ( ( lead & 0xe0 ) == 0xc0 ) { want = 2; } + else if ( ( lead & 0xf0 ) == 0xe0 ) { want = 3; } + else if ( ( lead & 0xf8 ) == 0xf0 ) { want = 4; } + unit[0] = c; + in.pos++; + unit_length = 1; + while ( unit_length < want && in.pos < in.size && + ( (unsigned char) in.text[in.pos] & 0xc0 ) == 0x80 ) + { + unit[unit_length++] = in.text[in.pos++]; + } + } + if ( out == NULL ) + { + placed += unit_length; // measured and not kept: a byte buffer's read sizes its node this way (§2.5) + } + else if ( placed + unit_length <= capacity ) + { + memcpy( out + placed, unit, (size_t) unit_length ); + placed += unit_length; + } + else + { + clamped = true; + } + } + if ( clamped ) { in.report->clamped++; } + if ( length != NULL ) { *length = placed; } + return true; +} + +// the numeric token at the cursor, copied out whole; false = not a number +// Scan one number, to JSON's OWN grammar (RFC 8259 §6) and not to a run of +// number-ish characters: +// +// number = [ "-" ] int [ frac ] [ exp ] +// int = "0" / ( digit1-9 *digit ) +// frac = "." 1*digit +// exp = ( "e" / "E" ) [ "-" / "+" ] 1*digit +// +// Scanning the production is what makes a typo in an authoring file a +// DIAGNOSTIC rather than a value: "1-2" scans as 1 and leaves "-2" where the +// object expects a comma, so the text is malformed — which is what §16.2 +// already promises. A permissive scan would hand "1-2" to a digit loop and +// report a clamp, and a config pipeline would never hear about it. Leading +// "+", leading zeros, ".5" and "3." are not JSON either. +inline bool TableJsonWalkNumber( TableJsonIn & in, bool * integral ) +{ + TableJsonSpace( in ); + bool whole = true; + if ( in.pos < in.size && in.text[in.pos] == '-' ) { in.pos++; } + // int: a lone zero, or a non-zero digit and any digits after it + if ( in.pos >= in.size ) { return false; } + if ( in.text[in.pos] == '0' ) + { + in.pos++; + } + else if ( in.text[in.pos] >= '1' && in.text[in.pos] <= '9' ) + { + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + } + else + { + return false; + } + // frac + if ( in.pos < in.size && in.text[in.pos] == '.' ) + { + in.pos++; + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + // exp + if ( in.pos < in.size && ( in.text[in.pos] == 'e' || in.text[in.pos] == 'E' ) ) + { + in.pos++; + if ( in.pos < in.size && ( in.text[in.pos] == '-' || in.text[in.pos] == '+' ) ) { in.pos++; } + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + *integral = whole; + return true; +} + +// the same production, with the token kept for conversion +inline bool TableJsonScanNumber( TableJsonIn & in, char * token, int32_t capacity, int32_t * length, bool * integral ) +{ + TableJsonSpace( in ); + int64_t start = in.pos; + if ( !TableJsonWalkNumber( in, integral ) ) { return false; } + int64_t count = in.pos - start; + if ( count <= 0 || count >= capacity ) { return false; } + memcpy( token, in.text + start, (size_t) count ); + token[count] = 0; + *length = (int32_t) count; + return true; +} + +// the token's exact double, through the runtime's own converter — which +// speaks the LOCALE's decimal point, so the token crosses back over that +// character on its way in +inline double TableJsonTokenDouble( const char * token, int32_t length, bool single ) +{ + char work[kTableJsonMaxNumber]; + memcpy( work, token, (size_t) length ); + work[length] = 0; + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int32_t i = 0; i < length; i++ ) + { + if ( work[i] == '.' ) { work[i] = point; } + } + } + if ( single ) { return (double) strtof( work, NULL ); } + return strtod( work, NULL ); +} + +// the token's exact integer, parsed digit by digit so no width and no +// locale can move it. Saturation is reported as a clamp, the wire's rule for +// a value outside what the reader can hold (§4). +inline int64_t TableJsonTokenInteger( const char * token, int32_t length, bool is_signed, bool * saturated ) +{ + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) + { + negative = token[i] == '-'; + i++; + } + uint64_t magnitude = 0; + bool over = false; + for ( ; i < length; i++ ) + { + uint64_t digit = (uint64_t) ( token[i] - '0' ); + if ( magnitude > ( UINT64_MAX - digit ) / 10 ) { over = true; break; } + magnitude = magnitude * 10 + digit; + } + if ( !is_signed ) + { + // -0 IS zero, and clamping it would report an event that did not + // happen; only a real negative magnitude is out of range here + if ( negative ) { *saturated = magnitude != 0; return 0; } + if ( over ) { *saturated = true; return (int64_t) UINT64_MAX; } + *saturated = false; + return (int64_t) magnitude; + } + if ( negative ) + { + if ( over || magnitude > ( uint64_t( 1 ) << 63 ) ) { *saturated = true; return INT64_MIN; } + *saturated = false; + if ( magnitude == ( uint64_t( 1 ) << 63 ) ) { return INT64_MIN; } + return -(int64_t) magnitude; + } + if ( over || magnitude > (uint64_t) INT64_MAX ) { *saturated = true; return INT64_MAX; } + *saturated = false; + return (int64_t) magnitude; +} + +// A number token into a wide kind's raw storage (docs/SPEC-TABLES.md §16.2). A +// 128-bit integer takes any token whose VALUE is integral; a fixed field any +// token whose value is EXACTLY representable in its Q I.F — a finer fraction +// is the wrong shape for the field, counted as a kind mismatch and never +// rounded, the rule SPEC.md §4.6 gives a fixed default. A magnitude past 128 +// bits saturates and counts as a clamp, as an int64 field saturates at +// INT64_MAX; the declared range clamps after it, on the RAW scale, as it does +// for every bounded scalar. +// +// The token is normalized to its digits with the decimal point after "point" +// of them. An integer part past 40 digits is above 2^128 whatever the digits +// are, and a value below 10^-40 is finer than 2^-127, the finest fraction any +// F can spell — so outside that band the answer is known without the +// arithmetic, and a token spelling 1e999999999 costs nothing to refuse. +inline bool TableJsonReadWide( TableJsonIn & in, const char * token, int32_t length, void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + int frac = f->frac_bits; + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { negative = token[i] == '-'; i++; } + const char * int_digits = token + i; + int32_t int_len = 0; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { int_len++; i++; } + const char * frac_digits = token + i; + int32_t frac_len = 0; + if ( i < length && token[i] == '.' ) + { + i++; + frac_digits = token + i; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { frac_len++; i++; } + } + int64_t exp = 0; + if ( i < length && ( token[i] == 'e' || token[i] == 'E' ) ) + { + i++; + bool exp_negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { exp_negative = token[i] == '-'; i++; } + while ( i < length && token[i] >= '0' && token[i] <= '9' ) + { + if ( exp < 100000 ) { exp = exp * 10 + ( token[i] - '0' ); } + i++; + } + if ( exp_negative ) { exp = -exp; } + } + // the digits, with the point after "point" of them; leading and trailing + // zeros stripped. digit( k ) reads the k-th of the int and frac runs. + int32_t start = 0, end = int_len + frac_len; + int64_t point = int_len + exp; + while ( start < end && ( start < int_len ? int_digits[start] : frac_digits[start - int_len] ) == '0' ) { start++; point--; } + while ( end > start && ( end - 1 < int_len ? int_digits[end - 1] : frac_digits[end - 1 - int_len] ) == '0' ) { end--; } + + TableJsonWide raw = { 0, 0 }; + bool saturated = false; + TableJsonWide signed_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) >> 1 }; + TableJsonWide signed_min = { 0, uint64_t( 1 ) << 63 }; + TableJsonWide unsigned_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) }; + if ( start == end ) + { + // zero, and -0 IS zero + } + else if ( point > 40 ) + { + saturated = true; + if ( !negative ) { raw = is_signed ? signed_max : unsigned_max; } + else if ( is_signed ) { raw = signed_min; } + } + else if ( point < -40 ) + { + in.report->kind_mismatch++; // finer than any F can spell + return true; + } + else + { + // the fraction FIRST, so an inexact value is the wrong shape whatever + // its magnitude: its digits, with the zeros a negative point puts in + // front, doubled F times; each doubling's carry is the next bit, and + // the value is exact iff nothing is left after the last one + char fd[kTableJsonMaxNumber + 48]; + int32_t fn = 0; + for ( int64_t z = point; z < 0; z++ ) { fd[fn++] = 0; } + for ( int32_t k = (int32_t) ( point > 0 ? point : 0 ) + start; k < end; k++ ) + { + fd[fn++] = (char) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ); + } + TableJsonWide fraction = { 0, 0 }; + for ( int b = 0; b < frac; b++ ) + { + int carry = 0; + for ( int32_t k = fn - 1; k >= 0; k-- ) + { + int d = fd[k] * 2 + carry; + fd[k] = (char) ( d % 10 ); + carry = d / 10; + } + fraction = TableJsonWideShl( fraction, 1 ); + fraction.lo |= (uint64_t) carry; + } + for ( int32_t k = 0; k < fn; k++ ) + { + if ( fd[k] != 0 ) + { + in.report->kind_mismatch++; + return true; + } + } + // then the whole part, saturating past 128 bits + TableJsonWide whole = { 0, 0 }; + for ( int64_t k = start; k < start + point && !saturated; k++ ) + { + uint32_t digit = k < end ? (uint32_t) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ) : 0; + if ( TableJsonWideMulAdd( &whole, 10, digit ) != 0 ) { saturated = true; } + } + if ( !saturated && frac > 0 && !TableJsonWideZero( TableJsonWideShr( whole, 128 - frac ) ) ) { saturated = true; } + if ( !saturated ) + { + raw = TableJsonWideShl( whole, frac ); + raw.lo |= fraction.lo; + raw.hi |= fraction.hi; + } + if ( is_signed ) + { + if ( !saturated && !negative && TableJsonWideNegative( raw ) ) { saturated = true; } + if ( !saturated && negative && TableJsonWideCompare( raw, signed_min, false ) > 0 ) { saturated = true; } + if ( saturated ) { raw = negative ? signed_min : signed_max; } + else if ( negative ) { raw = TableJsonWideNeg( raw ); } + } + else + { + if ( saturated ) { raw = unsigned_max; } + if ( negative && !TableJsonWideZero( raw ) ) { raw.lo = 0; raw.hi = 0; saturated = true; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->wide != NULL ) + { + TableJsonWide lo = { f->wide->lo[0], f->wide->lo[1] }; + TableJsonWide hi = { f->wide->hi[0], f->wide->hi[1] }; + if ( TableJsonWideCompare( raw, lo, is_signed ) < 0 ) { raw = lo; in.report->clamped++; } + else if ( TableJsonWideCompare( raw, hi, is_signed ) > 0 ) { raw = hi; in.report->clamped++; } + } + TableJsonWideStore( storage, f->elem_size, raw ); + return true; +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ); + +inline bool TableJsonSkipContainer( TableJsonIn & in, char close, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; // the opening bracket + bool first = true; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == close ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + if ( close == '}' ) + { + // the key is kept, because a skipped OBJECT may still be a + // pointer's: an `&node` opening it names a node the storage could + // not hold, and the numbering has to survive the drop (§16.7). + // Anywhere but first, the prefix is the reserved key out of place + // — in a pointered unit; a fixed unit skips the value whole. + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( key[0] == '&' && in.graph != NULL ) + { + if ( !first ) { in.report->malformed = true; in.bad = true; return false; } + if ( !TableJsonSkippedAmpersand( in, key, depth ) ) { return false; } + first = false; + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } + } + first = false; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return TableJsonSkipContainer( in, '}', depth ); + case '[': return TableJsonSkipContainer( in, ']', depth ); + case '"': return TableJsonScanString( in, NULL, 0, NULL ); + case 't': return TableJsonLiteral( in, "true" ); + case 'f': return TableJsonLiteral( in, "false" ); + case 'n': return TableJsonLiteral( in, "null" ); + case 0: in.bad = true; return false; + default: + { + // consumed, never converted: skipping needs no buffer, and this + // is the one walk a hostile text drives to the depth cap. It is + // the SAME production the value path scans, so an unknown key + // cannot smuggle past a number a named key would refuse. + bool integral = false; + if ( !TableJsonWalkNumber( in, &integral ) ) { in.bad = true; return false; } + return true; + } + } +} + +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration reads through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ); + +// place one scalar at one storage address +inline bool TableJsonReadScalar( TableJsonIn & in, void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; {} is None, and + // two keys is a text this walk will not guess at + const TableUnionInfo * arms = f->arms(); + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, 0 ); + if ( TableJsonPeek( in ) == '}' ) { in.pos++; return true; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t tag = 0; + for ( int64_t t = 1; t <= f->enum_max; t++ ) + { + if ( strcmp( f->enum_name( (uint64_t) t ), key ) == 0 ) { tag = t; break; } + } + if ( tag == 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + void * payload = (uint8_t *) storage + arms->arms[tag].offset; + const TableFieldInfo * arm = arms->arms[tag].field; + bool placed = true; + if ( arm != NULL ) + { + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2). A value of + // the wrong shape for that row is a KIND MISMATCH: the union + // reads None, the event is counted, and the enclosing object + // continues — the rule a FIELD's value lives under, one key + // down. A pointer arm's null is a null pointer, not a shape + // error, exactly as a pointer field's is (§16.7). + char got = TableJsonValueShape( in ); + if ( arm->kind == 17 && !arm->is_array && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + memset( payload, 0, (size_t) arms->arms[tag].size ); + } + else if ( got != TableJsonShape( arm ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( arm->kind == 17 && !arm->is_array ) + { + // A POINTER ARM'S VALUE IS THE POINTEE IN PLACE, or a + // node reference to one (§16.7) — the read a pointer + // FIELD takes, which is not the scalar walk + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadPointer( in, payload, arm, depth + 1 ) ) { return false; } + } + else + { + // SELECTION ZERO-ESTABLISHES THE ARM (SPEC §5): an arm + // takes no specified default, so zero is the establish + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadField( in, storage, arm, depth + 1 ) ) { return false; } + } + } + else if ( arms->arms[tag].table != NULL ) + { + if ( TableJsonValueShape( in ) != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else + { + arms->arms[tag].table->reset( payload ); + if ( !TableJsonReadTable( in, payload, arms->arms[tag].table, depth + 1 ) ) { return false; } + } + } + else + { + // A PAYLOAD-FREE ARM'S VALUE IS null (§2.6): the arm name + // selects it and there is nothing to place + if ( TableJsonValueShape( in ) != 'z' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( !TableJsonLiteral( in, "null" ) ) + { + return false; + } + } + if ( placed ) + { + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, (uint64_t) tag ); + } + } + char c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; // a second key: a one-of with two arms is not a value + return false; + } + if ( f->kind == 13 ) + { + f->table->reset( storage ); + return TableJsonReadTable( in, storage, f->table, depth + 1 ); + } + if ( TableJsonIsEnum( f ) ) + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + for ( int64_t v = 0; v <= f->enum_max; v++ ) + { + if ( strcmp( f->enum_name( (uint64_t) v ), name ) == 0 ) + { + TableJsonSetRaw( storage, f->elem_size, (uint64_t) v ); + return true; + } + } + // a name this build cannot name reads as None and counts as unknown, + // exactly as an unknown variant id does on the wire (§4) + TableJsonSetRaw( storage, f->elem_size, 0 ); + in.report->unknown++; + return true; + } + if ( TableJsonIsFlags( f ) ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + uint64_t bits = 0; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( c != '"' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + bool found = false; + for ( int64_t bit = 0; bit <= f->enum_max; bit++ ) + { + if ( strcmp( f->enum_name( (uint64_t) bit ), name ) == 0 ) + { + bits |= uint64_t( 1 ) << bit; + found = true; + break; + } + } + if ( !found ) { in.report->unknown++; } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + TableJsonSetRaw( storage, f->elem_size, bits ); + return true; + } + if ( f->kind == 1 ) + { + char c = TableJsonPeek( in ); + if ( c == 't' ) { if ( !TableJsonLiteral( in, "true" ) ) { return false; } TableJsonSetRaw( storage, f->elem_size, 1 ); return true; } + if ( !TableJsonLiteral( in, "false" ) ) { return false; } + TableJsonSetRaw( storage, f->elem_size, 0 ); + return true; + } + char token[kTableJsonMaxNumber]; + int32_t length = 0; + bool integral = false; + if ( !TableJsonScanNumber( in, token, kTableJsonMaxNumber, &length, &integral ) ) + { + in.bad = true; + return false; + } + if ( TableJsonKindWide( f->kind ) ) + { + return TableJsonReadWide( in, token, length, storage, f ); + } + if ( f->kind == 10 || f->kind == 11 ) + { + bool single = f->kind == 10; + double value = TableJsonTokenDouble( token, length, single ); + // A magnitude the field's format cannot hold is the WRONG SHAPE for + // the kind, and it never reaches storage: 1e400 is not a float64 and + // 1e300 is not a float32. Storing the infinity the conversion + // produced would leave an instance this walk called CLEAN that + // ToJsonMeasure then refuses forever (a non-finite float has no JSON + // spelling), and §16.1's one invariant is that a text which reads + // clean writes back. + if ( !TableJsonFinite( value ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( f->has_range ) + { + if ( value < f->range_min ) { value = f->range_min; in.report->clamped++; } + else if ( value > f->range_max ) { value = f->range_max; in.report->clamped++; } + } + if ( single ) + { + float narrow = (float) value; + if ( !TableJsonFinite( (double) narrow ) ) + { + in.report->kind_mismatch++; + return true; + } + memcpy( storage, &narrow, sizeof( narrow ) ); + } + else + { + memcpy( storage, &value, sizeof( value ) ); + } + return true; + } + // JSON HAS ONE NUMBER TYPE. 2.0 IS the integer 2 and 1e3 IS 1000, and a + // library that round-trips numbers through a double emits them that way — + // this walker's own float writer emits 1e+21. So an integer field takes + // any number whose VALUE is integral, however it was spelled; only a + // genuinely fractional value is the wrong shape for it. + bool is_signed = f->kind >= 2 && f->kind <= 5; + bool saturated = false; + int64_t value = 0; + if ( integral ) + { + value = TableJsonTokenInteger( token, length, is_signed, &saturated ); + } + else + { + double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( is_signed ) + { + if ( d >= 9223372036854775808.0 ) { value = INT64_MAX; saturated = true; } + else if ( d < -9223372036854775808.0 ) { value = INT64_MIN; saturated = true; } + else if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) d; } + } + else + { + if ( d < 0.0 ) + { + // a negative for an unsigned field clamps to zero, as the + // exact digit path already does + if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + value = 0; + saturated = true; + } + else if ( d >= 18446744073709551616.0 ) { value = (int64_t) UINT64_MAX; saturated = true; } + else if ( d != (double) (uint64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) (uint64_t) d; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->has_range ) + { + if ( (double) value < f->range_min ) { value = (int64_t) f->range_min; in.report->clamped++; } + else if ( (double) value > f->range_max ) { value = (int64_t) f->range_max; in.report->clamped++; } + } + // the field's own storage width is the last bound: a value past it + // clamps rather than wrapping, which is what the wire does too + if ( f->elem_size < 8 ) + { + if ( is_signed ) + { + int64_t high = ( int64_t( 1 ) << ( f->elem_size * 8 - 1 ) ) - 1; + int64_t low = -high - 1; + if ( value > high ) { value = high; in.report->clamped++; } + else if ( value < low ) { value = low; in.report->clamped++; } + } + else + { + uint64_t high = ( uint64_t( 1 ) << ( f->elem_size * 8 ) ) - 1; + if ( value < 0 ) { value = 0; in.report->clamped++; } + else if ( (uint64_t) value > high ) { value = (int64_t) high; in.report->clamped++; } + } + } + // at eight bytes the storage IS the parser's width, and an unsigned value + // past INT64_MAX rides here as a negative int64 by design — the token + // parser already turned a NEGATIVE token for an unsigned field into a + // clamped zero, so there is nothing left to bound. + TableJsonSetRaw( storage, f->elem_size, (uint64_t) value ); + return true; +} + +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ) +{ + uint8_t * storage = (uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonReadMap( in, (void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + + if ( f->kind == 12 ) + { + int32_t length = 0; + if ( !TableJsonScanString( in, (char *) storage, f->array_bound, &length ) ) { return false; } + storage[length] = 0; + TableJsonSetCount( base, f, length ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + // base64 decodes STRAIGHT INTO the field's storage, six bits at a + // time — no window, no temporary, so a bytes(N) of any declared + // extent reads the same way. A base64 body carries no escapes, so a + // backslash in one is simply not an alphabet character. + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + memset( storage, 0, (size_t) f->array_bound ); + TableJsonSetCount( base, f, 0 ); + const char * alphabet = TableJsonBase64Alphabet(); + int32_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + bool clamped = false; + bool malformed = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + const char * at = c != 0 ? strchr( alphabet, c ) : NULL; + if ( at == NULL ) { malformed = true; continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( at - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < f->array_bound ) + { + storage[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); + } + else + { + clamped = true; + } + } + } + if ( malformed ) + { + // a body that is not base64 is the wrong shape for the kind: the + // field keeps its default and the event is counted + in.report->kind_mismatch++; + return true; + } + if ( clamped ) { in.report->clamped++; } + TableJsonSetCount( base, f, placed ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + // every slot back to its declared defaults first, so a key the text + // omits keeps them and a repeated field key cannot leave an earlier + // occurrence's slots standing + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + void * slot = storage + (int64_t) i * f->elem_size; + if ( f->kind == 13 ) { f->table->reset( slot ); } + else { memset( slot, 0, (size_t) f->elem_size ); } + } + char shape = TableJsonElementShape( f ); + // A KEYED OBJECT'S KEYS ARE KEYS: a variant named twice is a duplicate + // key like any other, last-wins and counted (§16.2). Tracked the way + // a table's own field keys are — a bounded, allocation-free bitmask; + // a vocabulary wider than this still reads, its repeats simply stop + // being counted. + uint64_t seen[8] = {}; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t slot = -1; + for ( int64_t v = 0; v < f->array_bound; v++ ) + { + // nothing is stored for None, so "None" finds no slot and is + // an unknown key like any other name this reader cannot place + if ( !TableJsonKeyedSlotValid( f, v ) ) { continue; } + if ( strcmp( f->key_name( TableJsonKeyedSlotKey( v ) ), key ) == 0 ) { slot = v; break; } + } + if ( slot >= 0 && slot < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( slot & 63 ); + if ( ( seen[slot >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[slot >> 6] |= bit; + } + if ( slot < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; + } + if ( f->is_array ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + // LAST WINS has to be true of a repeated ARRAY key too, and it is + // wire-visible: a fixed array writes every slot, so a second, shorter + // occurrence overlaying a prefix would leave the first occurrence's + // tail standing. The field goes back to its declared defaults before + // this occurrence's elements are placed — the re-establishment a nested + // table and a union arm already get. A table element's defaults are + // its own (the reset hook); every other element kind's storage + // default is zero, which is what the generated array declares. + if ( f->kind == 13 ) + { + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + f->table->reset( storage + (int64_t) i * f->elem_size ); + } + } + else + { + memset( storage, 0, (size_t) f->array_bound * (size_t) f->elem_size ); + } + TableJsonSetCount( base, f, 0 ); + int32_t placed = 0; + char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( placed >= f->array_bound ) + { + // more elements than the reader's bound: the bounded prefix + // is kept and the excess counts, the wire's rule (§4) + in.report->clamped++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( f->kind == 17 ) + { + // an element of an ARRAY OF POINTERS (§2.1): null is a null slot, an + // object is the pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( storage + (int64_t) placed * f->elem_size, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed++; + } + else + { + if ( !TableJsonReadScalar( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + // a fixed array's tail keeps the defaults the prefill left there, + // exactly as a short wire count does + TableJsonSetCount( base, f, placed ); + return true; + } + return TableJsonReadScalar( in, storage, f, depth ); +} + +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ); + +// ONE table object: keys are field keys, unknown ones are skipped and +// counted, a repeated key is last-wins and counted. The instance is already +// at its declared defaults when this is entered, so a key the text never +// mentions keeps the default an absent field takes on the wire (§4). +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + return TableJsonReadTableKeys( in, base, info, depth, NULL ); +} + +// The keys of an object whose brace is already consumed. A pointer's object +// opens the same way a table's does, but its FIRST key may be `&node` (§16.7) +// and the adapter that reads it has to scan the key to know — so it hands the +// key it scanned in as `first_key`, with the colon consumed, and this places +// it before scanning the rest. +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ) +{ + // duplicate tracking, bounded and allocation-free: a table with more + // fields than this still reads, its repeats simply stop being counted + uint64_t seen[8] = {}; + for ( ;; ) + { + char key[kTableJsonMaxKey]; + char c = 0; + if ( first_key != NULL ) + { + memcpy( key, first_key, strlen( first_key ) + 1 ); // scanned into a buffer this size by the caller + first_key = NULL; + } + else + { + c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + } + int32_t index = -1; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + if ( strcmp( info->fields[i].json, key ) == 0 ) { index = i; break; } + } + if ( key[0] == '&' ) + { + // THE AMPERSAND PREFIX IS RESERVED TO THE FORM (docs/SPEC-TABLES.md + // §16.7). No declaration may take a key beginning with it, so this + // is never a field this build lacks — it is the sharing construct + // somewhere it cannot stand: `&node` is the FIRST key of a pointer's + // object and nothing else, and the adapter that reads a pointer + // has consumed it before these keys are read. MALFORMED, refused + // and counted; never counted as unknown, never skipped. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( index < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + const TableFieldInfo * f = &info->fields[index]; + if ( index < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( index & 63 ); + if ( ( seen[index >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[index >> 6] |= bit; + } + // PRESENCE OF THE KEY IS THE PRESENCE (§16.2): reaching this line + // is the key being present, so an optional is set present + // whatever its value — with one exception the page names: a JSON + // null, which reads as ABSENT rather than as a value. + char got = TableJsonValueShape( in ); + if ( f->kind == 17 && !f->is_array ) + { + // a pointer: null is a null pointer, an object is the pointee + // in place or an `&node` reference to one (§16.7), a string is + // a BYTE BUFFER's bytes (§2.5), and anything else is the wrong + // shape for the kind + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) base + f->offset, f->elem_size, 0 ); + } + else if ( got != TableJsonShape( f ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) base + f->offset, f, depth ) ) + { + return false; + } + } + else if ( f->optional && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + // absent, and back at its defaults: a repeated key whose last + // occurrence is null must not leave an earlier value standing + if ( f->table != NULL ) { f->table->reset( (uint8_t *) base + f->offset ); } + else { memset( (uint8_t *) base + f->offset, 0, (size_t) f->elem_size ); } + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 0 ); + } + else + { + if ( got != TableJsonShape( f ) ) + { + // the wrong JSON type for the kind: skipped, never coerced + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, base, f, depth ) ) + { + return false; + } + if ( f->optional ) + { + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 1 ); + } + } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +// ---- the two entry points the per-table wrappers name ---- + +inline bool TableJsonRead( void * value, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = NULL; + info->reset( value ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, value, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +inline int64_t TableJsonWrite( const void * value, const TableTypeInfo * info, char * buffer, int64_t capacity ) +{ + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = NULL; + if ( !TableJsonWriteValue( out, value, info, 0 ) ) { return -1; } + // THE CANONICAL TEXT ENDS WITH EXACTLY ONE NEWLINE (docs/SPEC-TABLES.md + // §16.1). Every writer emits it — this walk, the C# walk and + // "schema unpack" — and every reader accepts a text with or without one, + // because the trailing whitespace a read already skips is what makes the + // two the same text. It is a byte of the FORM rather than a file + // convention: a text that is written to a file, pasted into a diff and + // handed back through a pipe has to be one text in all three places, and a + // buffer whose last byte is a closing brace is the one shape that is not. + out.put( '\n' ); + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json walk: end ---- + +// ---- json graph walk: begin ---- +// +// THE VARIABLE CLASS's half of the text form (docs/SPEC-TABLES.md §16.7). The +// walk above places every kind but one; this defines the three adapters it +// calls for that one, and the two entry points a pointered table's wrappers +// name. The text is the fixed class's — a pointee is an object in place — and a +// node named more than once carries `&node`: defined once, with its fields, +// and referenced after by `{ "&node": N }` alone. + +// ---- the identity map ---- +// +// ONE map shape serves both directions. Writing keys it by a node's ADDRESS and +// counts the slots that name the node, so the second pass knows at a node's +// first occurrence whether it will be named again; reading keys it by the +// text's own label and answers the node it defined. Open addressing, a +// multiply-shift hash and quadrupling growth — TablePackMap's shape (§6.2), on +// the same terms: proportional to nodes, never to bytes, on the authoring +// side, and released before the call returns. + +struct TableJsonGraphEntry +{ + uint64_t key; // a node's address (write) or a label (read); 0 is an empty slot + int64_t count; // write: how many slots name this node + int64_t label; // write: the `&node` label assigned at its first write, 0 until then + uint8_t open; // the descent is still open: a reference here is a cycle (write), a self-reference (read) + uint32_t node; // read: the node's arena offset; 0 for a definition the reader dropped + const TableTypeInfo * type; // read: the node's table; NULL for a dropped one +}; + +struct TableJsonGraphMap +{ + TableJsonGraphEntry * entries; + int64_t capacity; // a power of two, or zero while empty + int64_t count; + TableAllocator allocator; // the caller's pair (§6.5): the builder's on read, the one handed to ToJson on write +}; + +inline void TableJsonGraphMapInit( TableJsonGraphMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TableJsonGraphMapShutdown( TableJsonGraphMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TableJsonGraphMapInit( map, map.allocator ); +} + +inline int64_t TableJsonGraphMapSlot( const TableJsonGraphMap & map, uint64_t key ) +{ + uint64_t hash = key * 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != 0 && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TableJsonGraphEntry * TableJsonGraphMapFind( TableJsonGraphMap & map, uint64_t key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +inline bool TableJsonGraphMapGrow( TableJsonGraphMap & map ) +{ + TableJsonGraphMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 64; + grown.count = 0; + grown.entries = (TableJsonGraphEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TableJsonGraphEntry ) ); // zeroed, by the pair's contract + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == 0 ) { continue; } + grown.entries[ TableJsonGraphMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// the entry for a key, made if it was not there; `taken` says which. NULL is the +// allocator refusing, and the walk refuses with it. +inline TableJsonGraphEntry * TableJsonGraphMapReach( TableJsonGraphMap & map, uint64_t key, bool & taken ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TableJsonGraphMapGrow( map ) ) { return NULL; } + } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + taken = entry->key != key; + if ( taken ) + { + entry->key = key; + map.count++; + } + return entry; +} + +// ---- reading: into a builder ---- + +struct TableJsonGraphIn +{ + TableWorker * worker; // where every node comes from + TableJsonGraphMap labels; // a label -> the node it defined +}; + +// `&node`'s value, the LABEL: a positive integer spelled as one — digits, no sign, no +// fraction, no exponent, no leading zero (§16.7). Anything else is malformed. +inline bool TableJsonScanLabel( TableJsonIn & in, uint64_t & label ) +{ + TableJsonSpace( in ); + if ( in.pos >= in.size || in.text[in.pos] < '1' || in.text[in.pos] > '9' ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + uint64_t value = 0; + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) + { + uint64_t digit = (uint64_t) ( in.text[in.pos] - '0' ); + if ( value > ( UINT64_MAX - digit ) / 10 ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + value = value * 10 + digit; + in.pos++; + } + label = value; + return true; +} + +// A BYTE BUFFER's text (docs/SPEC-TABLES.md §2.5, §16.2): a string. For a +// *string the string's bytes become the blob; for a *bytes the string is base64 +// and its decoded bytes do. The blob is allocated at EXACTLY the decoded +// length — the string is scanned once without keeping it to learn the length, +// and once into the node — so a blob of any size reads with no window and no +// bound to clamp against. A *bytes body that is not base64 is the wrong shape +// for the kind: the reference stays null and the event is counted. +inline bool TableJsonReadBlob( TableJsonIn & in, void * slot, const TableFieldInfo * f ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + TableRef * ref = (TableRef *) slot; + ref->value = 0; + if ( strcmp( f->type_name, "string" ) == 0 ) + { + const int64_t mark = in.pos; + int32_t length = 0; + if ( !TableJsonScanString( in, NULL, 0, &length ) ) { return false; } + in.pos = mark; + char * data = TableStringEmplace( *graph->worker, *ref, NULL, (int64_t) length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int32_t placed = 0; + return TableJsonScanString( in, data, length, &placed ); + } + // base64: the alphabet characters decide the length, six bits apiece + const char * alphabet = TableJsonBase64Alphabet(); + const int64_t mark = in.pos + 1; + int64_t symbols = 0; + bool malformed = false; + in.pos++; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + if ( c == 0 || strchr( alphabet, c ) == NULL ) { malformed = true; continue; } + symbols++; + } + if ( malformed ) + { + in.report->kind_mismatch++; + return true; + } + const int64_t length = ( symbols * 6 ) / 8; + uint8_t * data = TableBytesEmplace( *graph->worker, *ref, length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int64_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + for ( int64_t at = mark; ; at++ ) + { + char c = in.text[at]; + if ( c == '"' ) { break; } + const char * symbol = c != '=' ? strchr( alphabet, c ) : NULL; + if ( symbol == NULL ) { continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( symbol - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < length ) { data[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); } + } + } + return true; +} + +// A pointer's object. Its FIRST key decides what it is: `&node` naming a label not +// yet defined, with fields after it, is a DEFINITION; `&node` naming one already +// defined, alone, is a REFERENCE; any other key is a node named once, its +// object in place. The node comes from the +// builder's arena, and the slot holds its arena offset (§6.3). A pointer whose +// target is a BYTE BUFFER — no table — takes a string instead (§2.5). +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( f->table == NULL ) { return TableJsonReadBlob( in, slot, f ); } + // the pointee nests one level down, exactly as a by-value table does, and + // takes the same cap: a chain nests as deep as it is long (§16.7) + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + char c = TableJsonPeek( in ); + if ( c == '}' ) + { + // an empty object: a node at its defaults, named once + in.pos++; + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + return true; + } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( strcmp( key, "&node" ) != 0 ) + { + // a node named once: the pointee's object in place, and this key is + // its first field — unless it is the reserved prefix under a spelling + // this form does not have, which ReadTableKeys refuses + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return TableJsonReadTableKeys( in, node, f->table, depth + 1, key ); + } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->labels, label, taken ); + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + // ONE SPELLING, and what follows the label says which half it is: fields + // after a label the text has not defined DEFINE it, and a label alone that + // the text has defined REFERS to it. The other two are malformed — a label + // alone that the text never defined, which would otherwise read as a default + // node under a silent report, and a field after a label already defined, + // which would be a second definition. That is what keeps a typo loud. + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + bool bare = c == '}'; + if ( bare == taken ) { in.report->malformed = true; in.bad = true; return false; } + if ( bare ) + { + // A REFERENCE. A label is defined when its object CLOSES, so a + // reference met inside its own definition — at any depth of by-value + // nesting — names a node whose descent is still open: the cycle the + // wire refuses (§3.1), refused here where it is written. A definition + // the reader dropped names no node, so the slot stays null with + // nothing more counted — the drop was counted where it happened. A + // node of another table than the slot declares is a kind mismatch, as + // on the wire. + in.pos++; + if ( entry->open != 0 ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + if ( entry->type == NULL ) + { + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + if ( entry->type != f->table ) + { + memcpy( slot, &ref, sizeof( ref ) ); + in.report->kind_mismatch++; + return true; + } + ref.value = (int64_t) entry->node; + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + // A DEFINITION: the node is allocated, the label is its, and the keys after + // `&node` are its fields. The entry is OPEN until the object closes, so a + // reference to the label from inside the node's own fields is refused as + // the cycle it is; the node and its table are filled in at the close. + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + entry->open = 1; + if ( !TableJsonReadTableKeys( in, node, f->table, depth + 1, NULL ) ) { return false; } + entry = TableJsonGraphMapFind( graph->labels, label ); // the map may have grown under the descent + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + memcpy( &ref, slot, sizeof( ref ) ); + entry->node = (uint32_t) ref.value; + entry->type = f->table; + entry->open = 0; + return true; +} + +// An `&`-prefixed key opening an object the walk is SKIPPING — a value past an +// array's bound, an unknown key's value, a value of the wrong shape. A +// definition in there still takes its label, so the numbering survives whatever +// the storage could not hold (§16.7): the label is registered with no node, and a +// reference to it reads null. Any other prefixed key is the reserved prefix +// out of place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL || strcmp( key, "&node" ) != 0 ) { in.report->malformed = true; in.bad = true; return false; } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + if ( TableJsonGraphMapReach( graph->labels, label, taken ) == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return true; // a fresh entry is node 0, type NULL: a definition with no node +} + +// ---- writing: from a region's const root ---- + +struct TableJsonGraphOut +{ + TableJsonGraphMap nodes; // a node's address -> how many slots name it, and its `&node` once assigned + bool counting; // PASS ONE: count the references, refuse a cycle, emit nothing + int64_t next_label; +}; + +// The node a slot names: null as `null`, a node named once as its object in +// place, and a node named more than once under the construct. Which of the +// last two it is was learned in pass one; pass two spells it. +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphOut * graph = (TableJsonGraphOut *) out.graph; + if ( graph == NULL ) { return false; } + const void * node = f->resolve( slot ); + if ( node == NULL ) + { + out.raw( "null", 4 ); + return true; + } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->nodes, (uint64_t) (uintptr_t) node, taken ); + if ( entry == NULL ) { return false; } + if ( f->table == NULL ) + { + // A BYTE BUFFER (§2.5, §16.7): its text is a string, which has no + // first key to carry `&node`, so a blob named from more than one + // slot has no spelling this form can carry and the graph is refused — + // as a shared node with nothing to write is. A blob named once is its + // bytes in place: base64 for a *bytes, the string itself for a *string. + if ( graph->counting ) { entry->count++; return true; } + if ( entry->count > 1 ) { return false; } + const TableBlob * blob = (const TableBlob *) node; + if ( blob->length > (uint32_t) 0x7fffffff ) { return false; } + if ( strcmp( f->type_name, "string" ) == 0 ) { TableJsonWriteString( out, (const char *) ( blob + 1 ), (int32_t) blob->length ); } + else { TableJsonWriteBase64( out, (const uint8_t *) ( blob + 1 ), (int32_t) blob->length ); } + return true; + } + if ( graph->counting ) + { + // PASS ONE: one visit per node, every slot that names it counted, and + // a reference to a node whose descent is still open is a cycle — + // refused here as the wire refuses it (§3.1) + entry->count++; + if ( !taken ) { return entry->open == 0; } + entry->open = 1; + if ( !TableJsonWriteValue( out, node, f->table, depth ) ) { return false; } + entry = TableJsonGraphMapFind( graph->nodes, (uint64_t) (uintptr_t) node ); // the map may have grown under the descent + if ( entry == NULL ) { return false; } + entry->open = 0; + return true; + } + // PASS TWO: a node named once is its object in place; a node named more + // than once is DEFINED at its first occurrence — `&node` first, then its + // fields — and REFERENCED by `&node` alone after that, spelled the same way at + // every site. Labels run from 1 in first-write order and are the text's own, + // so a stray number in a hand-edited text is most often one never defined. + if ( entry->count <= 1 ) + { + return TableJsonWriteValue( out, node, f->table, depth ); + } + if ( depth > kTableJsonMaxDepth ) { return false; } + if ( entry->label != 0 ) + { + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + out.line( depth ); + out.put( '}' ); + return true; + } + entry->label = ++graph->next_label; + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + bool any = true; + int64_t before = out.offset; + if ( !TableJsonWriteFields( out, node, f->table, depth, any ) ) { return false; } + // a definition carries at least one field, because a label alone is a + // reference: a shared node with nothing to write has no definition this + // form can spell, and the writer refuses it as it refuses any value it + // cannot spell (§16.3) + if ( out.offset == before ) { return false; } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- the two entry points a pointered table's wrappers name ---- + +// The text into the builder's root. Every node the text names is allocated in +// the builder's arena through the field's own Emplace; the label map is the +// walk's, released before this returns. The root itself takes no label — nothing +// may name it (§16.7) — so an `&node` at the root is refused like any other key +// of the prefix. +inline bool TableJsonReadGraph( TableWorker & worker, void * root, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + if ( worker.arena == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } + TableJsonGraphIn graph; + graph.worker = &worker; + TableJsonGraphMapInit( graph.labels, worker.arena->allocator ); + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = &graph; + info->reset( root ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, root, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + TableJsonGraphMapShutdown( graph.labels ); + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +// The text of a region's const root: measured when the buffer is NULL, written +// when it is not, over one code path. Two passes over one walk — the first +// counts how many slots name each node and refuses a cycle, the second writes +// — so a node's first occurrence knows whether it will be named again. The +// ROOT's entry is open for the whole first pass, so a reference back at it is +// the cycle it is (§3.1), and it takes no label. +inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * info, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + if ( root == NULL ) { return -1; } + TableJsonGraphOut graph; + TableJsonGraphMapInit( graph.nodes, allocator ); + graph.counting = true; + graph.next_label = 0; + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph.nodes, (uint64_t) (uintptr_t) root, taken ); + if ( entry == NULL ) { TableJsonGraphMapShutdown( graph.nodes ); return -1; } + entry->open = 1; + TableJsonOut count; + count.buffer = NULL; + count.capacity = 0; + count.offset = 0; + count.overflow = false; + count.graph = &graph; + bool ok = TableJsonWriteValue( count, root, info, 0 ); + graph.counting = false; + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = &graph; + if ( ok ) { ok = TableJsonWriteValue( out, root, info, 0 ); } + TableJsonGraphMapShutdown( graph.nodes ); + if ( !ok ) { return -1; } + out.put( '\n' ); // the canonical text ends with exactly one newline (§16.1) + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json graph walk: end ---- + +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + +// ---- json map walk: begin ---- + +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} + +// the entry's two rows: fields[0] IS the key and fields[1] IS the value, which +// is what makes a user's own table of pairs the same bytes (§2.8) +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } + +inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } +inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } + +// AN INTEGER KEY IS THE INTEGER'S DECIMAL SPELLING, QUOTED, because a JSON +// object's keys are strings. Written digit by digit so no locale can move it. +inline void TableJsonWriteMapIntegerKey( TableJsonOut & out, const void * storage, const TableFieldInfo * key ) +{ + uint64_t magnitude = 0; + bool negative = false; + if ( TableJsonMapKeySigned( key ) ) + { + int64_t value = 0; + switch ( key->kind ) + { + case 2: value = (int64_t) *(const int8_t *) storage; break; + case 3: value = (int64_t) *(const int16_t *) storage; break; + case 4: value = (int64_t) *(const int32_t *) storage; break; + default: value = *(const int64_t *) storage; break; + } + negative = value < 0; + magnitude = negative ? ( ~(uint64_t) value ) + 1 : (uint64_t) value; + } + else + { + switch ( key->kind ) + { + case 6: magnitude = (uint64_t) *(const uint8_t *) storage; break; + case 7: magnitude = (uint64_t) *(const uint16_t *) storage; break; + case 8: magnitude = (uint64_t) *(const uint32_t *) storage; break; + default: magnitude = *(const uint64_t *) storage; break; + } + } + char digits[24]; + int32_t at = (int32_t) sizeof( digits ); + do { digits[--at] = (char) ( '0' + ( magnitude % 10 ) ); magnitude /= 10; } while ( magnitude != 0 ); + if ( negative ) { digits[--at] = '-'; } + TableJsonWriteString( out, digits + at, (int32_t) sizeof( digits ) - at ); +} + +inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const TableFieldInfo * key ) +{ + const uint8_t * storage = (const uint8_t *) entry + key->offset; + if ( TableJsonMapKeyIsString( key ) ) + { + // A STRING KEY IS THE STRING (§2.8): every JSON key of a map object is + // a KEY OF THE MAP and none is a field key, so the `&` prefix §16.7 + // reserves for field keys is ordinary data here. + TableJsonWriteString( out, (const char *) storage, *(const int32_t *) ( (const uint8_t *) entry + key->count_offset ) ); + return; + } + TableJsonWriteMapIntegerKey( out, (const void *) storage, key ); +} + +// ToJson WRITES ENTRIES IN ASCENDING KEY ORDER, so unpack then pack is +// byte-stable and a diff of two texts is a diff of two maps (§2.8, §17.2). +// A region holds them in that order already, so this is the array in place. +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "{}", 2 ); return true; } + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); + out.put( '{' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); + TableJsonWriteMapKey( out, entry, key ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// AN INTEGER KEY IS READ BY §16.2's INTEGER RULE AND BY NOTHING ELSE, so +// "2.0" and "1e3" are the integers 2 and 1000 and "-0" is zero. The token is +// walked as a JSON number over its own bytes; a token that rule calls +// malformed makes the KEY malformed, and a genuinely fractional value, or one +// outside the key kind's range, is kind_mismatch for that entry. +inline bool TableJsonMapKeyValue( const char * token, int32_t length, const TableFieldInfo * key, + int64_t & value, bool & fits ) +{ + fits = false; + TableReport scratch; + TableJsonIn probe = { token, (int64_t) length, 0, &scratch, false, NULL }; + bool integral = false; + if ( !TableJsonWalkNumber( probe, &integral ) ) { return false; } + if ( probe.pos != (int64_t) length ) { return false; } // trailing bytes: not a number + if ( !integral ) + { + const double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) { return true; } // a value no key kind holds + const double whole = d < 0 ? -d : d; + if ( whole != (double) (int64_t) whole ) { return true; } // genuinely fractional + } + bool saturated = false; + const bool is_signed = TableJsonMapKeySigned( key ); + value = integral ? TableJsonTokenInteger( token, length, is_signed, &saturated ) + : (int64_t) TableJsonTokenDouble( token, length, false ); + if ( saturated ) { return true; } // outside every width: kind_mismatch, never clamped + switch ( key->kind ) + { + case 2: fits = value >= -128 && value <= 127; break; + case 3: fits = value >= -32768 && value <= 32767; break; + case 4: fits = value >= -2147483647 - 1 && value <= 2147483647; break; + case 5: fits = true; break; + case 6: fits = value >= 0 && value <= 255; break; + case 7: fits = value >= 0 && value <= 65535; break; + case 8: fits = value >= 0 && (uint64_t) value <= 4294967295ull; break; + default: fits = integral; break; // uint64: the token's own magnitude + } + return true; +} + +// FromJson READS KEYS IN WHATEVER ORDER THE TEXT GIVES THEM. A repeated key is +// last-wins and counted duplicate, the object rule (§16.2) applied inside the +// map. An empty object is an empty map, and null is kind_mismatch. +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const char shape = TableJsonShape( value ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char token[kTableJsonMaxKey]; + int32_t token_length = 0; + if ( !TableJsonScanString( in, token, kTableJsonMaxKey - 1, &token_length ) ) { return false; } + token[token_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t key_value = 0; + bool place = true; + if ( !TableJsonMapKeyIsString( key ) ) + { + bool fits = false; + if ( !TableJsonMapKeyValue( token, token_length, key, key_value, fits ) ) + { + // A MALFORMED KEY STOPS THE READ where §16.1's rule stops it, + // with the instance holding what was placed before the stop. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( !fits ) { in.report->kind_mismatch++; place = false; } + } + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; + if ( place && entry == NULL ) + { + // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the + // wire's rule, because a clamped key is a merged entry (§2.8). + in.report->clamped++; + } + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) + { + in.report->duplicate++; // last-wins, the object rule inside the map + } + const char got = TableJsonValueShape( in ); + if ( entry == NULL ) + { + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( value->kind == 17 && !value->is_array ) + { + // A POINTER VALUE IS SHARED EXACTLY AS A POINTER FIELD IS (§2.8): + // null is a null slot, an object is the pointee in place or an + // &node reference to one (§16.7), anything else is the wrong shape — + // the same three the field-key loop gives a pointer field, because + // an entry's value IS a field line. + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) entry + value->offset, value->elem_size, 0 ); + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) entry + value->offset, value, depth + 1 ) ) + { + return false; + } + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, entry, value, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json map walk: end ---- + +// ---- json list walk: begin ---- + +// an unbounded array is the out-of-line array that is not a map (§8.1) +inline bool TableJsonIsList( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && !TableJsonIsMap( f ); +} + +// ToJson WRITES THE ELEMENTS IN INDEX ORDER, which is the only order there is, +// so unpack then pack is byte-stable without a rule of its own (§2.9, §17.2). +// A region holds the array in place, so this steps it at the descriptor's pitch. +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + const uint8_t * elements = TableJsonExtentElements( slot ); + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const uint8_t * element = elements + (int64_t) i * f->elem_size; + if ( f->kind == 17 ) + { + // a []*T's elements take the pointer row (§16.7): the pointee's + // object in place, null, or `&node` for a shared one + if ( !TableJsonWritePointer( out, element, f, depth + 1 ) ) { return false; } + } + else if ( !TableJsonWriteScalar( out, element, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; +} + +// FromJson READS EVERY ELEMENT THE TEXT CARRIES, appending each through the +// descriptor's place resolver: `[]` is an empty list, and null is +// kind_mismatch, the array row's own rule (§16.2). LAST WINS holds for a +// repeated key: the list goes back to EMPTY before this occurrence's elements +// land, the builder's storage being reclaimed at reset (§2.9). +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) slot, 8, 0 ); + TableJsonSetRaw( (uint8_t *) slot + 8, 4, 0 ); + const char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + void * element = f->place( *graph->worker, slot, NULL, 0, 0 ); + if ( element == NULL ) + { + // NOT ADDED: the arena could not carve another segment, or the + // count met the int32 cap. The text cannot be placed whole, and + // the read stops where §16.1's rule stops it. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( f->kind == 17 ) + { + // an element of a []*T (§2.9): null is a null slot, an object is the + // pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) element, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, element, f, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + // the wrong shape for the element kind: the slot keeps its + // defaults and the event counts, the array row's rule (§16.2) + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, element, f, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json list walk: end ---- + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +bool SampleFromJson( Sample & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, SampleTableType(), text, bytes, report ); +} + +int64_t SampleToJsonMeasure( const Sample & value ) +{ + return TableJsonWrite( &value, SampleTableType(), NULL, 0 ); +} + +int64_t SampleToJson( const Sample & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, SampleTableType(), buffer, capacity ); +} + +bool RowFromJson( RowBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Row * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, RowTableType(), text, bytes, report ); +} + +int64_t RowToJsonMeasure( const Row * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, RowTableType(), NULL, 0, allocator ); +} + +int64_t RowToJson( const Row * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, RowTableType(), buffer, capacity, allocator ); +} + +bool SheetFromJson( SheetBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Sheet * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, SheetTableType(), text, bytes, report ); +} + +int64_t SheetToJsonMeasure( const Sheet * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, SheetTableType(), NULL, 0, allocator ); +} + +int64_t SheetToJson( const Sheet * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, SheetTableType(), buffer, capacity, allocator ); +} + +bool ItemFromJson( Item & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, ItemTableType(), text, bytes, report ); +} + +int64_t ItemToJsonMeasure( const Item & value ) +{ + return TableJsonWrite( &value, ItemTableType(), NULL, 0 ); +} + +int64_t ItemToJson( const Item & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, ItemTableType(), buffer, capacity ); +} + +bool SquadFromJson( SquadBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Squad * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, SquadTableType(), text, bytes, report ); +} + +int64_t SquadToJsonMeasure( const Squad * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, SquadTableType(), NULL, 0, allocator ); +} + +int64_t SquadToJson( const Squad * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, SquadTableType(), buffer, capacity, allocator ); +} + +bool ArmyFromJson( ArmyBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Army * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, ArmyTableType(), text, bytes, report ); +} + +int64_t ArmyToJsonMeasure( const Army * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, ArmyTableType(), NULL, 0, allocator ); +} + +int64_t ArmyToJson( const Army * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, ArmyTableType(), buffer, capacity, allocator ); +} + +bool DeckFromJson( DeckBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Deck * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, DeckTableType(), text, bytes, report ); +} + +int64_t DeckToJsonMeasure( const Deck * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, DeckTableType(), NULL, 0, allocator ); +} + +int64_t DeckToJson( const Deck * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, DeckTableType(), buffer, capacity, allocator ); +} + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/HoldersTable.h b/testdata/golden/tables/lists/HoldersTable.h new file mode 100644 index 000000000..6ae1b0d5b --- /dev/null +++ b/testdata/golden/tables/lists/HoldersTable.h @@ -0,0 +1,11518 @@ +// Code generated by the schema compiler from Holders.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — protocol id 0xa5fbe602c119cdd9 (packets only: tables version by field id, not by protocol id) +// The TABLE wire (evolution-tolerant, docs/SPEC-TABLES.md): no serialize +// dependency — includable from any TU. + +#pragma once + +#include +#include // the prefill's scalar-array fills +#include // offsetof, for the reflection descriptors + +// ---- the hooks (docs/USAGE.md, "the C++ table runtime's hooks") ---- +// +// schema_assert — the runtime's own assert, and the refusal a debugger reads. +// NDEBUG removes it, exactly as it removes assert. A caller who already routes +// serialize's asserts writes `#define schema_assert serialize_assert` before +// including this header and both halves land in one handler. +#ifndef schema_assert +#include +#define schema_assert assert +#endif // #ifndef schema_assert + +// schema_fatal — what stands after the assert on a path that cannot continue. +// NDEBUG does not remove it. Supply it and is never included. +#ifndef schema_fatal +#include // abort +#define schema_fatal abort +#endif // #ifndef schema_fatal + +// schema_allocate / schema_release — what "no allocator handed in" means for +// this program. schema_allocate hands back ZEROED bytes and NULL on failure: +// an arena segment is copied whole, padding included, so anything left +// uninitialized here would reach a packed region. Supply both and +// is never included; hand a TableAllocator to a builder to route one +// structure's allocations somewhere else again. +#ifndef schema_allocate +#include // calloc, free +#define schema_allocate( bytes ) calloc( (size_t) 1, (size_t) ( bytes ) ) +#define schema_release( pointer ) free( pointer ) +#endif // #ifndef schema_allocate +#include // a node's lifetime starts in arena storage (placement new) +#include // one atomic per slab: the arena is lock-free by ownership + +#include "Holders.h" + +#ifndef LISTDEMO_SCHEMA_TABLE_PRIMITIVES +#define LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +// THE CODEC DOES NOT DEPEND ON THE COMPILER'S INLINING BUDGET. A table of a +// realistic field count emits one large body per type, and the cursor a body +// writes through lives in the caller's `TableWriter`: across a call boundary +// that cursor round-trips through memory, and a `uint8_t *` store may alias the +// writer itself, so every put reloads it. When a budget runs out mid-body the +// codec silently degrades to that shape. Forcing the primitives and the +// fixed-class bodies inline is what keeps the cursor in registers and lets +// adjacent constant framing bytes merge into one store. +#if defined( _MSC_VER ) +#define LISTDEMO_TABLE_INLINE __forceinline +#elif defined( __GNUC__ ) || defined( __clang__ ) +#define LISTDEMO_TABLE_INLINE inline __attribute__(( always_inline )) +#else +#define LISTDEMO_TABLE_INLINE inline +#endif + +namespace listdemo { + +// WHY A READ WAS REFUSED, by name (docs/SPEC-TABLES.md §3.3, §11). A REFUSAL +// is not one of §4's events: nothing is decoded, no counter moves and no +// damage is reported, so five zero counters and a false flag are what a clean +// read prints too and only the verdict tells them apart. The reason says which +// refusal it was. +// +// This is the MESSAGE PATH's vocabulary and not the cooked form's (§7.4): a +// caller meeting one of these has been refused a MESSAGE on a connection, +// which is a different recovery with a different owner than a file a header +// match turned down. +enum TableMessageReason +{ + newer_form, // a FORM BYTE this reader does not carry (§3) + no_vocabulary, // no table for this connection: the message arrived before the announcement, or after a refused one + second_announcement, // a second announcement on a connection: it sets nothing, amends nothing, and the connection closes + vocabulary_too_large, // an announcement above the receiver's declared bound, refused before an entry is touched + message_form_as_file // a form 2 wire where a FILE was expected: its table is somewhere else +}; + +// The table-wire read report — the permissive contract's ledger. Silence +// (all zero) means the data matched this reader's schema exactly. +struct TableReport +{ + int32_t unknown = 0; // unknown field ids skipped (newer data) + int32_t kind_mismatch = 0; // known id, changed type — skipped, never misdecoded + int32_t clamped = 0; // out-of-range values clamped to declared bounds + // a key the TEXT form saw twice: last wins, and the repeat is counted + // (docs/SPEC-TABLES.md §16.2). The wire never raises it — a body carrying an + // id twice is legal input whose last occurrence wins, silently (§3). + int32_t duplicate = 0; + bool malformed = false; // framing damage; decode stopped, partial result kept + // THE REFUSAL VERDICT, which is not one of §4's events and moves no counter + // (docs/SPEC-TABLES.md §3): a FORM BYTE this reader does not carry. Five + // zero counters and a false flag are what a clean read prints too, so the + // verdict is what tells the two apart. + bool refused = false; + // WHICH refusal, and it is read only when refused is set: a read that + // was not refused has no reason, and this member is the one the caller + // must not look at then (docs/SPEC-TABLES.md §3.3). + TableMessageReason reason = newer_form; +}; + + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; +// ---- reflection (tables only, docs/SPEC-TABLES.md) ---- +// +// Static field descriptors for every type in the table closure: name, wire +// id/kind, storage offset, bounds, ranges, enum names and branch guards — +// enough to walk, print, diff, edit or bind any table value at runtime with +// no RTTI and no schema files. TableType() returns X's descriptor. + +struct TableTypeInfo; + +// One arm of a union field: where its payload sits inside the union's storage +// and what its payload looks like. The arm's NAME and its table-wire id come +// from the field's enum_name/variant_id functions at the same tag, so nothing +// is spelled twice (docs/SPEC-TABLES.md §8). +struct TableFieldInfo; + +struct TableUnionArmInfo +{ + uint32_t offset; // offsetof the arm's payload within the union storage + const TableTypeInfo * table; // the arm payload's descriptor, or NULL + // AN ARM IS A FIELD LINE (docs/SPEC-TABLES.md §2.6): an arm that names no + // declared type or table carries the FIELD descriptor a field of that + // type would carry instead — offsets taken within the union storage — so + // a generic walk meets an arm's kind, width, bounds and companions where + // it meets a field's. Exactly one of the two is non-NULL on a set arm. + const TableFieldInfo * field; + uint32_t size; // the arm's whole storage, which selection zero-establishes +}; + +// A union field's shape: the tag, and the arms indexed by it. Arms run +// [0, enum_max]; index 0 is the EMPTY arm and carries no payload. +struct TableUnionInfo +{ + uint32_t tag_offset; // offsetof the tag within the union storage + uint32_t tag_size; // sizeof the tag + const TableUnionArmInfo * arms; +}; + +// The exact raw range of a wide-kind field (docs/SPEC-TABLES.md §8.2): two 128-bit +// values as 64-bit lanes, low lane first, two's complement for the signed kinds. +struct TableWideRange +{ + uint64_t lo[2]; + uint64_t hi[2]; +}; + +// the arena's allocation front, defined with the variable-length runtime +// below; a descriptor names it only through a pointer parameter. +struct TableWorker; + +struct TableFieldInfo +{ + const char * name; // schema field name, e.g. "health" + const char * json; // the TEXT form's key: the json = "key" attribute, else name (§16.3) + const char * type_name; // schema type name, e.g. "float32", "Grade" + uint64_t id; // table-wire field id: fnv1a64 of the name, of the was alias after a rename (§5) + uint8_t kind; // table-wire kind; for arrays/strings/bytes, the ELEMENT kind + bool is_array; // fixed or counted array (bytes included) + bool is_pointer; // a *T pointer field: storage is an 8-byte TableRef; the target is a table + // THE TWO THE TEXT FORM NEEDS (docs/SPEC-TABLES.md §16.7), and they + // are here for the same reason is_pointer is: the walk is ONE walk + // over descriptors and cannot spell a target's own At or + // Emplace. `resolve` reads a slot in a REGION and answers the + // node it names, or NULL; `emplace` allocates one in a BUILDER's + // arena and points the slot at it. NULL on every field that is not + // a pointer, and emitted only in a unit that declares one. + const void * (*resolve)( const void * slot ); + void * (*emplace)( TableWorker & worker, void * slot ); + bool counted; // a _count/_length int32 companion exists (counted arrays, strings, bytes) + bool optional; // a ?T field: a _present bool companion decides whether it rides + int32_t array_bound; // array capacity / string max length; 0 for plain scalars + uint32_t offset; // offsetof the storage member + uint32_t elem_size; // sizeof the member (element size for arrays) + uint32_t count_offset; // offsetof the _count/_length companion, or 0xffffffff + uint32_t present_offset; // offsetof the _present companion, or 0xffffffff + const TableTypeInfo * table; // nested table's descriptor, or NULL + bool has_range; // a declared [min, max] (int or float) + double range_min; // NOTE: int64 ranges beyond 2^53 lose precision here + double range_max; + // the WIDE kinds (18-29, docs/SPEC-TABLES.md §3, §8.2): frac_bits is a fixed + // field's F — its storage holds units × 2^F — and wide is the declared + // range on that RAW scale, exact, as two 128-bit two's-complement values + // in 64-bit lanes (low lane first). NULL where the declaration bounds + // nothing (a bare uint128) and for every other kind; frac_bits is 0 for + // every kind that is not fixed-point. range_min/range_max still carry + // the declared bounds as doubles — whole units for a fixed field — for + // a walker that only shows them. + uint8_t frac_bits; + const TableWideRange * wide; + int64_t enum_max; // enums: highest valid value (None = 0 always valid); + // unions: the arm count (tag range [0, enum_max]); + // flags: the highest declared BIT INDEX; else -1 + // the vocabulary's names, indexed the same way enum_max bounds: an enum's + // value -> name, a union's tag -> arm name, a FLAGS field's bit index -> + // variant name. NULL for every other kind. + const char * (*enum_name)( uint64_t value ); + // the TABLE-WIRE id of one variant (docs/SPEC-TABLES.md §5): for an enum, the + // hash of the variant's name; for a union, the hash of the arm's name. + // 0 is the reserved id — an enum's None, a union's empty. NULL for every + // other kind — a FLAGS field's variants have no per-variant wire id (§4), + // so a NULL here beside a non-NULL enum_name is what says "flags". + // Walk [0, enum_max] to enumerate a vocabulary and its ids. + uint64_t (*variant_id)( uint64_t value ); + // an ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): the array has one slot per + // variant of key_type_name, indexed by the variant's value, and its slots + // ride under variant ids rather than positions. key_name and key_id are + // the key's vocabulary — walk [0, array_bound) to print slots by name. + // NULL on every other field. + const char * key_type_name; + const char * (*key_name)( uint64_t value ); + uint64_t (*key_id)( uint64_t value ); + // union fields: the tag and its arms, behind a function so the whole + // descriptor stays CONSTANT-INITIALISED (a captureless lambda converts to + // a function pointer at compile time; the arms themselves are a static + // inside it). NULL for every other kind. + const TableUnionInfo * (*arms)(); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded +}; + +struct TableTypeInfo +{ + const char * name; // schema type name + uint32_t size; // sizeof the storage struct + int32_t num_fields; + const TableFieldInfo * fields; + // put one instance back at its declared defaults, in place. A generic + // walker that fills a value has to be able to establish the defaults an + // absent field takes, and it holds no type to spell — this is the one + // thing the descriptors could not express without it. Placement-new + // value-init, exactly what the wire's read path does, and no temporary. + void (*reset)( void * storage ); + // the DERIVED mode (docs/SPEC-TABLES.md): false = fixed-size, a plain + // relocatable struct; true = variable-length, built through a Builder + // and read through a region root. Nobody declares it; the compiler + // works it out. + bool variable; +}; + +struct TableWriter +{ + uint8_t * buffer; + int64_t capacity; + int64_t offset = 0; + bool overflow = false; + + // the parameters do not repeat the member names: a parameter that hides a + // member is a warning the estate's compilers disagree about (gcc's + // -Wshadow and cl's C4458 refuse it, clang's -Wshadow does not), and this + // is a header a consumer compiles under its OWN flags + TableWriter( uint8_t * to_buffer, int64_t to_capacity ) : buffer( to_buffer ), capacity( to_capacity ) {} + + LISTDEMO_TABLE_INLINE void raw( const void * data, int64_t bytes ) + { + if ( offset + bytes > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) bytes ); + offset += bytes; + } + LISTDEMO_TABLE_INLINE void put8( uint8_t v ) { raw( &v, 1 ); } + LISTDEMO_TABLE_INLINE void put16( uint16_t v ) { uint8_t b[2] = { uint8_t( v ), uint8_t( v >> 8 ) }; raw( b, 2 ); } + LISTDEMO_TABLE_INLINE void put32( uint32_t v ) { uint8_t b[4] = { uint8_t( v ), uint8_t( v >> 8 ), uint8_t( v >> 16 ), uint8_t( v >> 24 ) }; raw( b, 4 ); } + LISTDEMO_TABLE_INLINE void put64( uint64_t v ) { put32( uint32_t( v ) ); put32( uint32_t( v >> 32 ) ); } + // a 128-bit value as two lanes, the low half first (docs/SPEC-TABLES.md §3) + LISTDEMO_TABLE_INLINE void put128( uint64_t lo, uint64_t hi ) { put64( lo ); put64( hi ); } + // EVERY LENGTH, COUNT, INDEX AND ID REFERENCE IS ONE CANONICAL UNSIGNED + // LEB128 (docs/SPEC-TABLES.md §3): seven value bits a byte, the lowest + // group first, the high bit set on every byte but the last. One value has + // one spelling, so two conforming writers agree byte for byte. + LISTDEMO_TABLE_INLINE void putleb( uint64_t v ) + { + while ( v >= 0x80 ) { put8( uint8_t( v ) | 0x80 ); v >>= 7; } + put8( uint8_t( v ) ); + } +}; + +// TableLebBytes is one value's spelling length, which a MEASURE needs before +// the bytes exist — the length of a body has to be known before it is written, +// because a length whose own width moves cannot be patched in place. +inline int64_t TableLebBytes( uint64_t v ) +{ + int64_t n = 1; + while ( v >= 0x80 ) { v >>= 7; n++; } + return n; +} + +// THE ID TABLE, WRITER SIDE (docs/SPEC-TABLES.md §3). It holds every id the +// body used, once each, in FIRST-USE order over the whole wire, and the body +// names them by position: reference k is the kth entry, counted from 1, and +// reference 0 names NO ID. +// +// Its capacity is a COMPILE-TIME fact of the unit — the distinct names its +// table closure can spell — so a save allocates nothing: the table is a local +// of Measure and of Save. The bucket chain makes ref constant time and makes +// truncate constant time too, which is what an ELIDED field needs: a field +// that turns out not to ride costs nothing in the id table either, so the walk +// interns its id, builds the payload that decides, and undoes the entry when +// nothing rides. +struct TableIds +{ + static const int32_t kCapacity = 60; + static const int32_t kBuckets = 128; + + uint64_t ids[ kCapacity ]; + int32_t chain[ kCapacity ]; + int32_t head[ kBuckets ]; + int32_t count; + bool overflow; + // THE MESSAGE FORM'S SLOTS (docs/SPEC-TABLES.md §3.3). A form 2 wire + // names ids through the CONNECTION's table, which is the unit's whole + // vocabulary in a compiler-settled order — so every reference is known at + // compile time and rides at the header as a literal beside the id. This + // flag is what selects it: false interns the id in first-use order and + // writes a trailer, true answers the slot and writes none, and the walk + // that decides is one walk. + bool vocabulary; + + TableIds() : count( 0 ), overflow( false ), vocabulary( false ) + { + for ( int32_t i = 0; i < kBuckets; i++ ) { head[i] = -1; } + } + + static LISTDEMO_TABLE_INLINE uint32_t bucket_of( uint64_t id ) + { + return uint32_t( ( id * 0x9E3779B97F4A7C15ull ) >> 57 ) & uint32_t( kBuckets - 1 ); + } + + // the reference an id takes: its message-form SLOT under the connection's + // table, or the file's own first-use entry + LISTDEMO_TABLE_INLINE uint64_t ref( uint64_t id, uint64_t slot ) + { + if ( vocabulary ) { return slot; } + return intern( id ); + } + + // the FILE form's half, appending the id on first use + uint64_t intern( uint64_t id ) + { + const uint32_t b = bucket_of( id ); + for ( int32_t i = head[b]; i >= 0; i = chain[i] ) + { + if ( ids[i] == id ) { return uint64_t( i ) + 1; } + } + if ( count >= kCapacity ) { overflow = true; return 1; } + ids[count] = id; chain[count] = head[b]; head[b] = count; count++; + return uint64_t( count ); + } + + // undo every entry appended since mark. An entry removed is the most + // recent one in its bucket, so it sits at that bucket's head. + void truncate( int32_t mark ) + { + // a SLOT costs no entry, so an elided field has nothing to undo + if ( vocabulary ) { return; } + while ( count > mark ) + { + count--; + head[ bucket_of( ids[count] ) ] = chain[count]; + } + } +}; + +// TableIdsBytes is the trailer's own size: the entries, each a fixed +// little-endian u64, and the ENTRY COUNT, the one fixed-width number on the +// wire (docs/SPEC-TABLES.md §3). +inline int64_t TableIdsBytes( const TableIds & ids ) { return int64_t( ids.count ) * 8 + 8; } + +// TableIdsWrite puts the trailer where the walk ended: a writer never patches, +// because first-use order is known only when the walk ends. +inline void TableIdsWrite( TableWriter & w, const TableIds & ids ) +{ + for ( int32_t i = 0; i < ids.count; i++ ) { w.put64( ids.ids[i] ); } + w.put64( uint64_t( ids.count ) ); +} + +// THE ID TABLE, READER SIDE (docs/SPEC-TABLES.md §3). A reader locates it from +// the END of the wire and resolves it ONCE, at open: the entries are eight +// bytes each and a body names them by position, so every field dispatches +// through an index rather than through a search over hashes. +struct TableIdTable +{ + const uint8_t * entries = NULL; + int64_t count = 0; + + // the id a reference names. ref is 1-based and bounds-checked by the + // caller: a reference ABOVE the entry count is framing damage on the body + // that carries it, and 0 names no id at all. + uint64_t at( uint64_t ref ) const + { + const uint8_t * e = entries + ( ref - 1 ) * 8; + uint64_t lo = uint64_t( e[0] ) | uint64_t( e[1] ) << 8 | uint64_t( e[2] ) << 16 | uint64_t( e[3] ) << 24; + uint64_t hi = uint64_t( e[4] ) | uint64_t( e[5] ) << 8 | uint64_t( e[6] ) << 16 | uint64_t( e[7] ) << 24; + return lo | ( hi << 32 ); + } +}; + +struct TableReader +{ + const uint8_t * buffer; + int64_t size; + int64_t offset = 0; + TableReport * report; + const TableIdTable * ids = NULL; + // ONLY THE ROOT BODY CARRIES THE NODE TABLE (docs/SPEC-TABLES.md §3.1), so + // a body has to know which it is: the reserved id inside a NESTED body is + // malformed, because a second numbering cannot exist. Every reader made + // for a payload is nested; the two the wire surfaces make for a root say so. + bool nested = true; + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report ) + : buffer( from_buffer ), size( from_size ), report( to_report ) {} + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report, const TableIdTable * to_ids ) + : buffer( from_buffer ), size( from_size ), report( to_report ), ids( to_ids ) {} + + LISTDEMO_TABLE_INLINE bool has( int64_t bytes ) const { return offset + bytes <= size; } + // A LENGTH IS A 64-BIT NUMBER AND A BUFFER IS NOT (docs/SPEC-TABLES.md + // §3): every length, count and index on this wire has sixty-four bits of + // capability, so one past what remains must be compared UNSIGNED. Casting + // it to int64 first turns 0xFFFFFFFFFFFFFFFF into -1, and a negative + // length looks like room. + LISTDEMO_TABLE_INLINE bool room( uint64_t bytes ) const { return bytes <= (uint64_t) ( size - offset ); } + LISTDEMO_TABLE_INLINE uint8_t get8() { return buffer[offset++]; } + LISTDEMO_TABLE_INLINE uint16_t get16() { uint16_t v = uint16_t( buffer[offset] ) | uint16_t( buffer[offset+1] ) << 8; offset += 2; return v; } + LISTDEMO_TABLE_INLINE uint32_t get32() { uint32_t v = uint32_t( buffer[offset] ) | uint32_t( buffer[offset+1] ) << 8 | uint32_t( buffer[offset+2] ) << 16 | uint32_t( buffer[offset+3] ) << 24; offset += 4; return v; } + LISTDEMO_TABLE_INLINE uint64_t get64() { uint64_t lo = get32(); uint64_t hi = get32(); return lo | ( hi << 32 ); } + LISTDEMO_TABLE_INLINE void get128( uint64_t & lo, uint64_t & hi ) { lo = get64(); hi = get64(); } + + // ONE CANONICAL UNSIGNED LEB128 (docs/SPEC-TABLES.md §3), and a + // non-minimal spelling is MALFORMED: 0x80 0x00 and 0x00 both spell zero, + // and only the second is legal input. An encoding past ten bytes, or a + // tenth byte with a bit above the 64th value bit, is malformed on the same + // rule. false = framing damage on the body carrying it. + bool getleb( uint64_t & value ) + { + // A NUMBER THIS READER REFUSES LEAVES THE CURSOR WHERE IT WAS. The + // caller's next question is often "did this body end exactly at its + // L", and a rejected number that had moved the cursor would answer + // that question with the damage already stepped over. + const int64_t at = offset; + value = 0; + uint32_t shift = 0; + for ( int32_t i = 0; i < 10; i++ ) + { + if ( !has( 1 ) ) { offset = at; return false; } + const uint8_t b = get8(); + if ( i == 9 && b > 1 ) { offset = at; return false; } + value |= uint64_t( b & 0x7F ) << shift; + if ( ( b & 0x80 ) == 0 ) + { + if ( i > 0 && b == 0 ) { offset = at; return false; } // a redundant continuation + return true; + } + shift += 7; + } + offset = at; + return false; + } + + // resolve one id reference against the file's table. false = a reference + // ABOVE the entry count, or a 0 where an id is required, both of which + // are framing damage on the body that carries it. + bool getid( uint64_t & id ) + { + uint64_t ref = 0; + if ( !getleb( ref ) ) { return false; } + if ( ref == 0 || ids == NULL || ref > (uint64_t) ids->count ) { return false; } + id = ids->at( ref ); + return true; + } + + // skip one payload by kind; false = framing damage. FOUR RULES COVER THE + // SET (docs/SPEC-TABLES.md §3), and a kind outside it is not skippable — + // which is why the set is closed and why kind 31 exists. + bool skip( uint8_t kind ) + { + switch ( kind ) + { + // the fixed-width kinds, each by its width: 18-29 are the 128-bit integers and + // the fixed-point family at every storage width (docs/SPEC-TABLES.md §3) + case 1: case 2: case 6: case 20: case 25: return has( 1 ) ? ( offset += 1, true ) : false; + case 3: case 7: case 21: case 26: return has( 2 ) ? ( offset += 2, true ) : false; + case 4: case 8: case 10: case 22: case 27: return has( 4 ) ? ( offset += 4, true ) : false; + case 5: case 9: case 11: case 23: case 28: return has( 8 ) ? ( offset += 8, true ) : false; + case 18: case 19: case 24: case 29: return has( 16 ) ? ( offset += 16, true ) : false; + case 17: case 30: // a NODE INDEX (§3.1) and an ENUM's variant reference: one LEB128 and stop + { + uint64_t ignored = 0; + return getleb( ignored ); + } + case 12: case 13: case 14: case 16: case 31: case 32: // 31 is the ESCAPE, 32 the payload-free kind + { + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + case 15: // union: the arm id reference, then its kind, its L and its payload (reference 0 = empty) + { + uint64_t arm = 0; + if ( !getleb( arm ) ) return false; + if ( arm == 0 ) return true; + if ( !has( 1 ) ) return false; + offset += 1; // the arm's kind byte + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + } + return false; + } +}; + +// The RESERVED node-table id, the one id the language holds back +// (docs/SPEC-TABLES.md §3.1, §5). It rides in every unit, pointered or not, +// because every body has to know that a NESTED body claiming one is damaged. +static const uint64_t kTableNodeTableFieldId = 0xFFFFFFFFFFFFFFFFull; + +// TableWireForm is the FORM BYTE, and it is the whole header +// (docs/SPEC-TABLES.md §3). A reader that meets a byte it does not know +// refuses the wire by name and never reports damage. +const uint8_t kTableWireForm = 1; + +// TableOpen reads the form byte and the trailer, in that order, and hands back +// the ROOT BODY. It answers one of three verdicts, because five zero counters +// and a false flag are what a clean read prints too: +// +// TableOpenOk the form is known and the table read whole +// TableOpenRefused a FORM BYTE this reader does not carry: nothing is +// decoded, nothing is counted, and no damage is reported +// TableOpenDamaged a table that cannot be read whole — fewer than eight +// bytes, a count whose entries run past the front of the +// file, a count that leaves no room for the form byte, or +// ONE ID IN TWO ENTRIES. The whole wire is malformed, +// nothing is decoded, and one event is counted. +// TableOpenBodyStopped the form and the table were good and the ROOT BODY +// could not be walked to its own terminator. What it +// decoded before that is kept, as everywhere on this wire. +enum TableOpenVerdict { TableOpenOk, TableOpenRefused, TableOpenDamaged, TableOpenBodyStopped }; + +inline TableOpenVerdict TableOpen( const uint8_t * buffer, int64_t bytes, TableIdTable & table, int64_t & body_bytes ) +{ + if ( bytes < 1 ) { return TableOpenDamaged; } + if ( buffer[0] != kTableWireForm ) { return TableOpenRefused; } + if ( bytes < 9 ) { return TableOpenDamaged; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + uint64_t count = lo | ( hi << 32 ); + if ( count > (uint64_t) ( bytes / 8 ) ) { return TableOpenDamaged; } + const int64_t span = (int64_t) count * 8 + 8; + if ( span + 1 > bytes ) { return TableOpenDamaged; } + table.entries = buffer + bytes - span; + table.count = (int64_t) count; + // THE ENTRIES ARE DISTINCT: a table that carries one id twice is malformed + // for the whole wire, because no wire this schema writes carries a repeat + // and it would leave one more shape of table for a hostile writer to aim + // at (docs/SPEC-TABLES.md §3). + for ( int64_t i = 1; i < table.count; i++ ) + { + const uint64_t id = table.at( uint64_t( i ) + 1 ); + for ( int64_t j = 0; j < i; j++ ) + { + if ( table.at( uint64_t( j ) + 1 ) == id ) { return TableOpenDamaged; } + } + } + body_bytes = bytes - span - 1; + return TableOpenOk; +} + +// TableBodyExtent walks a body's framing to the zero reference that ends it, +// so a reader can tell a body that ENDED EARLY — leaving bytes no field claims +// — from one that is merely damaged. ANY BYTE BETWEEN THE ROOT'S TERMINATOR +// AND THE TABLE'S FIRST ENTRY IS MALFORMED, because no field claims it and the +// two ends of the file have met (docs/SPEC-TABLES.md §3). +inline bool TableBodyEndsEarly( const uint8_t * body, int64_t bytes, const TableIdTable & table ) +{ + TableReport ignored; + TableReader r( body, bytes, &ignored, &table ); + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { return false; } + if ( ref == 0 ) { return r.offset != bytes; } + if ( ref > (uint64_t) table.count ) { return false; } + if ( !r.has( 1 ) ) { return false; } + if ( !r.skip( r.get8() ) ) { return false; } + } +} + +// THE MESSAGE FORM (docs/SPEC-TABLES.md §3.3): a FILE carries its own id +// table and a MESSAGE STREAM announces one and then carries none. +// +// A form 2 wire is TWO PARTS, the form byte and the root body: the body ends +// at its own zero reference as it does in a file, there is no trailer, and the +// message's last byte is the body's terminator. Its references resolve against +// the CONNECTION's table, which is the unit's whole vocabulary in the order +// the compiler settled. +const uint8_t kTableWireMessageForm = 2; + +// The RESERVED build-version id, the second id the language holds back (§5, +// §11), beside the node table's. It is the announcement's one required field, +// and a reserved id in any body but the one whose transport it is, is +// malformed (§3.1). +static const uint64_t kTableBuildVersionFieldId = 0xFFFFFFFFFFFFFFFEull; + +// The reserved NODE-TABLE id's own slot in this unit's vocabulary (§3.3). A +// pointered message names the node table through it, exactly as every other +// field header names its id through a slot. +static const uint64_t kTableNodeTableFieldSlot = 38; + +// THE UNIT'S ANNOUNCEMENT, byte for byte: 61 entries and 508 bytes. It is an +// ordinary form 1 FILE — the form byte, a body carrying the BUILD VERSION +// under the reserved id at kind 9, and the trailer that IS the connection's +// table, slot 1 the reserved id and slots 2 and up the vocabulary under one +// numbering. +// +// The vocabulary is the unit's whole closure in the COOK PROJECTION's order +// (§20.2) — each record in the order the projection renders it and each +// record's fields in the order the projection renders them, then each enum's +// variants and each union's arms — followed by the tail the projection does +// not name: the reserved node-table id, the three blob type ids as bytes, +// string and wstring, and every table's own name id in the projection's sorted +// record order. The tail is UNCONDITIONAL, so an ordinary edit only ever grows +// it at its end and never moves a slot a generated field header carries as a +// literal. +static const int64_t kTableAnnounceBytes = 508; +static const uint8_t kTableAnnounce[ kTableAnnounceBytes ] = { + 0x01, 0x01, 0x09, 0xc7, 0x71, 0x45, 0xca, 0xda, 0x0e, 0x7c, 0x8d, 0x00, + 0xfe, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x30, 0xb1, 0x3a, 0xff, + 0x4a, 0xd9, 0xb1, 0x40, 0x20, 0xea, 0x4d, 0x40, 0x8e, 0xa7, 0x19, 0xaa, + 0x26, 0xa9, 0x02, 0x0c, 0x9b, 0x01, 0x48, 0x78, 0xe9, 0xea, 0x71, 0x6f, + 0x0f, 0x01, 0x82, 0xbf, 0x6f, 0x2c, 0x41, 0x4f, 0xbf, 0x84, 0x78, 0x3e, + 0xf3, 0xa4, 0x48, 0x44, 0x19, 0xab, 0xd7, 0x56, 0x05, 0x4a, 0xa3, 0x30, + 0x67, 0x55, 0x5b, 0x85, 0xc9, 0xe2, 0x4e, 0x30, 0x69, 0x6a, 0xb4, 0x81, + 0xfb, 0x67, 0x4d, 0x1a, 0xcf, 0x7b, 0x27, 0x21, 0x74, 0xa2, 0x79, 0x44, + 0x8e, 0xe2, 0xe5, 0xb1, 0x84, 0x76, 0xbc, 0x2e, 0xef, 0x83, 0x76, 0x1e, + 0xc5, 0x99, 0xf7, 0x82, 0x76, 0x4e, 0x0a, 0xd9, 0xa8, 0x2e, 0x86, 0x70, + 0x84, 0xed, 0xf2, 0x4a, 0xbb, 0xf0, 0x0c, 0x9b, 0xcc, 0xfb, 0x2d, 0x73, + 0x68, 0xb7, 0xf0, 0xae, 0x4c, 0x0c, 0xf6, 0x52, 0xbf, 0xe9, 0xd1, 0x2f, + 0x93, 0xcd, 0xda, 0xdb, 0x22, 0x72, 0x34, 0x7d, 0xf6, 0x0b, 0x72, 0x17, + 0x07, 0x17, 0x02, 0x86, 0x4c, 0xf5, 0x63, 0xaf, 0x54, 0x15, 0x02, 0x86, + 0x4c, 0xf4, 0x63, 0xaf, 0x3a, 0x70, 0x6e, 0x3e, 0x93, 0x43, 0xe5, 0x9d, + 0x3d, 0x62, 0xcb, 0x8f, 0xec, 0xfc, 0xf7, 0x39, 0x09, 0x06, 0x02, 0x86, + 0x4c, 0xeb, 0x63, 0xaf, 0x09, 0x4b, 0x4d, 0x57, 0xaa, 0x33, 0x47, 0xd2, + 0x31, 0x54, 0xaf, 0x1d, 0x19, 0x73, 0x50, 0x12, 0xb2, 0x0f, 0x40, 0x27, + 0x0b, 0x6b, 0x98, 0x01, 0x38, 0x81, 0x0a, 0xf1, 0x1f, 0x06, 0xa7, 0xa3, + 0x0f, 0x62, 0xad, 0x07, 0x77, 0x47, 0x82, 0x5f, 0x42, 0x4f, 0x4f, 0x30, + 0x0d, 0x39, 0x84, 0x1c, 0x86, 0x1b, 0x63, 0x8e, 0xba, 0xad, 0xbc, 0xc4, + 0xec, 0x10, 0x5b, 0x36, 0x19, 0x4a, 0xc9, 0x3d, 0xea, 0x0c, 0xe8, 0x30, + 0x94, 0xfd, 0xe4, 0x7c, 0xec, 0x22, 0x02, 0x86, 0x4c, 0xfc, 0x63, 0xaf, + 0x05, 0x28, 0x02, 0x86, 0x4c, 0xff, 0x63, 0xaf, 0x52, 0x26, 0x02, 0x86, + 0x4c, 0xfe, 0x63, 0xaf, 0xb1, 0x45, 0xc3, 0x44, 0x35, 0xab, 0xfe, 0x73, + 0xc0, 0x7f, 0xb3, 0x8a, 0xbe, 0x08, 0x63, 0x7f, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xe4, 0x4f, 0x1c, 0x4f, 0x47, 0xc0, 0x2e, 0x2f, + 0x58, 0xfc, 0xaf, 0xfa, 0xd8, 0xe0, 0x4b, 0x70, 0xc7, 0xd4, 0x7b, 0x26, + 0xb0, 0x9d, 0x29, 0x5f, 0xcc, 0x14, 0x15, 0x7f, 0xcb, 0xc2, 0x58, 0xd8, + 0x84, 0x9e, 0x3a, 0x55, 0x8b, 0x37, 0xe2, 0x06, 0x2a, 0x89, 0xf5, 0x12, + 0x09, 0xc1, 0xca, 0x0a, 0x44, 0xa2, 0x31, 0xc1, 0xad, 0xa7, 0xee, 0xee, + 0xe8, 0xcf, 0xbf, 0x43, 0x73, 0x18, 0x43, 0xd0, 0x42, 0xad, 0xf6, 0xf8, + 0x59, 0x86, 0x63, 0x91, 0xb7, 0xce, 0x00, 0x7c, 0xd1, 0xc5, 0x34, 0x20, + 0x06, 0x68, 0x47, 0x98, 0xd1, 0xa1, 0xcf, 0x52, 0x5f, 0x82, 0x58, 0xac, + 0x36, 0x15, 0x78, 0x5e, 0xb8, 0x8b, 0x59, 0x6f, 0xc9, 0xc6, 0x86, 0xbb, + 0xc3, 0x64, 0x89, 0x50, 0xd2, 0x8d, 0xa7, 0xf1, 0x80, 0xea, 0x3a, 0xb9, + 0xf1, 0x21, 0xf7, 0x41, 0x11, 0xed, 0xd9, 0xce, 0x96, 0x92, 0x43, 0x8a, + 0xfb, 0x06, 0xc9, 0xfe, 0x19, 0xe1, 0x13, 0xa0, 0xa7, 0x0a, 0xc7, 0x54, + 0x12, 0xd6, 0x40, 0xdc, 0x08, 0xf0, 0xf5, 0xc0, 0x24, 0x5f, 0xf8, 0x33, + 0xc8, 0xfb, 0x85, 0x9a, 0xaf, 0xe0, 0xc9, 0x0c, 0x91, 0x0a, 0x55, 0x60, + 0xf7, 0xa2, 0x07, 0xec, 0x8b, 0x6d, 0x02, 0x86, 0x43, 0xf3, 0xc2, 0x2e, + 0x87, 0x27, 0xcc, 0x86, 0xf0, 0xe0, 0x26, 0x8f, 0x3d, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, +}; + +// AnnounceMeasure is the announcement's byte count, which is a constant of the +// unit and not a walk. +inline int64_t AnnounceMeasure() { return kTableAnnounceBytes; } + +// Announce writes the announcement into the caller's buffer and answers the +// bytes written — exactly AnnounceMeasure's answer — or -1 when the buffer is +// too small. It allocates nothing and walks nothing. +inline int64_t Announce( uint8_t * buffer, int64_t capacity ) +{ + if ( buffer == NULL || capacity < kTableAnnounceBytes ) { return -1; } + memcpy( buffer, kTableAnnounce, (size_t) kTableAnnounceBytes ); + return kTableAnnounceBytes; +} + +// TableVocabulary is ONE DIRECTION of ONE CONNECTION's id table (§3.3): the +// entries an announcement carried, whole, under one numbering with slot 1 the +// reserved build-version id. +// +// A peer holds TWO of these for a connection, the one it writes with and the +// one it reads with, and neither is the other's. A restart opens a fresh +// connection with empty tables and nothing is cached across connections, so +// its whole life is one connection's. It BORROWS the announcement's bytes rather than +// copying them, so a receiver holds one table a direction and its memory is +// the bound below and nothing else. +struct TableVocabulary +{ + // THE CONFORMING DEFAULT BOUND (§3.3): 32 KiB a direction, eight times the + // 500-id unit that is already a large one. A connection's table is bounded + // by nothing the wire carries, so the receiver declares the maximum and an + // announcement above it is refused by name before an entry is touched. + static const int64_t kDefaultMaxEntries = 4096; + + TableIdTable table; + uint64_t build_version = 0; + bool announced = false; + int64_t max_entries = kDefaultMaxEntries; +}; + +// AnnounceRead reads an announcement into one direction's table (§3.3). +// +// THE BOUND IS CHECKED BEFORE ANYTHING IS ALLOCATED: the entry count is a +// fixed little-endian u64 at the end, so a receiver reads it, compares it and +// refuses without touching an entry. After that it is §3's ordinary FILE read, +// because the announcement IS a file, with EXACTLY ONE STRICT CHECK over its +// body: the reserved build-version field present, exactly once, under kind 9, +// eight bytes wide. Everything else is an ordinary field under §4's tolerance, +// so an unknown one is skipped and counted and the announcement can GAIN a +// field in a later minor without a lockstep redeploy. +// +// The FIRST announcement sets the table and it is the only one that can. A +// SECOND is refused by name: it does not replace the table, it does not amend +// it and it changes nothing. A refused announcement sets NO TABLE. +inline bool AnnounceRead( TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + if ( vocabulary.announced ) + { + to->refused = true; + to->reason = second_announcement; + return false; + } + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireForm ) + { + to->refused = true; + to->reason = buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + return false; + } + if ( bytes < 9 ) { to->malformed = true; return false; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + if ( ( lo | ( hi << 32 ) ) > (uint64_t) vocabulary.max_entries ) + { + to->refused = true; + to->reason = vocabulary_too_large; + return false; + } + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else { to->refused = true; to->reason = newer_form; } + return false; + } + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) { to->malformed = true; return false; } + // the body, under §4's tolerance and this form's one strict check + TableReader r( buffer + 1, body_bytes, to, &table ); + uint64_t version = 0; + int32_t seen = 0; + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { to->malformed = true; return false; } + if ( ref == 0 ) { break; } + if ( ref > (uint64_t) table.count || !r.has( 1 ) ) { to->malformed = true; return false; } + const uint64_t id = table.at( ref ); + const uint8_t kind = r.get8(); + if ( id != kTableBuildVersionFieldId ) + { + to->unknown++; + if ( !r.skip( kind ) ) { to->malformed = true; return false; } + continue; + } + if ( kind != 9 || !r.has( 8 ) ) { to->refused = true; to->reason = no_vocabulary; return false; } + version = r.get64(); + seen++; + } + if ( seen != 1 ) { to->refused = true; to->reason = no_vocabulary; return false; } + vocabulary.table = table; + vocabulary.build_version = version; + vocabulary.announced = true; + return true; +} + +inline float table_bits_to_float( uint32_t bits ) { float f; memcpy( &f, &bits, 4 ); return f; } +inline uint32_t table_float_to_bits( float f ) { uint32_t b; memcpy( &b, &f, 4 ); return b; } +inline double table_bits_to_double( uint64_t bits ) { double d; memcpy( &d, &bits, 8 ); return d; } +inline uint64_t table_double_to_bits( double d ) { uint64_t b; memcpy( &b, &d, 8 ); return b; } + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +#ifndef LISTDEMO_SCHEMA_TABLE_ARENA +#define LISTDEMO_SCHEMA_TABLE_ARENA + +namespace listdemo { + +// ---- variable-length tables: tuning constants (docs/SPEC-TABLES.md) ---- +// +// The segment size and the count multiply to exactly 2^32: the u32 reference +// is the arena's hard ceiling, and these constants saturate it rather than +// leaving address space unreachable. Slab handout costs one atomic per slab, +// so per-node allocation costs no synchronization at all. + +static const uint32_t kTableSegmentBits = 22; // 4 MiB segments +static const uint32_t kTableSegmentSize = 1u << kTableSegmentBits; +static const uint32_t kTableSegmentMask = kTableSegmentSize - 1u; +static const uint32_t kTableMaxSegments = 1u << ( 32 - kTableSegmentBits ); // 1024 -> 4 GiB +static const uint32_t kTableSlabBytes = 64u * 1024u; // one atomic per slab +static const uint32_t kTableAlign = 8; // every node starts 8-aligned +static const uint32_t kTableAllocFailed = 0xFFFFFFFFu; + +// ---- THE CALLER'S ALLOCATOR (docs/SPEC-TABLES.md §6.5) ---- +// +// Every allocation the variable-length runtime makes goes through one of +// these — the arena's segments, the pack walk's identity map, the numbering's +// entry array, the packed region, and the tool path's node directory. There is +// no other call to the C library on this path, so a counting allocator sees +// every byte and a game's own heap can own all of it. +// +// It is the shape TableBlockAllocator already has (§19.1): two function +// pointers and a context the caller carries. What it adds is a CONTRACT ON +// alloc — the bytes come back ZEROED. Lock copies whole nodes, PADDING +// INCLUDED, so anything left uninitialized reaches a packed region; the default +// pair reaches that through calloc, which costs nothing measurable because a +// fresh segment is untouched pages either way. +struct TableAllocator +{ + void * ( *alloc )( void * context, int64_t bytes ); // ZEROED bytes, NULL on failure + void ( *free )( void * context, void * pointer ); + void * context; +}; + +// The default pair, and it is the one every entry point takes when the caller +// names none. It calls schema_allocate / schema_release, so a program with its +// own C-library replacement can move the floor without writing a struct at all. +inline void * table_default_alloc( void * context, int64_t bytes ) { (void) context; return schema_allocate( bytes ); } +inline void table_default_free( void * context, void * pointer ) { (void) context; schema_release( pointer ); } + +inline TableAllocator TableDefaultAllocator() +{ + TableAllocator allocator; + allocator.alloc = table_default_alloc; + allocator.free = table_default_free; + allocator.context = NULL; + return allocator; +} + +// ---- TableRef: a relocatable reference (never a machine pointer) ---- +// +// Two encodings, one slot, and the FORM says which is in force: +// +// in the arena — the node's arena offset (segment index in the high bits) +// in a region — the SELF-RELATIVE byte delta from this slot's own address, +// so a deref is one add, needs no base pointer, and a whole +// region relocates by memcpy with zero fix-up +// +// 0 is null in both, and a slot can never name the node that contains it, so +// zero names nothing real in either form. +// +// A REGION DELTA HAS NO REQUIRED SIGN (§6.3). A region is packed depth-first, +// so a node's FIRST reference points forward; every LATER reference to that +// same node points BACK at the one body it already has, which is exactly what +// makes one node one node in a region. Sharing and a back-reference are the +// same fact, and nothing validates a reference by its sign. +// +// IT IS EIGHT BYTES, SIGNED, so ONE REGION REACHES EVERYTHING (§6.3, §7): a +// four-byte slot bounded a region at 2 GiB, and the scale a cook exists for is +// *"100mbs or many gigabytes of data in Assets.bin"*. +struct TableRef +{ + int64_t value = 0; + bool null() const { return value == 0; } +}; + +// TableSlot is what Alloc hands back: usable as the node pointer (write +// fields through it) AND as the reference to store in a pointer field. +template struct TableSlot +{ + T * ptr = NULL; + TableRef ref; + T * operator->() const { return ptr; } + T & operator*() const { return *ptr; } + operator T *() const { return ptr; } + operator TableRef() const { return ref; } + bool null() const { return ptr == NULL; } +}; + +inline uint32_t TableAlignUp( uint32_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( kTableAlign - 1 ); } +inline int64_t TableAlignUp64( int64_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( int64_t( kTableAlign ) - 1 ); } + +// ---- a BYTE BUFFER's node (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// A *bytes or *string slot is a TableRef like every pointer slot, and it names +// a BLOB NODE: this eight-byte header and then the bytes, at offset eight so +// the data is eight-aligned. A *string blob carries one more zero byte after +// its data, so a region hands back a C string with no copy. The node's extent +// is the header plus its bytes, rounded to the arena's alignment like every +// node's; on the wire it is a record whose body is the bytes (§3.1). +struct TableBlob +{ + uint32_t length; + uint32_t zero; +}; + +static const int64_t kTableBlobHeader = 8; // length (u32), then four zero bytes +static const int64_t kTableBlobMaxLength = 0xFFFFFFFF; // a record's length is a u32 (§3.1) + +// the node's storage: the header, the bytes, a string's terminator, rounded +// to the arena's alignment like every node +inline int64_t TableBlobStorage( int64_t length, bool terminated ) +{ + return TableAlignUp64( kTableBlobHeader + length + ( terminated ? 1 : 0 ) ); +} + +// What a read answers: a pointer INTO the region and the length, NULL and +// zero for a null slot. Off a locked region, a loaded one or an opened cook +// the pointer is one add from the slot, and nothing is copied. +struct TableBytesView +{ + const uint8_t * data; + int64_t length; +}; + +struct TableStringView +{ + const char * data; // zero-terminated + int64_t length; +}; + +// What AllocBytes and AllocString hand back: the bytes to write through, the +// length asked for, and the reference to store in the slot — the three +// answers TableSlot gives for a table node. +struct TableBytesSlot +{ + uint8_t * data = NULL; + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +struct TableStringSlot +{ + char * data = NULL; // room for length bytes and the terminator, already zero + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +// ---- the arena: segmented, slab-handed, lock-free by ownership ---- +// +// Allocation is thread-local inside a worker's slab — no atomics on the node +// path. A worker takes its next slab with ONE compare-exchange, and a new +// segment is published with one more. Nothing ever moves: a segment, once +// allocated, lives untouched until the arena is torn down, so a T* obtained +// from Alloc stays valid while other workers allocate, and an offset stays +// correct while the arena grows. +// +// The model this DELIBERATELY refuses: one buffer under a lock, grown by +// realloc. A realloc moves the buffer under workers mid-write; offsets fix +// identity but not the raw references already resolved from them, and the +// resulting corruption is invisible until much later. Segments never move, so +// that bug class cannot be written here. +// +// Slack: at most one slab tail per worker plus one slab per segment (a slab +// that will not fit is skipped rather than split), i.e. under 2% of a segment +// plus threads x 64 KiB. That is the price of never synchronizing per node. +struct TableArena +{ + std::atomic segments[ kTableMaxSegments ]; + std::atomic cursor; // (segment << kTableSegmentBits) | bytes handed out + bool locked = false; // MONOTONIC: Lock() is one-way, there is no unlock + // THE ARENA CARRIES ITS OWN, so everything downstream of a builder — + // segments, pack map, numbering, region, node directory — allocates through + // the one pair the caller named, with nothing to thread by hand. + TableAllocator allocator; +}; + +inline void TableArenaInit( TableArena & arena, TableAllocator allocator ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + arena.segments[i].store( NULL, std::memory_order_relaxed ); + } + arena.cursor.store( 0, std::memory_order_relaxed ); + arena.locked = false; + arena.allocator = allocator; +} + +inline void TableArenaShutdown( TableArena & arena ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + uint8_t * segment = arena.segments[i].exchange( NULL, std::memory_order_acq_rel ); + if ( segment != NULL ) { arena.allocator.free( arena.allocator.context, segment ); } + } + arena.cursor.store( 0, std::memory_order_relaxed ); +} + +// one L1 load plus an add: the segment table is 8 KiB and stays hot +inline uint8_t * TableArenaAt( const TableArena & arena, uint32_t offset ) +{ + return arena.segments[ offset >> kTableSegmentBits ].load( std::memory_order_relaxed ) + ( offset & kTableSegmentMask ); +} + +// TableArenaGrabSlab hands one worker its next private slab. Returns +// kTableAllocFailed when the arena's address space or the allocator is +// exhausted — a loud refusal, never a silent smaller slab. +inline uint32_t TableArenaGrabSlab( TableArena & arena ) +{ + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t segment = cursor >> kTableSegmentBits; + uint32_t used = cursor & kTableSegmentMask; + // strictly less: a slab is never split across segments, and the tail + // is the documented slack + if ( used + kTableSlabBytes < kTableSegmentSize ) + { + if ( arena.segments[segment].load( std::memory_order_acquire ) == NULL ) + { + // THE SEGMENT COMES BACK ZEROED, which is the allocator's + // contract and not an extra pass here: Lock copies whole nodes, + // PADDING INCLUDED, so anything uninitialized reaches a packed + // region. Value-initializing a node with placement new zeroes + // its MEMBERS and not its padding, so the zeroing has to happen + // at the segment or not at all. It costs nothing measurable: a + // fresh segment is untouched pages either way, and the default + // pair's calloc has the kernel hand them over zeroed. + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, (int64_t) kTableSegmentSize ); + if ( memory == NULL ) { return kTableAllocFailed; } + uint8_t * expected = NULL; + if ( !arena.segments[segment].compare_exchange_strong( expected, memory, std::memory_order_acq_rel ) ) + { + // another worker published this segment first + arena.allocator.free( arena.allocator.context, memory ); + } + } + if ( arena.cursor.compare_exchange_weak( cursor, cursor + kTableSlabBytes, std::memory_order_acq_rel ) ) + { + return ( segment << kTableSegmentBits ) | used; + } + continue; + } + uint32_t next_segment = segment + 1; + if ( next_segment >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + arena.cursor.compare_exchange_weak( cursor, next_segment << kTableSegmentBits, std::memory_order_acq_rel ); + } +} + +// TableArenaGrabSpan reserves a SPAN of the arena's address space for one node +// larger than a slab — a BYTE BUFFER of any size (docs/SPEC-TABLES.md §2.5) — +// and allocates it as one contiguous block. It takes whole segment indices +// from the cursor, starting at the index after the cursor's so nothing else +// is ever handed out inside the span, and publishes the block under the first +// of them; the indices the span covers past that one stay NULL, which is +// enough, because only a node's START is ever resolved through the segment +// table and a blob's bytes follow its header inside the one allocation. The +// unused tail of the segment the cursor was in is slack, like a slab tail. +// Returns kTableAllocFailed when the address space or the allocator is +// exhausted — a loud refusal, never a smaller blob. +inline uint32_t TableArenaGrabSpan( TableArena & arena, int64_t bytes ) +{ + if ( bytes <= 0 || bytes > ( (int64_t) kTableMaxSegments - 2 ) * (int64_t) kTableSegmentSize ) { return kTableAllocFailed; } + const uint32_t spanned = (uint32_t) ( ( bytes + kTableSegmentSize - 1 ) >> kTableSegmentBits ); + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t start = ( cursor >> kTableSegmentBits ) + 1; + if ( start + spanned >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + uint32_t next = ( start + spanned ) << kTableSegmentBits; + if ( !arena.cursor.compare_exchange_weak( cursor, next, std::memory_order_acq_rel ) ) { continue; } + // the span is this worker's now: nothing else can publish under its + // first index, so a plain store suffices, and the block comes back + // ZEROED like every segment — the blob's bytes and its tail are zeros + // until written + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, bytes ); + if ( memory == NULL ) { return kTableAllocFailed; } + arena.segments[start].store( memory, std::memory_order_release ); + return start << kTableSegmentBits; + } +} + +// ---- TableWorker: one thread's allocation front ---- +// +// The threading contract, stated plainly: +// * Alloc on YOUR OWN worker is safe concurrently with any other worker's. +// No locks, no atomics per node. +// * Writing fields of a node ANOTHER worker allocated is your own +// synchronization problem — this runtime does not arbitrate it. +// * Lock and Save are single-threaded: call them after the workers have +// joined. +struct TableWorker +{ + TableArena * arena = NULL; + uint32_t next = 0; + uint32_t end = 0; + + template TableSlot Alloc() + { + static_assert( alignof( T ) <= kTableAlign, "a table node's alignment must fit the arena's" ); + TableSlot slot; + if ( arena == NULL || arena->locked ) { return slot; } + uint32_t bytes = TableAlignUp( (uint32_t) sizeof( T ) ); + if ( bytes > kTableSlabBytes ) { return slot; } // a node larger than a slab: refused, never split + if ( end == 0 || next + bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return slot; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + uint32_t at = next; + next += bytes; + // A NODE IS BORN IN TWO HALVES: start its lifetime in the raw + // storage, then write the declared defaults ONE MEMBER AT A TIME. + // + // It is "T", not "T{}". Value-initialising the whole aggregate says + // the same thing and costs cl O(BYTES) TO COMPILE — it expands element + // by element in its front end — while both halves here cost + // O(declarations). The slab cap below refuses a large node at RUN + // TIME and bounds nothing at compile time: the cost is paid by + // whatever T a caller instantiates this with. + // Padding is not the difference: value-initialisation zeroes MEMBERS + // and not padding either way, which is why the segment is calloc'd. + // + // TableReset is an OVERLOAD SET, one per closure member, reached from + // this template by argument-dependent lookup on T's own namespace — + // Alloc is a template and cannot spell Reset. + // + // The reset is here because ONE DEFINITION SAYS WHAT THE DECLARED + // DEFAULTS ARE, and it is Reset. Default-initialisation lands on + // the same values today, because a member with a non-zero default + // carries a member initializer that says so — but that is the class + // definition agreeing with Reset, not the arena reading it, and #320's + // fix was itself a pass that MOVED initialisation between the two. + // The arena reads the definition. + slot.ptr = new ( TableArenaAt( *arena, at ) ) T; + TableReset( *slot.ptr ); + slot.ref.value = at; + return slot; + } + + // Alloc a BYTE BUFFER's node of exactly length bytes (docs/SPEC-TABLES.md + // §2.5): the blob header and its bytes, zeroed, in this thread's slab when + // it fits and in a span of the arena's own when it does not. NULL is the + // arena locked, a length below zero or past a record's u32, or the + // allocator refusing. The offset comes back for the reference. + TableBlob * AllocBlob( int64_t length, bool terminated, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( length < 0 || length > kTableBlobMaxLength ) { return NULL; } + const int64_t bytes = TableBlobStorage( length, terminated ); + if ( bytes > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, bytes ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + } + else + { + if ( end == 0 || next + (uint32_t) bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) bytes; + } + TableBlob * blob = (TableBlob *) TableArenaAt( *arena, at ); + blob->length = (uint32_t) length; // the bytes after it are the segment's zeros + blob->zero = 0; + return blob; + } + + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries + // no type id, takes no index and has no Reset, so it goes through the same + // slab and span the blob path uses rather than through Alloc. + uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( bytes <= 0 || align > (int64_t) kTableAlign ) { return NULL; } + const int64_t rounded = TableAlignUp64( bytes ); + if ( rounded > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, rounded ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + return TableArenaAt( *arena, at ); + } + if ( end == 0 || next + (uint32_t) rounded > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) rounded; + return TableArenaAt( *arena, at ); // the segment came back zeroed + } + // a *bytes node: the bytes to write through, and the reference to store + TableBytesSlot AllocBytes( int64_t length ) + { + TableBytesSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, false, at ); + if ( blob == NULL ) { return slot; } + slot.data = (uint8_t *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } + + // a *string node: room for length bytes and the zero byte after them + TableStringSlot AllocString( int64_t length ) + { + TableStringSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, true, at ); + if ( blob == NULL ) { return slot; } + slot.data = (char *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } +}; + +// ---- TablePackMap: the pack walk's identity map (docs/SPEC-TABLES.md §3.1, §6.2) ---- +// +// ONE ENTRY PER REACHABLE NODE, and that map IS identity: a node must know +// where it landed to be named a second time, so Lock packs a shared node ONCE +// and every later reference resolves to the one body it already has. That is +// the same first-visit numbering the wire uses, so the pack order and the node +// order are one order. +// +// COLOURING AN ENTRY WHILE ITS DESCENT IS OPEN COSTS ONE BIT, and it is what +// makes a data cycle free to refuse: a reference to an entry still open is a +// cycle, and Lock returns failure rather than recursing away. The ROOT's entry +// is open for the whole walk. +// +// The map is proportional to NODES, never to bytes, and it lives on the +// AUTHORING side, where §6.5 licenses allocation. Nothing on the reading path +// ever builds one. +struct TablePackEntry +{ + const void * key; // the node's address in the graph being packed + int64_t offset; // where that node landed in the region + uint8_t open; // its descent is still open: a reference here is a cycle +}; + +struct TablePackMap +{ + TablePackEntry * entries = NULL; + int64_t capacity = 0; // a power of two, or zero while empty + int64_t count = 0; + TableAllocator allocator; // the caller's, carried from the walk that built it +}; + +inline void TablePackMapInit( TablePackMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TablePackMapShutdown( TablePackMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TablePackMapInit( map, map.allocator ); +} + +// The two walks behind Lock re-derive the SAME map from the same graph — the +// numbering is never carried between them (§3.1) — so the second starts from +// an empty map and keeps the capacity the first paid for. +inline void TablePackMapReset( TablePackMap & map ) +{ + if ( map.entries != NULL ) { memset( map.entries, 0, (size_t) map.capacity * sizeof( TablePackEntry ) ); } + map.count = 0; +} + +// open addressing, linear probing, a multiply-shift hash over the address: a +// node key is a pointer and its low bits are alignment, so the low bits alone +// would collide on every node of one type +inline int64_t TablePackMapSlot( const TablePackMap & map, const void * key ) +{ + uint64_t hash = (uint64_t) (uintptr_t) key; + hash *= 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != NULL && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TablePackEntry * TablePackMapFind( TablePackMap & map, const void * key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TablePackEntry * entry = &map.entries[ TablePackMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +// QUADRUPLING, not doubling, and the reason is measured: growth rehashes every +// entry, and on a graph of 131,071 nodes the doubling schedule spent 45% of +// Lock in rehashing alone. Quadrupling from 1024 buys 1.35x on that graph and +// keeps the map NODE-proportional (§6.2) — under 128 bytes a node at its +// worst, right after a grow, and about 64 on average. +inline bool TablePackMapGrow( TablePackMap & map ) +{ + TablePackMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 1024; + grown.entries = (TablePackEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TablePackEntry ) ); + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == NULL ) { continue; } + grown.entries[ TablePackMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// REACH a node: one probe answers both questions the walk has. A true "taken" +// says this is a FIRST visit, and the entry is now the node's, coloured open +// at "offset"; otherwise the entry is the one the node already has, and its +// open bit says cycle or sharing. NULL is an allocation failure, and it is a +// refusal like any other: Lock fails rather than packing a graph it cannot +// track. +// +// It is one call and not a find followed by an insert because the walk asks +// this question twice per node — once to measure, once to pack — and every +// probe is a miss into a table larger than L2. +inline TablePackEntry * TablePackMapReach( TablePackMap & map, const void * key, int64_t offset, bool & taken, int64_t & slot ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TablePackMapGrow( map ) ) { return NULL; } + } + slot = TablePackMapSlot( map, key ); + TablePackEntry * entry = &map.entries[slot]; + taken = entry->key != key; // an empty slot is a first visit; the key is never NULL + if ( taken ) + { + entry->key = key; + entry->offset = offset; + entry->open = 1; + map.count++; + } + return entry; +} + +// The descent finished: the node keeps its entry — identity outlives the +// descent — and stops being a cycle. The "hint" is the slot Reach returned, and it +// is checked against the key rather than trusted, so a rehash between the two +// costs a second probe instead of correctness. +inline void TablePackMapClose( TablePackMap & map, const void * key, int64_t hint ) +{ + if ( hint >= 0 && hint < map.capacity && map.entries[hint].key == key ) + { + map.entries[hint].open = 0; + return; + } + TablePackEntry * entry = TablePackMapFind( map, key ); + if ( entry != NULL ) { entry->open = 0; } +} + +// ---- resolution contexts: which encoding a walk is reading ---- + +struct TableArenaCtx { const TableArena * arena; }; +struct TableRegionCtx {}; + +// ---- a BYTE BUFFER's resolution (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// The same two encodings a table pointer has, resolved the same way: a +// self-relative delta in a region — one add, no base — and an arena offset +// while the builder is mutable. The blob is reached through its header, and a +// view is the header plus eight and the header's first word. Nothing here +// allocates and nothing copies: off a locked region, a loaded one or an +// opened cook the view points INTO the region. +inline const TableBlob * TableBlobAt( const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableRegionCtx &, const TableRef & ref ) { return TableBlobAt( ref ); } +inline const TableBlob * TableBlobAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} + +inline TableBytesView TableBytesViewOf( const TableBlob * blob ) +{ + TableBytesView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const uint8_t *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} +inline TableStringView TableStringViewOf( const TableBlob * blob ) +{ + TableStringView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const char *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} + +// the const form's hot path: one add, no base +inline TableBytesView TableBytesAt( const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ref ) ); } +inline TableStringView TableStringAt( const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ref ) ); } +// and the context forms a walk uses: a region context, an arena context, or +// the arena itself while the builder is mutable +template inline TableBytesView TableBytesAt( const Ctx & ctx, const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ctx, ref ) ); } +template inline TableStringView TableStringAt( const Ctx & ctx, const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ctx, ref ) ); } + +// allocate a blob in the arena and point the slot at it; the slot holds the +// arena offset, as every slot does while the builder is mutable +inline uint8_t * TableBytesEmplace( TableWorker & worker, TableRef & slot, int64_t length ) +{ + TableBytesSlot allocated = worker.AllocBytes( length ); + slot = allocated.ref; + return allocated.data; +} +// the text is copied in when one is given; a NULL text leaves the zeros for +// the caller to fill +inline char * TableStringEmplace( TableWorker & worker, TableRef & slot, const char * text, int64_t length ) +{ + TableStringSlot allocated = worker.AllocString( length ); + slot = allocated.ref; + if ( allocated.data != NULL && text != NULL && length > 0 ) { memcpy( allocated.data, text, (size_t) length ); } + return allocated.data; +} + +// ---- the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table, and a +// pointer field rides as an INDEX into it under kind 17. The encoding is +// flat: no pointer edge is a nesting level, so a chain's length is not a depth, +// and two references to one node are one node. +// +// THE FIELD RIDES ONCE: an L with sixty-four bits of capability frames a +// numbering of any size, so the whole numbering is one contiguous payload and a +// save's node bodies have no aggregate ceiling. + +static const uint64_t kTableNodeIndexNull = 0; // absence and null are one value +static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts the table + +// The not-materialized sentinel (§6.3): a record whose type id this build could +// not name. Distinct from every real offset including the root's 0, so an index +// resolving through it yields NULL and can never fabricate the root. +static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; + +// ---- the numbering, on the SAVE side ---- +// +// One entry per reachable node in FIRST-VISIT order, so entry k is node index +// k + 2. The two thunks are what let one loop write a table of mixed types: the +// numbering walk knows each target's type STATICALLY at the site it numbers it, +// so it stores the instantiation there and the loop never asks what a node is. +struct TableNumbering; + +struct TableNodeEntry +{ + const void * node; + uint64_t type_id; + // the type id's MESSAGE-FORM SLOT (docs/SPEC-TABLES.md §3.3), stored where + // the numbering walk stores the id itself and for the same reason: the + // target's type is known STATICALLY at the site that numbers it, so a + // form 2 save reads the slot out of the entry instead of looking an id up. + // Every pointer target's type id is an entry of the announcement, which is + // what makes the slot a compile-time fact of a POINTERED message too. + uint64_t type_slot; + int64_t ( * measure )( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ); + bool ( * save )( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ); +}; + +struct TableNumbering +{ + TablePackMap seen; // node -> index; the ROOT is index 1, open for the whole walk + TableNodeEntry * entries = NULL; + int64_t count = 0; + int64_t capacity = 0; +}; + +// The numbering allocates through the map's pair rather than carrying a second +// copy of it: one numbering is one walk, and a walk has one allocator. +inline void TableNumberingInit( TableNumbering & n, TableAllocator allocator ) +{ + TablePackMapInit( n.seen, allocator ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +inline void TableNumberingShutdown( TableNumbering & n ) +{ + TableAllocator allocator = n.seen.allocator; + TablePackMapShutdown( n.seen ); + allocator.free( allocator.context, n.entries ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +// The index a numbered node was given, for the save that writes it into a +// pointer slot. False means the two walks disagree about the graph, which is a +// refusal and never a guess. +inline bool TableNumberingIndex( const TableNumbering & n, const void * node, uint64_t & index ) +{ + if ( n.seen.capacity == 0 ) { return false; } + const TablePackEntry & entry = n.seen.entries[ TablePackMapSlot( n.seen, node ) ]; + if ( entry.key != node ) { return false; } + index = (uint64_t) entry.offset; + return true; +} + +inline bool TableNumberingAppend( TableNumbering & n, const TableNodeEntry & entry ) +{ + if ( n.count == n.capacity ) + { + // GROW BY COPY, never by realloc: the allocator hook is a PAIR, and a + // game's heap is not required to have a resize primitive at all. The + // schedule quadruples, so the copying is amortized to a constant per + // entry and the growth is the same growth it always was. + int64_t capacity = n.capacity != 0 ? n.capacity * 4 : 256; + TableAllocator allocator = n.seen.allocator; + TableNodeEntry * grown = (TableNodeEntry *) allocator.alloc( allocator.context, capacity * (int64_t) sizeof( TableNodeEntry ) ); + if ( grown == NULL ) { return false; } + if ( n.entries != NULL ) + { + memcpy( grown, n.entries, (size_t) n.count * sizeof( TableNodeEntry ) ); + allocator.free( allocator.context, n.entries ); + } + n.entries = grown; + n.capacity = capacity; + } + n.entries[n.count++] = entry; + return true; +} + +// The thunks the numbering stores. Each resolves to the closure member's own +// MeasureBody / SaveBodyFields through an overload set in the member's DECLARING +// file, reached by argument-dependent lookup at instantiation — the same bridge +// the arena's TableReset uses, and the reason a numbering may span the files of +// one unit without any file naming another's members. +template +inline int64_t TableNodeMeasureThunk( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ) +{ + return TableNodeMeasure( *(const Ctx *) ctx, numbering, ids, *(const T *) node ); +} + +template +inline bool TableNodeSaveThunk( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ) +{ + return TableNodeSave( *(const Ctx *) ctx, numbering, w, ids, *(const T *) node ); +} + +// ---- a BYTE BUFFER's record (docs/SPEC-TABLES.md §2.5, §3.1) ---- +// +// A blob rides as a node record under one of two RESERVED type ids — the fold +// a table's name takes, over the keywords "bytes" and "string", which no table +// can be named — with the bytes as its body and nothing framed inside. These +// two thunks are what the numbering stores for a blob, as it stores a +// member's codec for a table: the length, and the bytes verbatim. +static const uint64_t kTableBytesTypeId = 0x2f2ec0474f1c4fe4ull; // fnv1a64( "bytes" ) +static const uint64_t kTableStringTypeId = 0x704be0d8faaffc58ull; // fnv1a64( "string" ) + +template +inline int64_t TableBlobMeasureThunk( const void *, const TableNumbering &, TableIds &, const void * node ) +{ + return (int64_t) ( (const TableBlob *) node )->length; +} + +template +inline bool TableBlobSaveThunk( const void *, const TableNumbering &, TableWriter & w, TableIds &, const void * node ) +{ + const TableBlob * blob = (const TableBlob *) node; + w.raw( (const void *) ( blob + 1 ), (int64_t) blob->length ); + return true; +} + +// TableNodeTableMeasure and TableNodeTableSave are the framing, and they are +// ONE fill rule written twice — measure derives it from the graph and save +// derives the same one, which is what makes measure == save hold across a +// pointer graph (§3.1). +// +// The field rides ONCE, under the reserved id, kind 12: the payload opens with +// the count and then carries the records back to back, each a type id +// REFERENCE, a length and a body. The reserved id is interned BEFORE the +// records, and a record's type id before its body, which is the first-use order +// the trailer is written in (§3). +template +inline int64_t TableNodeTablePayload( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + int64_t payload = TableLebBytes( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + payload += TableLebBytes( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return -1; } + payload += TableLebBytes( (uint64_t) body ) + body; + } + return payload; +} + +template +inline int64_t TableNodeTableMeasure( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return 0; } // a root that reaches no nodes writes none of them + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return -1; } + return TableLebBytes( ref ) + 1 + TableLebBytes( (uint64_t) payload ) + payload; +} + +template +inline bool TableNodeTableSave( const Ctx & ctx, TableWriter & w, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return true; } + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return false; } + w.putleb( ref ); + w.put8( 12 ); // kind 12 is the opaque byte payload: a reader that cannot name the id skips by L + w.putleb( (uint64_t) payload ); + w.putleb( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + w.putleb( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return false; } + w.putleb( (uint64_t) body ); + if ( !n.entries[k].save( (const void *) &ctx, n, w, ids, n.entries[k].node ) ) { return false; } + } + return true; +} + +// ---- the numbering, on the LOAD side: a region's NODE DIRECTORY (§6.3) ---- +// +// The wire's numbering made resident: one entry per numbered node, in index +// order, position i describing node index i + 1 — so position 0 is the ROOT at +// offset 0. It is ATTRIBUTION, and attribution is separable: nothing that reads +// a structure touches it, a deref is one add on a self-relative offset, and a +// caller may release it once Load returns. +struct TableNodeDirEntry +{ + uint64_t offset; + uint64_t type_id; +}; + +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; + +// TableNodeMap is what a pointer slot resolves through while a body decodes. +struct TableNodeMap +{ + uint8_t * base = NULL; + const TableNodeDirEntry * entries = NULL; + int64_t count = 0; // the ROOT's entry included, so it is records + 1 + bool good = false; // the node table read whole; a numbering that failed resolves nothing + // WHERE THE NODES LIVE, and therefore what a resolved slot holds: a region + // takes the SELF-RELATIVE delta so a deref is one add, and the tool's + // builder path takes the node's ARENA OFFSET (§6.3). + bool arena = false; + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. + TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; +}; + +// TableNodeResolve places one node index in a pointer slot, and every failure +// is one of §4's events with the pointer left null. The declared TARGET type id +// is checked at every index, the root's included: the root carries no record +// and therefore no wire type id, so the READER'S OWN root type is what the +// claim is checked against. +inline void TableNodeResolve( const TableNodeMap & map, TableRef & slot, uint64_t index, uint64_t target, TableReport * report ) +{ + slot.value = 0; + if ( index == kTableNodeIndexNull || !map.good ) { return; } + if ( index - 1 >= (uint64_t) map.count ) + { + report->malformed = true; // an index above node_count + 1 + return; + } + const TableNodeDirEntry & entry = map.entries[index - 1]; + if ( entry.offset == kTableNodeAbsent ) + { + // a node whose type id this build could not name KEEPS ITS INDEX, and + // every pointer naming it reads null. The unknown was counted once, at + // the node, not once per pointer. + return; + } + if ( entry.type_id != target ) + { + report->kind_mismatch++; + return; + } + slot.value = map.arena ? (int64_t) entry.offset + : (int64_t) ( ( map.base + entry.offset ) - (const uint8_t *) &slot ); +} + +// ---- the record SCAN, and it is the whole of load's bound (§3.1) ---- +// +// Reading follows no reference. The scan walks the root body's top-level fields, +// finds the ONE under the reserved id, and reads records out of its payload in +// order — the field rides once, so nothing is copied to make a body contiguous +// and the generated body decoder never learns the transport exists. +struct TableNodeScan +{ + TableReader fields; // over the ROOT body, skipping past everything else + const uint8_t * payload; // the node-table field's payload + int64_t payload_size; + int64_t payload_offset; + bool opened; // the root body has been walked for the field + uint64_t declared; + int64_t records; + bool present; // the root body carries a node table at all + bool malformed; + const TableIdTable * ids; +}; + +inline TableNodeScan TableNodeScanBegin( const uint8_t * body, int64_t size, TableReport * report, const TableIdTable * ids ) +{ + TableNodeScan s = { TableReader( body, size, report, ids ), NULL, 0, 0, false, 0, 0, false, false, ids }; + return s; +} + +// find the node-table field, or answer false when the root body has none. A +// body carrying an id more than once is legal input and THE LAST OCCURRENCE +// WINS (docs/SPEC-TABLES.md §3), so the walk runs to the terminator and keeps +// the last rather than stopping at the first. +inline bool TableNodeScanOpen( TableNodeScan & s ) +{ + if ( s.opened ) { return false; } + s.opened = true; + for ( ;; ) + { + uint64_t ref = 0; + if ( !s.fields.getleb( ref ) ) { break; } + if ( ref == 0 ) { break; } // the terminator + if ( s.ids == NULL || ref > (uint64_t) s.ids->count ) { break; } + const uint64_t id = s.ids->at( ref ); + if ( !s.fields.has( 1 ) ) { break; } + const uint8_t kind = s.fields.get8(); + if ( id == kTableNodeTableFieldId ) + { + s.present = true; + if ( kind != 12 ) { s.malformed = true; return false; } + uint64_t length = 0; + if ( !s.fields.getleb( length ) || !s.fields.room( length ) ) { s.malformed = true; return false; } + s.payload = s.fields.buffer + s.fields.offset; + s.payload_size = (int64_t) length; + s.fields.offset += (int64_t) length; + continue; + } + if ( !s.fields.skip( kind ) ) { break; } + } + if ( s.payload == NULL ) { return false; } + TableReader head( s.payload, s.payload_size, s.fields.report, s.ids ); + if ( !head.getleb( s.declared ) ) { s.malformed = true; return false; } + s.payload_offset = head.offset; + return true; +} + +// the next record, or false at the end of the table — s.malformed says whether +// the end was the end or the framing giving out +inline bool TableNodeScanNext( TableNodeScan & s, uint64_t & type_id, const uint8_t * & body, int64_t & length ) +{ + if ( !s.opened && !TableNodeScanOpen( s ) ) { return false; } + if ( s.payload == NULL || s.payload_offset >= s.payload_size ) { return false; } + TableReader rec( s.payload, s.payload_size, s.fields.report, s.ids ); + rec.offset = s.payload_offset; + uint64_t ref = 0; + if ( !rec.getleb( ref ) || ref == 0 || s.ids == NULL || ref > (uint64_t) s.ids->count ) + { + s.malformed = true; // a type id reference of 0, or one past the table + return false; + } + type_id = s.ids->at( ref ); + uint64_t declared_length = 0; + if ( !rec.getleb( declared_length ) ) + { + s.malformed = true; // a record whose length is damaged + return false; + } + if ( declared_length > (uint64_t) ( s.payload_size - rec.offset ) ) + { + s.malformed = true; // a record whose length runs past its field + return false; + } + body = s.payload + rec.offset; + length = (int64_t) declared_length; + s.payload_offset = rec.offset + length; + s.records++; + return true; +} + +// The record scan is AUTHORITATIVE: node_count is data from the wire, and a +// count that disagrees with the scan is malformed. Nothing is sized from it +// before the scan has confirmed it. +inline bool TableNodeScanWhole( TableNodeScan & s ) +{ + if ( s.malformed ) { return false; } + if ( !s.present ) { return true; } // no node table at all is not a broken one + return s.declared == (uint64_t) s.records; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_ARENA + +#ifndef LISTDEMO_SCHEMA_TABLE_EXTENT +#define LISTDEMO_SCHEMA_TABLE_EXTENT + +namespace listdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_EXTENT + +#ifndef LISTDEMO_SCHEMA_TABLE_MAP +#define LISTDEMO_SCHEMA_TABLE_MAP + +namespace listdemo { + +// ---- a MAP: a sorted entry array, and the lookup over it (§2.8) ---- +// +// On the wire, in a region and in a cook a map is an array of one generated +// ENTRY table held in ascending key order. What this adds is Find — a binary +// search over that array where it lies — and a builder that inserts, replaces +// and erases by key. Nothing here is stored: a region and a cook carry the +// array and the count, and not one byte about a hash or a probe. + +// entries carved from ONE call to the allocator pair; a new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableMapSegmentEntries = 32; + +// TableDeclRef names a type in an unevaluated context and is never defined — +// what 's declval is for, without the include the generated corpus +// refuses to pay for (the iterator_traits note, §13.9). +template T & TableDeclRef(); + +// THE ORDER IS TOTAL, AND IT IS THE SAME IN NINE LANGUAGES (§2.8). Integers +// compare by VALUE, signed for the signed kinds and unsigned for the unsigned. +// Strings compare by BYTES, unsigned, a shorter string that is a prefix of a +// longer one first: memcmp over the common length, then the lengths. Never a +// locale, never a code point, never a case fold. +inline int TableKeyOrder( uint64_t a, uint64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( int64_t a, int64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( const char * a, int32_t a_length, const char * b, int32_t b_length ) +{ + const int32_t common = a_length < b_length ? a_length : b_length; + if ( common > 0 ) + { + const int order = memcmp( (const void *) a, (const void *) b, (size_t) common ); + if ( order != 0 ) { return order < 0 ? -1 : 1; } + } + return a_length < b_length ? -1 : ( a_length > b_length ? 1 : 0 ); +} + +// the length of a NUL-terminated key at a call site, bounded by the storage it +// has to fit: a key one byte longer than the bound is refused, never truncated +inline int32_t TableKeyLength( const char * key, int32_t bound ) +{ + if ( key == NULL ) { return 0; } + for ( int32_t i = 0; i <= bound; i++ ) { if ( key[i] == 0 ) { return i; } } + return bound + 1; // longer than the bound: the caller refuses it +} + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.8, §7.2) ---- +// +// An int64 self-relative reference to the entry array and an int32 count, then +// padding to eight. The reference is a TableRef like a pointer's: in the arena +// it names the builder's HEAD, in a region it is the delta from the slot to +// the first entry, and 0 is the empty map in both. +template struct TableMap +{ + TableRef entries; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Entry * Entries() const + { + return entries.value != 0 ? (const Entry *) ( (const uint8_t *) &entries + entries.value ) : NULL; + } + int32_t size() const { return count; } + + // FIND: floor( log2 n ) + 1 key compares, in place, no allocation. NULL + // when absent, and on a map[K]*T the RESOLVED pointer, which is what a + // pointer field's accessor answers. + template const Entry * FindEntry( Key key ) const + { + const Entry * base = Entries(); + int32_t low = 0, high = count; + while ( low < high ) + { + const int32_t mid = low + ( high - low ) / 2; + const int order = TableEntryOrder( base[mid], key ); + if ( order == 0 ) { return base + mid; } + if ( order < 0 ) { low = mid + 1; } else { high = mid; } + } + return NULL; + } + // the return type is DEDUCED, so it is worked out when a call site + // instantiates Find and not when the holder's record declares the slot — + // which is what lets the entry's own overloads be declared after it + template auto Find( Key key ) const + { + return TableEntryFound( FindEntry( key ) ); + } + + // ---- iteration: ASCENDING key order, the key beside the value ---- + // + // A proxy BY VALUE, the keyed array's shape (§2.4): for ( auto [ key, + // value ] : map ). It carries no iterator_traits, for the reason + // TableKeyed's does not (§13.9). + struct ConstEntry + { + decltype( TableEntryKey( TableDeclRef() ) ) key; + decltype( TableEntryFound( (const Entry *) NULL ) ) value; + }; + + struct ConstIterator + { + const Entry * at; + ConstEntry operator*() const { return ConstEntry{ TableEntryKey( *at ), TableEntryFound( at ) }; } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Entries() }; } + ConstIterator end() const { return ConstIterator{ Entries() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.8, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first entry is inserted. Each +// segment is a fixed number of entries carved from one call to the allocator +// pair. An entry's address is stable for the arena's life, so a value handed +// back by an insert stays valid while other entries arrive. +struct TableMapHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an insert appends into + int32_t live; + int32_t dead; +}; + +template struct TableMapSegment +{ + TableRef next; + int32_t used; // entries carved from this segment + int32_t padding; + uint32_t dead[ ( kTableMapSegmentEntries + 31 ) / 32 ]; // Erase marks one bit, never the entry + Entry entries[ kTableMapSegmentEntries ]; +}; + +inline bool TableMapSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// ---- the ORDERED CURSOR the four writing walks read (§2.8) ---- +// +// Measure, Save, Lock and Cook each write a map's entries in ascending key +// order with no key twice, deriving the order from the builder's entries as +// each walk derives the numbering (§3.1). Nothing passes between them, so +// measure == save over a map is a real check on two sorts agreeing. +// +// A REGION is already sorted, so its cursor is the array in place and +// allocates nothing. The BUILDER's is the sort: an array of entry pointers +// allocated through the pair and released before the walk returns, because +// sorting the segments themselves would move entries whose addresses a caller +// holds. +template struct TableMapCursor +{ + const Entry * const * order = NULL; // the builder's form: sorted pointers + const Entry * entries = NULL; // the region's form: the array in place + int32_t count = 0; + TableAllocator allocator; + bool ok = false; + const Entry * operator[]( int32_t index ) const + { + return order != NULL ? order[index] : entries + index; + } +}; + +// heapsort: O( n log n ) once per map, no recursion, no allocation past the +// pointer array the caller already paid for +template inline void TableMapSort( const Entry ** order, int32_t count ) +{ + for ( int32_t start = count / 2 - 1; start >= 0; start-- ) + { + int32_t root = start; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= count ) { break; } + if ( child + 1 < count && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * swap = order[root]; order[root] = order[child]; order[child] = swap; + root = child; + } + } + for ( int32_t end = count - 1; end > 0; end-- ) + { + const Entry * swap = order[0]; order[0] = order[end]; order[end] = swap; + int32_t root = 0; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= end ) { break; } + if ( child + 1 < end && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * hold = order[root]; order[root] = order[child]; order[child] = hold; + root = child; + } + } +} + +// the REGION form: the array is already sorted, so the cursor is the array +template +inline TableMapCursor TableMapOrder( const TableRegionCtx &, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.entries = map.Entries(); + cursor.count = map.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: gather the LIVE entries out of the segment chain in +// insertion order, then sort. A dead entry costs nothing on any wire (§2.8). +template +inline TableMapCursor TableMapOrder( const TableArena & arena, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.allocator = arena.allocator; + cursor.count = map.count; + if ( map.entries.value == 0 || map.count <= 0 ) { cursor.ok = map.count == 0; cursor.count = 0; return cursor; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + if ( head->live != map.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + const Entry ** order = (const Entry **) arena.allocator.alloc( arena.allocator.context, (int64_t) map.count * (int64_t) sizeof( const Entry * ) ); + if ( order == NULL ) { return cursor; } + int32_t at = 0; + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 && at < map.count ) + { + const TableMapSegment * segment = (const TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used && at < map.count; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + order[at++] = segment->entries + i; + } + segment_ref = segment->next; + } + if ( at != map.count ) + { + arena.allocator.free( arena.allocator.context, order ); + return cursor; + } + TableMapSort( order, map.count ); + cursor.order = order; + cursor.ok = true; + return cursor; +} + +template +inline TableMapCursor TableMapOrder( const TableArenaCtx & ctx, const TableMap & map ) +{ + return TableMapOrder( *ctx.arena, map ); +} + +template inline void TableMapRelease( TableMapCursor & cursor ) +{ + if ( cursor.order != NULL ) { cursor.allocator.free( cursor.allocator.context, (void *) cursor.order ); } + cursor.order = NULL; +} + +// ---- the builder's five (§2.8) ---- +// +// Insert APPENDS after one LINEAR SCAN of the live entries for the key it may +// replace, Find is that same scan, and Erase is the scan and one bit. The +// builder builds NO INDEX, and that is a rule: the sort happens once, at Lock, +// Save or Cook, and every lookup that matters runs over the sorted region. + +// the head, allocated when the first entry is inserted +template +inline TableMapHead * TableMapReach( TableWorker & worker, TableMap & map ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( map.entries.value != 0 ) { return (TableMapHead *) TableArenaAt( *worker.arena, (uint32_t) map.entries.value ); } + uint32_t at = 0; + TableMapHead * head = (TableMapHead *) worker.AllocRaw( (int64_t) sizeof( TableMapHead ), (int64_t) alignof( TableMapHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + map.entries.value = (int64_t) at; + return head; +} + +// one entry's storage, appended: the current segment when it has room, a new +// one carved from one call to the pair when it does not +template +inline Entry * TableMapAppend( TableWorker & worker, TableMapHead * head, TableMap & map ) +{ + TableMapSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableMapSegmentEntries ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableMapSegment *) worker.AllocRaw( (int64_t) sizeof( TableMapSegment ), (int64_t) alignof( TableMapSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableMapSegment * previous = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Entry * entry = segment->entries + segment->used; + segment->used++; + head->live++; + map.count++; + return entry; +} + +// the LINEAR SCAN: the live entries in insertion order, O( n ) key compares +template +inline Entry * TableMapScan( const TableArena & arena, const TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) == 0 ) { return segment->entries + i; } + } + segment_ref = segment->next; + } + return NULL; +} + +// ERASE marks the entry DEAD, one bit in the segment's slot and not in the +// entry table, and decrements the live count. Its storage is reclaimed at +// RESET and never reused mid-build, because reusing a slot would make "an +// entry's address is stable" false for exactly one case. +template +inline bool TableMapErase( TableArena & arena, TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return false; } + TableMapHead * head = (TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) != 0 ) { continue; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + map.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INSERTION order, live entries only (§2.8) ---- +template struct TableMapEach +{ + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableMapSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableMapSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + auto operator*() const { return TableEntryEach( segment->entries + index ); } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableMapEach TableMapEachOf( const TableArena & arena, const TableMap & map ) +{ + TableMapEach each = { &arena, TableRef() }; + if ( map.entries.value != 0 ) + { + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + each.first = head->first; + } + return each; +} + +// ---- the LOAD side: where a decoded entry lands (§2.8) ---- +// +// THE READER TRUSTS NOTHING and spends one compare per entry. Every load path +// applies the same rules and produces one report (§4), so the region load of +// §6.5 and LoadBuilder never disagree about a wire. These two shapes are what +// makes that true with one generated decoder: a REGION carves the entry array +// out of the holder node's own extent, and the TOOL's path appends into the +// builder's arena, and the decoder above them cannot tell which it has. + +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. + +// TableMapFill is one map field being decoded: where the next entry lands, and +// the entry that last LANDED, which is what the ascending check compares +// against. +template struct TableMapFill +{ + TableMap * map = NULL; + Entry * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; +}; + +template +inline TableMapFill TableMapFillBegin( const TableNodeMap & nodes, TableMap & map, uint32_t n ) +{ + TableMapFill fill; + fill.map = ↦ + map.entries.value = 0; + map.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Entry ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Entry ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Entry *) base; + fill.capacity = (int32_t) n; + map.entries.value = (int64_t) ( base - (const uint8_t *) &map.entries ); + fill.ok = true; + return fill; +} + +// the entry that last LANDED — NULL before the first +template inline Entry * TableMapFillLast( TableMapFill & fill ) +{ + if ( fill.map->count <= 0 ) { return NULL; } + if ( fill.array != NULL ) { return fill.array + ( fill.map->count - 1 ); } + return TableMapLive( *fill.worker->arena, *fill.map, fill.map->count - 1 ); +} + +// the next slot, at the entry type's declared defaults +template inline Entry * TableMapFillNext( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + if ( fill.map->count >= fill.capacity ) { return NULL; } + Entry * entry = fill.array + fill.map->count; + TableReset( *entry ); + fill.map->count++; + return entry; + } + TableMapHead * head = TableMapReach( *fill.worker, *fill.map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( *fill.worker, head, *fill.map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// A MAP WITH HALF ITS KEYS IS NOT A MAP (§2.8): at the first entry whose key +// kind disagrees with the reader's declaration the map resets to EMPTY, one +// kind_mismatch is counted for the map, and its remaining bytes are skipped. +template inline void TableMapFillReset( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + fill.map->entries.value = 0; + fill.map->count = 0; + return; + } + if ( fill.map->entries.value != 0 ) + { + TableMapHead * head = (TableMapHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.map->entries.value ); + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + } + fill.map->count = 0; +} + +// an EMPTY map's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableMapFillEnd( TableMapFill & fill ) +{ + if ( fill.array != NULL && fill.map->count == 0 ) { fill.map->entries.value = 0; } +} + +// the k-th LIVE entry of a builder map, in insertion order — what the tool +// path's ascending check compares against +template +inline Entry * TableMapLive( const TableArena & arena, const TableMap & map, int32_t index ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + int32_t at = 0; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( at == index ) { return segment->entries + i; } + at++; + } + segment_ref = segment->next; + } + return NULL; +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.8, §6.5) ---- +// +// LoadMeasure's term for a map is N x sizeof( Entry ) rounded to +// alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this +// reads no field: it walks the map's own header and, where an entry's value +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect, +// and a refusal carries its reason (§6.5). +// A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its +// own L and the body's terminator, and under this form's variable lengths that +// footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a +// map's L can carry, and therefore what a LoadMeasure may be asked for. +static const int64_t kTableMapEntryFloor = 2; + +inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry + at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); + at += (int64_t) n * entry_size; + if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// ---- the TEXT form's placement (docs/SPEC-TABLES.md §2.8, §16) ---- +// +// The text is a plain JSON object keyed by the KEY, and the generic walk fills +// it through the ENTRY'S OWN descriptor — so all it needs from here is one +// entry at one key, handed back at its defaults. It is the builder's Insert +// with the ENTRY returned rather than its value, because the walk writes the +// value through a field row and not through a typed pointer. +template +inline Entry * TableMapPlace( TableWorker & worker, TableMap & map, Key key ) +{ + if ( worker.arena == NULL ) { return NULL; } + Entry * found = TableMapScan( *worker.arena, map, key ); + if ( found != NULL ) + { + TableResetMapValue( *found ); // a repeated key is LAST-WINS, whole + return found; + } + TableMapHead * head = TableMapReach( worker, map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( worker, head, map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// ---- the OPTIONAL RUNTIME INDEX (§2.8) ---- +// +// Open addressing with LINEAR PROBING over the sorted array, built AT LOAD for +// a map large enough that log n compares over a cold array cost more than one +// hash and a probe. IT IS NEVER STORED: the caller measures it, owns its +// storage, builds it in one pass and releases it whenever. +// +// ITS HASH AND ITS LOAD FACTOR ARE NOT A CROSS-PORT CONTRACT, and that is a +// rule. What a port is held to is the CONTRACT of the lookup: the same value +// the sorted array's Find returns for the same key, and no allocation past the +// storage the caller handed in. +struct TableMapIndex +{ + int32_t * slots = NULL; // entry indices, +1; 0 is an empty slot + int32_t capacity = 0; + bool good = false; +}; + +// this runtime's own, and no port reproduces it: fnv1a64 over the key's bytes +inline uint64_t TableMapHash( const void * bytes, int32_t length ) +{ + uint64_t hash = 0xCBF29CE484222325ull; + const uint8_t * at = (const uint8_t *) bytes; + for ( int32_t i = 0; i < length; i++ ) { hash ^= (uint64_t) at[i]; hash *= 0x100000001B3ull; } + return hash; +} +inline uint64_t TableMapHash( uint64_t key ) { return TableMapHash( (const void *) &key, (int32_t) sizeof( key ) ); } + +// this runtime's own load factor, and no port reproduces it either: the next +// power of two at or above twice the count, so a probe run stays short +inline int32_t TableMapIndexSlots( int32_t count ) +{ + int32_t slots = 8; + while ( slots < count * 2 ) { slots *= 2; } + return slots; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_MAP + +#ifndef LISTDEMO_SCHEMA_TABLE_LIST +#define LISTDEMO_SCHEMA_TABLE_LIST + +namespace listdemo { + +// ---- an UNBOUNDED ARRAY: a counted array whose count the data decides (§2.9) ---- +// +// On the wire, in a region and in a cook a list is the kind 14 body a [..N]T +// writes, its elements by-value records inside the holder's node extent. What +// this adds is the slot, a builder that appends into segments that never +// move, and a const surface that indexes and iterates in place. There is no +// sort, no key and no lookup: the order is INSERTION order, and it is +// identity the way position is identity in a fixed array. + +// elements carved from ONE call to the allocator pair. A new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableListSegmentElements = 32; + +// THE ELEMENT STORAGE: T itself, and a TableRef slot for a []*T, whose +// elements are references exactly as a pointer field's slot is (§2.1) +template struct TableListStorage { typedef T Element; }; +template struct TableListStorage { typedef TableRef Element; }; + +// WHAT THE CONST FORM ANSWERS: the element by reference, and on a []*T the +// RESOLVED pointer, one add on the self-relative delta, NULL for a null slot, +// exactly as At answers it (§6.2, §6.3) +template struct TableListConst +{ + typedef const T & Result; + static Result At( const T * element ) { return *element; } +}; +template struct TableListConst +{ + typedef const T * Result; + static Result At( const TableRef * element ) + { + return element->value != 0 ? (const T *) ( (const uint8_t *) element + element->value ) : NULL; + } +}; + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.9, §7.2) ---- +// +// An int64 self-relative reference to the element array and an int32 count, +// then padding to eight. The reference is a TableRef like a pointer's: in the +// arena it names the builder's HEAD, in a region it is the delta from the slot +// to the first element, and 0 is the empty list in both. It is the map's slot +// exactly, because it is the same two facts. +template struct TableList +{ + typedef typename TableListStorage::Element Element; + + TableRef elements; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Element * Elements() const + { + return elements.value != 0 ? (const Element *) ( (const uint8_t *) &elements + elements.value ) : NULL; + } + int32_t size() const { return count; } + + // INDEXING IS BOUNDS-CHECKED IN EVERY BUILD (§2.4, §2.9): the extent is a + // number that CAME FROM A FILE, so an index past it is not a mistake a + // release build gets to make cheaply. There is no undefined-behavior path + // here in any configuration. The assert carries the message where a + // debugger can read it and NDEBUG removes that. The fatal is what stands + // after it. Both go through the hooks: define schema_assert and + // schema_fatal and this refusal lands in your own handler. + void RefuseIndex( int32_t index ) const + { + if ( (uint32_t) index >= (uint32_t) count ) + { + schema_assert( false && "an unbounded array is indexed inside its count, which came from a file" ); + schema_fatal(); + } + } + typename TableListConst::Result operator[]( int32_t index ) const + { + RefuseIndex( index ); + return TableListConst::At( Elements() + index ); + } + + // ---- iteration: INDEX order, the element and no key ---- + // + // It carries no iterator_traits, for the reason TableKeyed's does not + // (§13.9). + struct ConstIterator + { + const Element * at; + typename TableListConst::Result operator*() const { return TableListConst::At( at ); } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Elements() }; } + ConstIterator end() const { return ConstIterator{ Elements() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.9, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first element is added. Each +// segment is a fixed number of elements carved from one call to the allocator +// pair. An element's address is stable for the arena's life, so a T * handed +// back by Add stays valid while other elements arrive. +struct TableListHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an Add appends into + int32_t live; + int32_t dead; +}; + +template struct TableListSegment +{ + TableRef next; + int32_t used; // elements carved from this segment + int32_t padding; + uint32_t dead[ ( kTableListSegmentElements + 31 ) / 32 ]; // Erase marks one bit, never the element + Element elements[ kTableListSegmentElements ]; +}; + +inline bool TableListSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// the head, allocated when the first element is added +template +inline TableListHead * TableListReach( TableWorker & worker, TableList & list ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( list.elements.value != 0 ) { return (TableListHead *) TableArenaAt( *worker.arena, (uint32_t) list.elements.value ); } + uint32_t at = 0; + TableListHead * head = (TableListHead *) worker.AllocRaw( (int64_t) sizeof( TableListHead ), (int64_t) alignof( TableListHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + list.elements.value = (int64_t) at; + return head; +} + +// one element's storage, appended: the current segment when it has room, a +// new one carved from one call to the pair when it does not. NULL means NOT +// ADDED: an arena that cannot carve another segment, or a count at the int32 +// cap (§2.2, §2.9). +template +inline typename TableList::Element * TableListAppend( TableWorker & worker, TableListHead * head, TableList & list ) +{ + typedef typename TableList::Element Element; + if ( list.count >= INT32_MAX ) { return NULL; } // the int32 storage cap + TableListSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableListSegmentElements ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableListSegment *) worker.AllocRaw( (int64_t) sizeof( TableListSegment ), (int64_t) alignof( TableListSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableListSegment * previous = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Element * element = segment->elements + segment->used; + segment->used++; + head->live++; + list.count++; + return element; +} + +// ADD, whole: the head, the append, and the element at its declared defaults +// (§2.9). The text form's placement is this same call, because a list has no +// key to place under (§16). +template +inline typename TableList::Element * TableListPlace( TableWorker & worker, TableList & list ) +{ + typedef typename TableList::Element Element; + TableListHead * head = TableListReach( worker, list ); + if ( head == NULL ) { return NULL; } + Element * element = TableListAppend( worker, head, list ); + if ( element == NULL ) { return NULL; } + new ( element ) Element(); // value-init: the declared defaults, and null for a slot + return element; +} + +// ERASE, ADDRESSED BY THE POINTER (§2.9): the element Add handed back is the +// handle, because a list has no key and the address is the one thing the +// builder promises never moves (§6.4). It marks the element DEAD, one bit in +// the segment's slot and not in the element storage, and decrements the live +// count. False when the pointer is not this list's. Its storage is reclaimed +// at RESET and never reused mid-build, the map's rule for the map's reason. +template +inline bool TableListErase( TableArena & arena, TableList & list, const typename TableList::Element * element ) +{ + typedef typename TableList::Element Element; + if ( list.elements.value == 0 || element == NULL ) { return false; } + TableListHead * head = (TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableListSegment * segment = (TableListSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + if ( element >= segment->elements && element < segment->elements + segment->used ) + { + const int32_t i = (int32_t) ( element - segment->elements ); + if ( TableListSegmentDead( segment->dead, i ) ) { return false; } // already erased + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + list.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INDEX order, live elements only (§2.9) ---- +template struct TableListEach +{ + typedef typename TableList::Element Element; + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableListSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableListSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + Element * operator*() const { return segment->elements + index; } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableListEach TableListEachOf( const TableArena & arena, const TableList & list ) +{ + TableListEach each = { &arena, TableRef() }; + if ( list.elements.value != 0 ) + { + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + each.first = head->first; + } + return each; +} + +// ---- the INDEX-ORDER CURSOR the four writing walks read (§2.9) ---- +// +// Measure, Save, Lock and Cook each visit a list's live elements in the order +// they were added, and they allocate nothing to do it: a region's cursor is +// the array in place, and the builder's walks the segment chain. Indexing the +// builder's form is SEQUENTIAL by construction, every walk steps i, i + 1, +// i + 2, so the cursor remembers where the last access landed and moves one +// live slot per step. An access behind the memo restarts from the first +// segment, which no walk here does. +template struct TableListCursor +{ + const Element * elements = NULL; // the region's form: the array in place + const TableArena * arena = NULL; // the builder's form: the segments + TableRef first; + int32_t count = 0; + bool ok = false; + // the memo: the segment and slot the last access landed on, and the live + // index that slot holds + mutable const TableListSegment * segment = NULL; + mutable int32_t within = -1; + mutable int32_t logical = -1; + + const Element * At( int32_t index ) const + { + if ( elements != NULL ) { return elements + index; } + if ( segment == NULL || index < logical ) + { + segment = first.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL; + within = -1; + logical = -1; + } + while ( logical < index ) + { + for ( ;; ) + { + within++; + while ( segment != NULL && within >= segment->used ) + { + segment = segment->next.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + within = 0; + } + if ( segment == NULL ) { return NULL; } // the slot and the head disagree + if ( !TableListSegmentDead( segment->dead, within ) ) { break; } + } + logical++; + } + return segment->elements + within; + } + const Element & operator[]( int32_t index ) const { return *At( index ); } +}; + +// the REGION form: the array is the cursor +template +inline TableListCursor::Element> TableListElements( const TableRegionCtx &, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.elements = list.Elements(); + cursor.count = list.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: the live elements out of the segment chain, in the +// order they were added. A dead element costs nothing on any wire (§2.9). +template +inline TableListCursor::Element> TableListElements( const TableArena & arena, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.arena = &arena; + cursor.count = list.count; + if ( list.elements.value == 0 || list.count <= 0 ) { cursor.ok = list.count == 0; cursor.count = 0; return cursor; } + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + if ( head->live != list.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + cursor.first = head->first; + cursor.ok = true; + return cursor; +} + +template +inline TableListCursor::Element> TableListElements( const TableArenaCtx & ctx, const TableList & list ) +{ + return TableListElements( *ctx.arena, list ); +} + +// ---- the LOAD side: where a decoded element lands (§2.9) ---- +// +// The same two shapes the map's fill takes, because the decoder above them +// cannot tell which it has: a REGION carves the element array out of the +// holder node's own extent, PRE-ORDER, and the TOOL's path appends into the +// builder's arena. +template struct TableListFill +{ + typedef typename TableList::Element Element; + TableList * list = NULL; + Element * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; + bool refused = false; // a count above the int32 cap on the tool's path: LoadBuilder answers NULL +}; + +template +inline TableListFill TableListFillBegin( const TableNodeMap & nodes, TableList & list, uint64_t n ) +{ + typedef typename TableList::Element Element; + TableListFill fill; + fill.list = &list; + list.elements.value = 0; + list.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( n > (uint64_t) INT32_MAX ) + { + // A COUNT ABOVE THE int32 STORAGE CAP (§2.2, §2.9): into a region it was + // refused by LoadMeasure before this ran, and into a builder it is the + // refusal LoadBuilder answers NULL for, moving no counter + fill.refused = nodes.carve->worker != NULL; + return fill; + } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Element ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Element ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Element *) base; + fill.capacity = (int32_t) n; + list.elements.value = (int64_t) ( base - (const uint8_t *) &list.elements ); + fill.ok = true; + return fill; +} + +// the next slot, at the element's declared defaults. NULL when the arena +// could not carve, which the decoder reports as framing damage +template inline typename TableList::Element * TableListFillNext( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count >= fill.capacity ) { return NULL; } + Element * element = fill.array + fill.list->count; + new ( element ) Element(); + fill.list->count++; + return element; + } + return TableListPlace( *fill.worker, *fill.list ); +} + +// A SLOT WHOSE ELEMENT NEVER LANDED is given back (§2.9, §4): the array keeps +// what it decoded, and an element whose own framing gave out before one byte +// of it decoded was not decoded. The region's form uncounts it, and the builder's +// marks it dead, which is what the storage rule allows mid-build. +template inline void TableListFillDrop( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count > 0 ) { fill.list->count--; } + return; + } + if ( fill.list->elements.value == 0 ) { return; } + TableListHead * head = (TableListHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.list->elements.value ); + if ( head->last.value == 0 ) { return; } + TableListSegment * segment = (TableListSegment *) TableArenaAt( *fill.worker->arena, (uint32_t) head->last.value ); + if ( segment->used <= 0 ) { return; } + const int32_t i = segment->used - 1; + if ( TableListSegmentDead( segment->dead, i ) ) { return; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + fill.list->count--; +} + +// an EMPTY list's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableListFillEnd( TableListFill & fill ) +{ + if ( fill.array != NULL && fill.list->count == 0 ) { fill.list->elements.value = 0; } +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.9, §6.5) ---- +// +// N x sizeof( T ) rounded to alignof( T ), AT EVERY DEPTH. N is framing and +// not a value, so this reads no field: it walks the list's own header and, +// where a table element holds a list or a map of its own, the elements' +// headers under it. Every -1 carries its REASON (§6.5): the int32 cap first, +// because a count past it cannot fit any body, and then the body's own L. +inline bool TableListWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t elem_size, int64_t elem_align, uint8_t elem_kind, int64_t elem_floor, + TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != elem_kind ) { return true; } // another element kind: §4's ordinary kind mismatch, the field reads empty + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; } // an N the list's L cannot carry + at = ( at + elem_align - 1 ) & ~( elem_align - 1 ); + at += (int64_t) n * elem_size; + if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_LIST + +#ifndef LISTDEMO_SCHEMA_BUILD_VERSION +#define LISTDEMO_SCHEMA_BUILD_VERSION + +namespace listdemo { + +// THE BUILD VERSION (docs/SPEC-TABLES.md §20): one digest over every fact the bytes +// this build produces depend on — the type wire's protocol id, every record's +// layout as the compiler's own C ABI model computes it, and the facts that +// decide what a load PUTS in those slots. It is the number a cook's header +// carries and the number Open compares, and the number a block's prologue +// carries and BlockOpen compares: a build version answers "which build?" and +// not "which form?", and what separates the two forms is their MAGIC. +// +// There are TWO ids in the design and they are not interchangeable: the +// PROTOCOL ID is the type wire's and nothing else, and the BUILD VERSION is +// what everything cooked or blocked is keyed by. A table edit moves this and +// never the protocol id; a type edit moves both. +static const uint64_t BuildVersion = 0x8d7c0edaca4571c7ull; + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_BUILD_VERSION + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK +#define LISTDEMO_SCHEMA_TABLE_COOK + +namespace listdemo { + +// ---- the cooked form (docs/SPEC-TABLES.md §7) ---- +// +// A cooked file is a HEADER, a DATA part and an ATTRIBUTION part, in that +// order. Every word of the header is a u64 written in the byte order the cook +// was produced in, and the header is 64 bytes: +// +// 0 magic 0x4b4f4f434d484353, read BYTEWISE before anything else +// 8 build_version the unit's id (docs/SPEC-TABLES.md §20) +// 16 byte_order 1 little, 2 big — the order that WROTE the file +// 24 data_length the region's bytes, rounded up to alignment +// 32 attribution_length the directory's bytes, or 0 +// 40 alignment the region's alignment, never below eight +// 48 reserved zero +// 56 reserved zero +// +// The DATA part is Lock's region written verbatim (§7.2) — the root at its +// base — and it is what a runtime points at. The ATTRIBUTION part is the node +// directory (§6.3), and NOTHING THAT READS THE STRUCTURE TOUCHES IT: it is +// written beside the data for schema cook-check, so a build that ships no +// tooling need not carry it at all. +static const int64_t kTableCookHeaderBytes = 64; + +// THE MAGIC'S VALUE, and a consumer written from the page needs the constant +// rather than a description of one. It is "SCHMCOOK" read as ASCII in the byte +// order a little-endian store produces — the same shape the block form's +// SCHMABLK takes, so a hex dump of a little-endian cook is legible and the two +// accelerators sit in one vocabulary. +// +// IT IS STORED IN THE PRODUCER'S ORDER, which is what makes it the byte-order +// check as well as the form check: a consumer reads back this build's +// constant, or that constant byte-reversed — which identifies a cook of the +// OTHER order — or something that is not a cook. All three answers but the +// first refuse, and a cook and a BLOCK are separated here too, because a +// form's identity belongs in its magic rather than in a second digest. +static const uint64_t TableCookMagic = 0x4b4f4f434d484353ull; + +// THIS BUILD's byte order, as the header's own word carries it. The magic is +// what REFUSES a foreign order; this word is what RECORDS which order wrote +// the file, so a refusal names the order rather than inferring it and a tool +// dumping a cook reads the fact. A file whose magic matched and whose order +// word did not is corrupt, and there is no reading that recovers it. +// +// The BUILD VERSION cannot do either job: §20.1 digests byteorder as a +// GENERATION input, little for every target schema generates for today, so +// two builds of one schema for two orders emit the same id. +#if defined( __BYTE_ORDER__ ) && defined( __ORDER_BIG_ENDIAN__ ) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__ +static const uint64_t TableCookByteOrder = 2; // big +#else +static const uint64_t TableCookByteOrder = 1; // little +#endif + +// The greatest region alignment a cooked file may name. The DATA part begins +// at align_up( 64, alignment ), which is 64 for every unit this language can +// declare — the largest alignment it has is sixteen — so a word past this cap +// describes a file no build of this schema wrote (docs/SPEC-TABLES.md §7.1). +static const uint64_t TableCookMaxAlign = 64; + +// The header read, BYTEWISE. memcpy is the portable spelling of "these eight +// bytes, in this machine's order"; every compiler this repo builds under folds +// it to one load, and it is the only read in the whole of Open that is not a +// comparison. +inline uint64_t table_cook_read64( const uint8_t * p ) +{ + uint64_t v; + memcpy( &v, p, sizeof( v ) ); + return v; +} + +// TableCookOpen: THE WHOLE CHECK, in one place, because §7 states the +// enumeration once and every generated Open is that one enumeration plus +// its own root's two layout facts. +// +// THE CHECK, in order: the magic read bytewise, the byte order it establishes, +// the build version against this build's own, both RESERVED words zero, the +// region alignment the header names, the two part lengths against the length +// the caller passed — a truncated file and a file with trailing bytes are the +// same refusal — the root's own storage inside the data part, and the +// alignment of the base. +// +// AND THAT IS ALL OF IT. On a match the bytes ARE what this build wrote, in +// this build's layout and this build's byte order, so there is nothing to +// validate and nothing to fix up: the caller gets the root. Nothing per node +// happens here, which is what makes open O(1) in the file's size; a walk of +// any shape would forfeit that, and validating an untrusted file is schema +// cook-check's job and a person's decision (§7.4). +// +// EVERY NUMBER BELOW COMES OUT OF THE FILE, so the arithmetic is unsigned and +// each term is BOUNDED BEFORE IT IS ADDED: a forged length near 2^64 must +// refuse, and an addition that wrapped would be the defect the comparison +// after it was supposed to catch. Nothing past length is read on any path, +// including every refusing one. +inline const uint8_t * TableCookOpen( const void * bytes, uint64_t length, uint64_t root_size, uint64_t root_align ) +{ + if ( bytes == NULL ) { return NULL; } + if ( length < (uint64_t) kTableCookHeaderBytes ) { return NULL; } + const uint8_t * raw = (const uint8_t *) bytes; + // the MAGIC, bytewise and first: it is what establishes the byte order + // every other header word is read in, so nothing else may be read before + // it. A byte-reversed constant is a cook of the other order and refuses + // here, which is why the order never reaches a fix-up pass. + if ( table_cook_read64( raw ) != TableCookMagic ) { return NULL; } + if ( table_cook_read64( raw + 16 ) != TableCookByteOrder ) { return NULL; } + if ( table_cook_read64( raw + 8 ) != BuildVersion ) { return NULL; } + // the RESERVED words: a non-zero one means a writer used a form this build + // does not understand, and Open refuses rather than ignoring it. + if ( table_cook_read64( raw + 48 ) != 0 ) { return NULL; } + if ( table_cook_read64( raw + 56 ) != 0 ) { return NULL; } + const uint64_t data_length = table_cook_read64( raw + 24 ); + const uint64_t attribution_length = table_cook_read64( raw + 32 ); + const uint64_t alignment = table_cook_read64( raw + 40 ); + // THE ALIGNMENT WORD IS DATA, and it is the one header field the rest of + // the check does arithmetic WITH rather than only comparison against. A + // region's alignment is a power of two, never below eight (the floor that + // puts the attribution part on an eight-byte boundary without a second + // padding rule) and never past the cap above; a word that is none of those + // rounds nothing and aligns nothing, so it is refused before it is used. + if ( alignment < 8 || alignment > TableCookMaxAlign ) { return NULL; } + if ( ( alignment & ( alignment - 1 ) ) != 0 ) { return NULL; } + // and it must be an alignment THE ROOT CAN SIT AT, since the root is at + // the region's base: both are powers of two, so "at least the root's" + // is one division. + if ( ( alignment % root_align ) != 0 ) { return NULL; } + // The DATA part begins at align_up( 64, alignment ). It is DERIVED and not + // a header field, because a fact a reader computes is a fact two writers + // cannot disagree about. + const uint64_t data_offset = ( (uint64_t) kTableCookHeaderBytes + alignment - 1 ) & ~( alignment - 1 ); + if ( length < data_offset ) { return NULL; } + // the two part lengths against the length the caller passed. The whole + // file is data_offset + data_length + attribution_length, and a length + // that is not EXACTLY that refuses — truncation and trailing bytes are one + // refusal, and both terms are subtracted rather than added so no sum can + // carry. + if ( data_length > length - data_offset ) { return NULL; } + if ( attribution_length != length - data_offset - data_length ) { return NULL; } + // the ROOT sits at the region's base, so the region has to hold it: a + // shorter data part describes a root partly outside the file, which is the + // one way a match-and-point reader could hand back storage it never + // received. + if ( data_length < root_size ) { return NULL; } + const uint8_t * base = raw + data_offset; + // the alignment of the BASE. The header pads the data part to the region's + // alignment, so a base an allocator or mmap gave you is already aligned — + // mmap gives page alignment for free — and a base that is not is a caller's + // buffer this form cannot be read out of. + if ( ( (uintptr_t) base % (uintptr_t) alignment ) != 0 ) { return NULL; } + return base; +} + +// ---- the cooked form, the WRITE side (docs/SPEC-TABLES.md §7.6) ---- +// +// THE BYTE ORDER IS THE TARGET'S, NOT THE HOST'S. A cook is produced in the +// byte order of the build that will read it (§7), so the fixing happens here — +// offline, once, on the writing side — and never at Open. Passing +// TableByteOrder::Big on a little-endian machine produces a big-endian build's +// file, and nothing about the writing host reaches the bytes. +enum class TableByteOrder +{ + Little = 1, // the header's byte_order word, and the order every scalar is written in + Big = 2, +}; + +// One store, width as an argument. Every call site passes a literal width, so +// the loop folds to a store (and a byte swap on the foreign order); a name per +// width would claim four §11 names to save nothing. +inline void table_cook_put( uint8_t * at, uint64_t value, int32_t width, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * i ) ); } + } + else + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * ( width - 1 - i ) ) ); } + } +} + +// A 128-bit store as two lanes: sixteen bytes, the low lane first in the +// little order and the high lane first — each lane big-endian — in the big +// order, exactly as a u64 is one lane of eight (docs/SPEC-TABLES.md §7.2). +inline void table_cook_put128( uint8_t * at, uint64_t lo, uint64_t hi, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) { table_cook_put( at, lo, 8, order ); table_cook_put( at + 8, hi, 8, order ); } + else { table_cook_put( at, hi, 8, order ); table_cook_put( at + 8, lo, 8, order ); } +} + +// A buffer piece: the USED bytes and nothing else. The tail is already zero — +// the whole extent was zeroed before any field was written — so this copies the +// used prefix and leaves the rest, which is what makes a string's unused tail a +// consequence of one memset rather than a rule per buffer. A used length past +// the buffer, or below zero, is a value no reader could have produced and it is +// clamped rather than trusted: this writes inside the caller's buffer on every +// input. +inline void table_cook_bytes( uint8_t * at, const void * source, int64_t used, int64_t capacity ) +{ + if ( used <= 0 ) { return; } + const int64_t n = used < capacity ? used : capacity; + memcpy( at, source, (size_t) n ); +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE +#define LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// ---- the cooked form's WRITE side for a POINTERED root (docs/SPEC-TABLES.md §7.6) ---- +// +// A pointered root's cook is the region of §7.2: every node the numbering +// reached (§3.1), once, at its own type's alignment, in index order, the root +// at offset zero. This is that region while it is being laid out and written — +// the tool's own Layout and Write, in one struct. +// +// The OFFSETS are one per node, the root's zero at position 0 and node index k +// at position k - 1, which is the directory's own order (§6.3); they are the +// one allocation the write makes beyond the numbering, and they go through the +// same pair. A measure needs no offsets and leaves the pointer NULL. +struct TableCookRegion +{ + const TableNumbering * numbering = NULL; // node -> index, from the walk that placed it + int64_t * offsets = NULL; // index - 1 -> the node's region offset; NULL while measuring + int64_t count = 0; // nodes, the root included + int64_t bytes = 0; // the data part's length, rounded to align + int64_t align = 0; // the region's alignment: the nodes' greatest, never below eight + uint8_t * base = NULL; // where the data part is being written; NULL while measuring +}; + +// A reference slot: the SELF-RELATIVE delta from the slot's own address to the +// node's start (§6.3), and zero for null. The node is found by the address the +// numbering keyed it under, which is the same address the walk resolved through +// the same context — so a reference the numbering does not carry is a slot the +// walk never reached (a counted array's slot past its count, an absent +// optional's value) holding a node the region will not hold, and it is refused +// rather than written as a delta to nowhere. +inline bool table_cook_ref( const TableCookRegion & region, uint8_t * at, const void * pointee, TableByteOrder order ) +{ + if ( pointee == NULL ) { table_cook_put( at, 0, 8, order ); return true; } + uint64_t index = 0; + if ( !TableNumberingIndex( *region.numbering, pointee, index ) ) { return false; } + if ( index == 0 || index > (uint64_t) region.count ) { return false; } + const int64_t delta = region.offsets[index - 1] - (int64_t) ( at - region.base ); + table_cook_put( at, (uint64_t) delta, 8, order ); + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// table Sample — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Sample { + int32_t v = 0; +}; + +// table Row — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Row { + TableList items; // Sample: the element array, empty until an Add + int32_t label = 0; +}; + +// table Sheet — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Sheet { + TableList rows; // Row: the element array, empty until an Add + TableRef pinned; // *Row — null until assigned +}; + +// table Item — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Item { + int32_t count = 0; +}; + +// table SquadRosterEntry — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct SquadRosterEntry { + uint8_t key = 0; + Item value; +}; + +// table Squad — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Squad { + TableMap roster; // map[uint8]Item — the sorted entry array, empty until an insert + int32_t name = 0; +}; + +// table Army — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Army { + TableList squads; // Squad: the element array, empty until an Add + int32_t after = 0; +}; + +// table Deck — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Deck { + Row hands[3]; // used count beside it; count in [0, 3] + int32_t hands_count = 0; + int32_t after = 0; +}; + +// ---- prefill: the declared defaults, in place (docs/SPEC-TABLES.md) ---- + +inline void SampleReset( Sample & value ); +inline void RowReset( Row & value ); +inline void SheetReset( Sheet & value ); +inline void ItemReset( Item & value ); +inline void SquadRosterEntryReset( SquadRosterEntry & value ); +inline void SquadReset( Squad & value ); +inline void ArmyReset( Army & value ); +inline void DeckReset( Deck & value ); + +inline void SampleReset( Sample & value ) +{ + value.v = 0; +} + +inline void RowReset( Row & value ) +{ + value.items.elements.value = 0; // Sample: empty + value.items.count = 0; + value.items.padding = 0; + value.label = 0; +} + +inline void SheetReset( Sheet & value ) +{ + value.rows.elements.value = 0; // Row: empty + value.rows.count = 0; + value.rows.padding = 0; + value.pinned.value = 0; // *Row — null +} + +inline void ItemReset( Item & value ) +{ + value.count = 0; +} + +inline void SquadRosterEntryReset( SquadRosterEntry & value ) +{ + value.key = 0; + ItemReset( value.value ); +} + +inline void SquadReset( Squad & value ) +{ + value.roster.entries.value = 0; // map[uint8]Item: empty + value.roster.count = 0; + value.roster.padding = 0; + value.name = 0; +} + +inline void ArmyReset( Army & value ) +{ + value.squads.elements.value = 0; // Squad: empty + value.squads.count = 0; + value.squads.padding = 0; + value.after = 0; +} + +inline void DeckReset( Deck & value ) +{ + RowReset( value.hands[0] ); + for ( int32_t i = 1; i < 3; i++ ) { value.hands[i] = value.hands[0]; } + value.hands_count = 0; + value.after = 0; +} + +// ---- the arena's reset hook (docs/SPEC-TABLES.md §6) ---- +// +// TableWorker::Alloc is a template and cannot name a member's Reset, so +// the arena reaches it through this overload set by argument-dependent +// lookup. It is how a node born in raw arena storage comes to hold the +// declared defaults without value-initialising the whole aggregate. + +inline void TableReset( Sample & value ) { SampleReset( value ); } +inline void TableReset( Row & value ) { RowReset( value ); } +inline void TableReset( Sheet & value ) { SheetReset( value ); } +inline void TableReset( Item & value ) { ItemReset( value ); } +inline void TableReset( SquadRosterEntry & value ) { SquadRosterEntryReset( value ); } +inline void TableReset( Squad & value ) { SquadReset( value ); } +inline void TableReset( Army & value ) { ArmyReset( value ); } +inline void TableReset( Deck & value ) { DeckReset( value ); } + +// ---- pointer targets: allocation and resolution (docs/SPEC-TABLES.md §2) ---- +// +// A reference resolves differently in the two forms, and the CONTEXT says +// which: in the arena it is an offset; in a region it is a self-relative +// delta, so the const deref below is one add and needs no base pointer. + +// Row is a pointer target. +inline const Row * RowAt( const TableRef & ref ) // the const form's hot path: one add, no base +{ + return ref.value != 0 ? (const Row *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline Row * RowAt( TableRef & ref ) +{ + return ref.value != 0 ? (Row *) ( (uint8_t *) &ref + ref.value ) : NULL; +} +inline const Row * RowAt( const TableRegionCtx &, const TableRef & ref ) { return RowAt( ref ); } +inline const Row * RowAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const Row *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +// while the builder is mutable, resolve against the arena itself +inline Row * RowAt( TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (Row *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} +inline const Row * RowAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const Row *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} +// allocate one Row in the arena; the slot holds the arena offset +inline Row * RowEmplace( TableWorker & worker, TableRef & slot ) +{ + TableSlot allocated = worker.Alloc(); + slot = allocated.ref; + return allocated.ptr; +} + +// ---- codecs: measure/save/load per closure member ---- + +inline int64_t SampleMeasureBody( TableIds & ids, const Sample & value ); +LISTDEMO_TABLE_INLINE bool SampleSaveBody( TableWriter & w, TableIds & ids, const Sample & value ); +LISTDEMO_TABLE_INLINE bool SampleLoadBody( TableReader & r, Sample & value ); +template inline int64_t RowMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Row & value ); +template inline bool RowSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Row & value ); +template inline bool RowSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Row & value ); +inline bool RowLoadBody( TableReader & r, const TableNodeMap & nodes, Row & value ); +template inline int64_t SheetMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Sheet & value ); +template inline bool SheetSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Sheet & value ); +template inline bool SheetSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Sheet & value ); +inline bool SheetLoadBody( TableReader & r, const TableNodeMap & nodes, Sheet & value ); +inline int64_t ItemMeasureBody( TableIds & ids, const Item & value ); +LISTDEMO_TABLE_INLINE bool ItemSaveBody( TableWriter & w, TableIds & ids, const Item & value ); +LISTDEMO_TABLE_INLINE bool ItemLoadBody( TableReader & r, Item & value ); +inline int64_t SquadRosterEntryMeasureBody( TableIds & ids, const SquadRosterEntry & value ); +LISTDEMO_TABLE_INLINE bool SquadRosterEntrySaveBody( TableWriter & w, TableIds & ids, const SquadRosterEntry & value ); +LISTDEMO_TABLE_INLINE bool SquadRosterEntryLoadBody( TableReader & r, SquadRosterEntry & value ); +template inline int64_t SquadMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Squad & value ); +template inline bool SquadSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Squad & value ); +template inline bool SquadSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Squad & value ); +inline bool SquadLoadBody( TableReader & r, const TableNodeMap & nodes, Squad & value ); +template inline int64_t ArmyMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Army & value ); +template inline bool ArmySaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Army & value ); +template inline bool ArmySaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Army & value ); +inline bool ArmyLoadBody( TableReader & r, const TableNodeMap & nodes, Army & value ); +template inline int64_t DeckMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Deck & value ); +template inline bool DeckSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Deck & value ); +template inline bool DeckSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Deck & value ); +inline bool DeckLoadBody( TableReader & r, const TableNodeMap & nodes, Deck & value ); + +// ---- pointer-graph walkers: number (measure/save), pack (Lock) ---- + +template inline bool RowNumber( const Ctx & ctx, TableNumbering & numbering, const Row & value ); +template inline int64_t RowPackMeasure( const Ctx & ctx, TablePackMap & seen, const Row & value ); +template inline bool RowPack( const Ctx & ctx, TablePackMap & seen, const Row & src, Row & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool SheetNumber( const Ctx & ctx, TableNumbering & numbering, const Sheet & value ); +template inline int64_t SheetPackMeasure( const Ctx & ctx, TablePackMap & seen, const Sheet & value ); +template inline bool SheetPack( const Ctx & ctx, TablePackMap & seen, const Sheet & src, Sheet & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool SquadNumber( const Ctx & ctx, TableNumbering & numbering, const Squad & value ); +template inline int64_t SquadPackMeasure( const Ctx & ctx, TablePackMap & seen, const Squad & value ); +template inline bool SquadPack( const Ctx & ctx, TablePackMap & seen, const Squad & src, Squad & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool ArmyNumber( const Ctx & ctx, TableNumbering & numbering, const Army & value ); +template inline int64_t ArmyPackMeasure( const Ctx & ctx, TablePackMap & seen, const Army & value ); +template inline bool ArmyPack( const Ctx & ctx, TablePackMap & seen, const Army & src, Army & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool DeckNumber( const Ctx & ctx, TableNumbering & numbering, const Deck & value ); +template inline int64_t DeckPackMeasure( const Ctx & ctx, TablePackMap & seen, const Deck & value ); +template inline bool DeckPack( const Ctx & ctx, TablePackMap & seen, const Deck & src, Deck & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +// ---- the numbering's bridge to each member's codec (docs/SPEC-TABLES.md §3.1) ---- + +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Row & value ) { return RowMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Row & value ) { return RowSaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Sheet & value ) { return SheetMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Sheet & value ) { return SheetSaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Squad & value ) { return SquadMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Squad & value ) { return SquadSaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Army & value ) { return ArmyMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Army & value ) { return ArmySaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Deck & value ) { return DeckMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Deck & value ) { return DeckSaveBody( ctx, numbering, w, ids, value ); } + +// ---- SquadRosterEntry: the order, the key and the value (docs/SPEC-TABLES.md §2.8) ---- +// +// The four overloads the map runtime's templates reach by argument-dependent +// lookup. Nothing outside this file names them. +static_assert( alignof( SquadRosterEntry ) <= kTableAlign, "a map entry's alignment must fit the arena's" ); + +inline int TableEntryOrder( const SquadRosterEntry & a, const SquadRosterEntry & b ) +{ + return TableKeyOrder( (uint64_t) a.key, (uint64_t) b.key ); // integers compare by VALUE, unsigned here +} +inline int TableEntryOrder( const SquadRosterEntry & entry, uint8_t key ) +{ + return TableKeyOrder( (uint64_t) entry.key, (uint64_t) key ); +} +inline uint8_t TableEntryKey( const SquadRosterEntry & entry ) { return entry.key; } +inline void TableEntrySetKey( SquadRosterEntry & entry, uint8_t key ) { entry.key = key; } +inline const Item * TableEntryFound( const SquadRosterEntry * entry ) { return entry != NULL ? &entry->value : NULL; } +inline Item * TableEntryValue( SquadRosterEntry * entry ) { return &entry->value; } +struct SquadRosterEntryEach { uint8_t key; decltype( TableEntryValue( (SquadRosterEntry *) NULL ) ) value; }; +inline SquadRosterEntryEach TableEntryEach( SquadRosterEntry * entry ) { return SquadRosterEntryEach{ TableEntryKey( *entry ), TableEntryValue( entry ) }; } +inline void TableResetMapValue( SquadRosterEntry & value ) +{ + ItemReset( value.value ); +} + +// SquadRosterEntryReadKey: the key, before the slot is chosen (docs/SPEC-TABLES.md §2.8). +// Field order inside a body is not contractual (§3), so this scans rather +// than assuming a position — and this implementation writes the key first, +// so on any wire it wrote the scan ends at the first header. +struct SquadRosterEntryKeyRead +{ + uint8_t key; + bool found; // the body carried the key's id + bool kind_bad; // it carried it under another kind: the MAP's event + bool over; // longer than this reader's bound: the ENTRY is dropped + bool malformed; // the entry's framing gave out +}; + +inline SquadRosterEntryKeyRead SquadRosterEntryReadKey( const uint8_t * body, int64_t length, const TableIdTable * ids ) +{ + SquadRosterEntryKeyRead out = { 0, false, false, false, false }; + TableReport scratch; // the scan's own framing damage is the MAP's, raised by the caller + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { out.malformed = true; return out; } + if ( field_ref == 0 ) { return out; } // the terminator: no key field is the key's DEFAULT + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { out.malformed = true; return out; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { out.malformed = true; return out; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x3dc94a19365b10ecull ) // `key`, the ordinary hash of an ordinary name + { + out.kind_bad = field_kind != 6; // THE KEY KIND IS THE READER'S DECLARATION + out.found = !out.kind_bad; + if ( !out.kind_bad ) + { + if ( !r.has( 1 ) ) { out.malformed = true; return out; } + out.key = (uint8_t) r.get8(); + continue; // the LAST occurrence is the one §3 keeps + } + } + if ( !r.skip( field_kind ) ) { out.malformed = true; return out; } + } +} + +inline int64_t SampleMeasureBody( TableIds & ids, const Sample & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.v != 0 ) { bytes += TableLebBytes( ids.ref( 0xaf63eb4c86020609ull, 23 ) ) + 1 + 4; } // v + return bytes; +} + +inline int64_t SampleMeasure( const Sample & value ) +{ + TableIds ids; + const int64_t body = SampleMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool SampleSaveBody( TableWriter & w, TableIds & ids, const Sample & value ) +{ + if ( value.v != 0 ) + { + w.putleb( ids.ref( 0xaf63eb4c86020609ull, 23 ) ); w.put8( 4 ); // v + w.put32( uint32_t( value.v ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t SampleSave( const Sample & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !SampleSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == SampleMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool SampleLoadBody( TableReader & r, Sample & value ) +{ + SampleReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xaf63eb4c86020609ull: // v + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.v = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict SampleLoadVerdict( Sample & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + SampleReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + SampleReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !SampleLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool SampleLoad( Sample & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return SampleLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t SampleMeasureMessage( const Sample & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = SampleMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t SampleSaveMessage( const Sample & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !SampleSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == SampleMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool SampleLoadMessage( Sample & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + SampleReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return SampleLoadBody( r, value ); +} + +template +inline int64_t RowMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Row & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // items: a kind 14 array of kind 13 elements, INDEX order (§2.9) + TableListCursor cursor_items = TableListElements( ctx, value.items ); + if ( !cursor_items.ok ) { return -1; } // the slot and the head disagree + if ( cursor_items.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_items = ids.ref( 0x3e7884bf4f412c6full, 6 ); + int64_t body_items = 0; + body_items += 1 + TableLebBytes( (uint64_t) ( cursor_items.count ) ); // the element kind byte and the count + for ( int32_t elem_i_items = 0; elem_i_items < cursor_items.count; elem_i_items++ ) + { + const int64_t elem_bytes_items = SampleMeasureBody( ids, cursor_items[elem_i_items] ); + if ( elem_bytes_items < 0 ) { return -1; } + body_items += TableLebBytes( (uint64_t) ( elem_bytes_items ) ) + ( elem_bytes_items ); + } + bytes += TableLebBytes( ref_items ) + 1 + TableLebBytes( (uint64_t) ( body_items ) ) + ( body_items ); + } + } + if ( value.label != 0 ) { bytes += TableLebBytes( ids.ref( 0x39f7fcec8fcb623dull, 22 ) ) + 1 + 4; } // label + return bytes; +} + +template +inline bool RowSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Row & value ) +{ + (void) ctx; (void) numbering; + { + TableListCursor cursor_items = TableListElements( ctx, value.items ); // items + if ( !cursor_items.ok ) { return false; } + if ( cursor_items.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_items = ids.ref( 0x3e7884bf4f412c6full, 6 ); + int64_t body_items = 0; + body_items += 1 + TableLebBytes( (uint64_t) ( cursor_items.count ) ); // the element kind byte and the count + for ( int32_t elem_i_items = 0; elem_i_items < cursor_items.count; elem_i_items++ ) + { + const int64_t elem_bytes_items = SampleMeasureBody( ids, cursor_items[elem_i_items] ); + if ( elem_bytes_items < 0 ) { return false; } + body_items += TableLebBytes( (uint64_t) ( elem_bytes_items ) ) + ( elem_bytes_items ); + } + w.putleb( ref_items ); w.put8( 14 ); w.putleb( (uint64_t) body_items ); // items + w.put8( 13 ); w.putleb( (uint64_t) ( cursor_items.count ) ); + for ( int32_t elem_i_items = 0; elem_i_items < cursor_items.count; elem_i_items++ ) + { + { + const int64_t elem_len_items = SampleMeasureBody( ids, cursor_items[elem_i_items] ); + if ( elem_len_items < 0 ) return false; + w.putleb( (uint64_t) elem_len_items ); + if ( !SampleSaveBody( w, ids, cursor_items[elem_i_items] ) ) return false; + } + } + } + } + if ( value.label != 0 ) + { + w.putleb( ids.ref( 0x39f7fcec8fcb623dull, 22 ) ); w.put8( 4 ); // label + w.put32( uint32_t( value.label ) ); + } + return !w.overflow; +} + +template +inline bool RowSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Row & value ) +{ + if ( !RowSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool RowLoadBody( TableReader & r, const TableNodeMap & nodes, Row & value ) +{ + (void) nodes; + RowReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x3e7884bf4f412c6full: // items + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.items, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Sample * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + uint64_t elem_len_items = 0; + if ( !sub.getleb( elem_len_items ) || !sub.room( elem_len_items ) ) { r.report->malformed = true; break; } + { + TableReader elem_items( sub.buffer + sub.offset, (int64_t) elem_len_items, r.report, r.ids ); + SampleLoadBody( elem_items, ( *slot ) ); + } + sub.offset += (int64_t) elem_len_items; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x39f7fcec8fcb623dull: // label + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.label = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +template +inline int64_t SheetMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Sheet & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // rows: a kind 14 array of kind 13 elements, INDEX order (§2.9) + TableListCursor cursor_rows = TableListElements( ctx, value.rows ); + if ( !cursor_rows.ok ) { return -1; } // the slot and the head disagree + if ( cursor_rows.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_rows = ids.ref( 0xa3a7061ff10a8138ull, 27 ); + int64_t body_rows = 0; + body_rows += 1 + TableLebBytes( (uint64_t) ( cursor_rows.count ) ); // the element kind byte and the count + for ( int32_t elem_i_rows = 0; elem_i_rows < cursor_rows.count; elem_i_rows++ ) + { + const int64_t elem_bytes_rows = RowMeasureBody( ctx, numbering, ids, cursor_rows[elem_i_rows] ); + if ( elem_bytes_rows < 0 ) { return -1; } + body_rows += TableLebBytes( (uint64_t) ( elem_bytes_rows ) ) + ( elem_bytes_rows ); + } + bytes += TableLebBytes( ref_rows ) + 1 + TableLebBytes( (uint64_t) ( body_rows ) ) + ( body_rows ); + } + } + { + const Row * pointee_pinned = RowAt( ctx, value.pinned ); // *Row + // A POINTER RIDES AS A NODE INDEX (docs/SPEC-TABLES.md §3.1): the + // header and the index and nothing below it, because the pointee's + // body is in the node table and not here. NULL IS ELIDED — absence + // and null are one value — and a non-null pointer ALWAYS rides, even + // when its node's body is entirely default. + if ( pointee_pinned != NULL ) + { + uint64_t index_pinned = 0; + if ( !TableNumberingIndex( numbering, (const void *) pointee_pinned, index_pinned ) ) { return -1; } + bytes += TableLebBytes( ids.ref( 0x5f82477707ad620full, 28 ) ) + 1 + TableLebBytes( index_pinned ); + } + } + return bytes; +} + +template +inline bool SheetSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Sheet & value ) +{ + { + TableListCursor cursor_rows = TableListElements( ctx, value.rows ); // rows + if ( !cursor_rows.ok ) { return false; } + if ( cursor_rows.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_rows = ids.ref( 0xa3a7061ff10a8138ull, 27 ); + int64_t body_rows = 0; + body_rows += 1 + TableLebBytes( (uint64_t) ( cursor_rows.count ) ); // the element kind byte and the count + for ( int32_t elem_i_rows = 0; elem_i_rows < cursor_rows.count; elem_i_rows++ ) + { + const int64_t elem_bytes_rows = RowMeasureBody( ctx, numbering, ids, cursor_rows[elem_i_rows] ); + if ( elem_bytes_rows < 0 ) { return false; } + body_rows += TableLebBytes( (uint64_t) ( elem_bytes_rows ) ) + ( elem_bytes_rows ); + } + w.putleb( ref_rows ); w.put8( 14 ); w.putleb( (uint64_t) body_rows ); // rows + w.put8( 13 ); w.putleb( (uint64_t) ( cursor_rows.count ) ); + for ( int32_t elem_i_rows = 0; elem_i_rows < cursor_rows.count; elem_i_rows++ ) + { + { + const int64_t elem_len_rows = RowMeasureBody( ctx, numbering, ids, cursor_rows[elem_i_rows] ); + if ( elem_len_rows < 0 ) return false; + w.putleb( (uint64_t) elem_len_rows ); + if ( !RowSaveBody( ctx, numbering, w, ids, cursor_rows[elem_i_rows] ) ) return false; + } + } + } + } + { + const Row * pointee_pinned = RowAt( ctx, value.pinned ); // *Row + if ( pointee_pinned != NULL ) + { + uint64_t index_pinned = 0; + if ( !TableNumberingIndex( numbering, (const void *) pointee_pinned, index_pinned ) ) { return false; } + w.putleb( ids.ref( 0x5f82477707ad620full, 28 ) ); w.put8( 17 ); // pinned — a NODE INDEX into the flat node table + w.putleb( index_pinned ); + } + } + return !w.overflow; +} + +template +inline bool SheetSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Sheet & value ) +{ + if ( !SheetSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool SheetLoadBody( TableReader & r, const TableNodeMap & nodes, Sheet & value ) +{ + SheetReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xa3a7061ff10a8138ull: // rows + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.rows, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Row * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + uint64_t elem_len_rows = 0; + if ( !sub.getleb( elem_len_rows ) || !sub.room( elem_len_rows ) ) { r.report->malformed = true; break; } + { + TableReader elem_rows( sub.buffer + sub.offset, (int64_t) elem_len_rows, r.report, r.ids ); + RowLoadBody( elem_rows, nodes, ( *slot ) ); + } + sub.offset += (int64_t) elem_len_rows; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x5f82477707ad620full: // pinned + { + if ( kind != 17 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + // A POINTER FIELD'S PAYLOAD IS A NUMBER (docs/SPEC-TABLES.md §3.1): it is + // bounds-checked and resolved through the numbering, never FOLLOWED, so + // there is no traversal here and therefore no traversal bound. + { + uint64_t node_index = 0; + if ( !r.getleb( node_index ) ) { r.report->malformed = true; return false; } + TableNodeResolve( nodes, value.pinned, node_index, 0xa013e119fec906fbull, r.report ); // *Row + } + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +inline int64_t ItemMeasureBody( TableIds & ids, const Item & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.count != 0 ) { bytes += TableLebBytes( ids.ref( 0xb1e5e28e4479a274ull, 11 ) ) + 1 + 4; } // count + return bytes; +} + +inline int64_t ItemMeasure( const Item & value ) +{ + TableIds ids; + const int64_t body = ItemMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool ItemSaveBody( TableWriter & w, TableIds & ids, const Item & value ) +{ + if ( value.count != 0 ) + { + w.putleb( ids.ref( 0xb1e5e28e4479a274ull, 11 ) ); w.put8( 4 ); // count + w.put32( uint32_t( value.count ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t ItemSave( const Item & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !ItemSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == ItemMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool ItemLoadBody( TableReader & r, Item & value ) +{ + ItemReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xb1e5e28e4479a274ull: // count + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.count = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict ItemLoadVerdict( Item & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + ItemReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + ItemReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !ItemLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool ItemLoad( Item & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return ItemLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t ItemMeasureMessage( const Item & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = ItemMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t ItemSaveMessage( const Item & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !ItemSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == ItemMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool ItemLoadMessage( Item & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + ItemReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return ItemLoadBody( r, value ); +} + +inline int64_t SquadRosterEntryMeasureBody( TableIds & ids, const SquadRosterEntry & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.key != 0 ) { bytes += TableLebBytes( ids.ref( 0x3dc94a19365b10ecull, 31 ) ) + 1 + 1; } // key + { + const int32_t mark_value = ids.count; + const uint64_t ref_value = ids.ref( 0x7ce4fd9430e80ceaull, 32 ); + const int64_t body_value = ItemMeasureBody( ids, value.value ); + if ( body_value < 0 ) { return -1; } + if ( body_value > 1 ) { bytes += TableLebBytes( ref_value ) + 1 + TableLebBytes( (uint64_t) ( body_value ) ) + ( body_value ); } // value + else { ids.truncate( mark_value ); } // an all-default nested table elides, and costs no entry + } + return bytes; +} + +LISTDEMO_TABLE_INLINE bool SquadRosterEntrySaveBody( TableWriter & w, TableIds & ids, const SquadRosterEntry & value ) +{ + if ( value.key != 0 ) + { + w.putleb( ids.ref( 0x3dc94a19365b10ecull, 31 ) ); w.put8( 6 ); // key + w.put8( uint8_t( value.key ) ); + } + { + const int32_t mark_value = ids.count; + const uint64_t ref_value = ids.ref( 0x7ce4fd9430e80ceaull, 32 ); + const int64_t body_value = ItemMeasureBody( ids, value.value ); + if ( body_value < 0 ) return false; // storage invariant, refused as measure refuses it + if ( body_value > 1 ) // all-default nested elides + { + w.putleb( ref_value ); w.put8( 13 ); w.putleb( (uint64_t) body_value ); // value + if ( !ItemSaveBody( w, ids, value.value ) ) return false; + } + else { ids.truncate( mark_value ); } + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +LISTDEMO_TABLE_INLINE bool SquadRosterEntryLoadBody( TableReader & r, SquadRosterEntry & value ) +{ + SquadRosterEntryReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x3dc94a19365b10ecull: // key + { + if ( kind != 6 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t decoded_v = uint8_t( r.get8( ) ); + value.key = decoded_v; + break; + } + case 0x7ce4fd9430e80ceaull: // value + { + if ( kind != 13 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + { + TableReader sub( r.buffer + r.offset, (int64_t) body_len, r.report, r.ids ); + ItemLoadBody( sub, value.value ); + if ( sub.offset != sub.size ) + { + r.report->malformed = true; + ItemReset( value.value ); + } + } + r.offset += (int64_t) body_len; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +template +inline int64_t SquadMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Squad & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // roster: a kind 14 array of kind 13 elements, ASCENDING (§2.8) + TableMapCursor order_roster = TableMapOrder( ctx, value.roster ); + if ( !order_roster.ok ) { return -1; } // the sort could not run + if ( order_roster.count > 0 ) + { + const uint64_t ref_roster = ids.ref( 0x1c84390d304f4f42ull, 29 ); + int64_t body_roster = 1 + TableLebBytes( (uint64_t) order_roster.count ); // the element kind byte and the count + for ( int32_t i = 0; i < order_roster.count; i++ ) + { + const int64_t elem_roster = SquadRosterEntryMeasureBody( ids, *order_roster[i] ); + if ( elem_roster < 0 ) { TableMapRelease( order_roster ); return -1; } + body_roster += TableLebBytes( (uint64_t) ( elem_roster ) ) + ( elem_roster ); // BUT THE ENTRY ALWAYS RIDES: identity here is the key + } + bytes += TableLebBytes( ref_roster ) + 1 + TableLebBytes( (uint64_t) ( body_roster ) ) + ( body_roster ); + } + TableMapRelease( order_roster ); + } + if ( value.name != 0 ) { bytes += TableLebBytes( ids.ref( 0xc4bcadba8e631b86ull, 30 ) ) + 1 + 4; } // name + return bytes; +} + +template +inline bool SquadSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Squad & value ) +{ + (void) ctx; (void) numbering; + { + TableMapCursor order_roster = TableMapOrder( ctx, value.roster ); // roster + if ( !order_roster.ok ) { return false; } + if ( order_roster.count > 0 ) // an EMPTY map elides, the by-value rule (§3) + { + const uint64_t ref_roster = ids.ref( 0x1c84390d304f4f42ull, 29 ); + int64_t body_roster = 1 + TableLebBytes( (uint64_t) order_roster.count ); + for ( int32_t i = 0; i < order_roster.count; i++ ) + { + const int64_t elem_roster = SquadRosterEntryMeasureBody( ids, *order_roster[i] ); + if ( elem_roster < 0 ) { TableMapRelease( order_roster ); return false; } + body_roster += TableLebBytes( (uint64_t) ( elem_roster ) ) + ( elem_roster ); + } + w.putleb( ref_roster ); w.put8( 14 ); w.putleb( (uint64_t) body_roster ); + w.put8( 13 ); w.putleb( (uint64_t) order_roster.count ); + for ( int32_t i = 0; i < order_roster.count; i++ ) + { + const int64_t elem_len_roster = SquadRosterEntryMeasureBody( ids, *order_roster[i] ); + if ( elem_len_roster < 0 ) { TableMapRelease( order_roster ); return false; } + w.putleb( (uint64_t) elem_len_roster ); + if ( !SquadRosterEntrySaveBody( w, ids, *order_roster[i] ) ) { TableMapRelease( order_roster ); return false; } + } + } + TableMapRelease( order_roster ); + } + if ( value.name != 0 ) + { + w.putleb( ids.ref( 0xc4bcadba8e631b86ull, 30 ) ); w.put8( 4 ); // name + w.put32( uint32_t( value.name ) ); + } + return !w.overflow; +} + +template +inline bool SquadSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Squad & value ) +{ + if ( !SquadSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool SquadLoadBody( TableReader & r, const TableNodeMap & nodes, Squad & value ) +{ + (void) nodes; + SquadReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x1c84390d304f4f42ull: // roster + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + if ( !r.getleb( count ) ) { r.report->malformed = true; r.offset = body_end; break; } + // A MAP HEADER WHOSE ELEMENT KIND IS NOT 13 is the ordinary array + // kind mismatch of §4, and nothing about a map is special-cased + if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + TableMapFill fill = TableMapFillBegin( nodes, value.roster, (uint32_t) count ); + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + uint8_t last_key = 0; + bool landed = false; + for ( uint64_t i = 0; i < count; i++ ) + { + uint64_t elem_len = 0; + if ( !sub.getleb( elem_len ) || !sub.room( elem_len ) ) { r.report->malformed = true; break; } + const uint8_t * elem_body = sub.buffer + sub.offset; + sub.offset += (int64_t) elem_len; + SquadRosterEntryKeyRead read = SquadRosterEntryReadKey( elem_body, (int64_t) elem_len, r.ids ); + // THE KEY KIND IS CHECKED FIRST: a key read under another kind + // desynchronizes the rest of the scan, and the honest answer to a + // body whose key is not this reader's kind is the KIND, not the + // framing damage that follows from it. + if ( read.kind_bad ) + { + // A MAP WITH HALF ITS KEYS IS NOT A MAP (§2.8): the map resets + // to EMPTY, ONE kind_mismatch is counted for it, and the rest + // is skipped. Events counted inside earlier entries stand. + r.report->kind_mismatch++; + TableMapFillReset( fill ); + break; + } + if ( read.malformed ) { r.report->malformed = true; break; } + if ( read.over ) { r.report->clamped++; continue; } // skipped by its L, one count per entry + const int order = landed ? TableKeyOrder( (uint64_t) last_key, (uint64_t) read.key ) : -1; + if ( order > 0 ) + { + // DESCENDING: not a body any conforming writer produced. The map + // keeps the ascending prefix it has, the rest skips by the map's + // L, and the PARENT reads on past the field's length (§4). + r.report->malformed = true; + break; + } + SquadRosterEntry * slot = NULL; + if ( order == 0 ) + { + // EQUAL: a DUPLICATE. The slot that entry took is reset to the + // entry's defaults by the decode below, so LAST WINS WHOLE and an + // elided field of the repeat reads as its default. The map's + // count excludes it. + slot = TableMapFillLast( fill ); + r.report->duplicate++; + } + else + { + slot = TableMapFillNext( fill ); // ASCENDING: the next slot + } + if ( slot == NULL ) { r.report->malformed = true; break; } + { + TableReader elem( elem_body, (int64_t) elem_len, r.report, r.ids ); + SquadRosterEntryLoadBody( elem, *slot ); + } + last_key = read.key; // the WIRE keys of the entries that LAND + landed = true; + } + TableMapFillEnd( fill ); + } + r.offset = body_end; // the remaining entries skip by the map's L + break; + } + case 0xc4bcadba8e631b86ull: // name + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.name = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +template +inline int64_t ArmyMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Army & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // squads: a kind 14 array of kind 13 elements, INDEX order (§2.9) + TableListCursor cursor_squads = TableListElements( ctx, value.squads ); + if ( !cursor_squads.ok ) { return -1; } // the slot and the head disagree + if ( cursor_squads.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_squads = ids.ref( 0x7848019b0c02a926ull, 4 ); + int64_t body_squads = 0; + body_squads += 1 + TableLebBytes( (uint64_t) ( cursor_squads.count ) ); // the element kind byte and the count + for ( int32_t elem_i_squads = 0; elem_i_squads < cursor_squads.count; elem_i_squads++ ) + { + const int64_t elem_bytes_squads = SquadMeasureBody( ctx, numbering, ids, cursor_squads[elem_i_squads] ); + if ( elem_bytes_squads < 0 ) { return -1; } + body_squads += TableLebBytes( (uint64_t) ( elem_bytes_squads ) ) + ( elem_bytes_squads ); + } + bytes += TableLebBytes( ref_squads ) + 1 + TableLebBytes( (uint64_t) ( body_squads ) ) + ( body_squads ); + } + } + if ( value.after != 0 ) { bytes += TableLebBytes( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ) + 1 + 4; } // after + return bytes; +} + +template +inline bool ArmySaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Army & value ) +{ + { + TableListCursor cursor_squads = TableListElements( ctx, value.squads ); // squads + if ( !cursor_squads.ok ) { return false; } + if ( cursor_squads.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_squads = ids.ref( 0x7848019b0c02a926ull, 4 ); + int64_t body_squads = 0; + body_squads += 1 + TableLebBytes( (uint64_t) ( cursor_squads.count ) ); // the element kind byte and the count + for ( int32_t elem_i_squads = 0; elem_i_squads < cursor_squads.count; elem_i_squads++ ) + { + const int64_t elem_bytes_squads = SquadMeasureBody( ctx, numbering, ids, cursor_squads[elem_i_squads] ); + if ( elem_bytes_squads < 0 ) { return false; } + body_squads += TableLebBytes( (uint64_t) ( elem_bytes_squads ) ) + ( elem_bytes_squads ); + } + w.putleb( ref_squads ); w.put8( 14 ); w.putleb( (uint64_t) body_squads ); // squads + w.put8( 13 ); w.putleb( (uint64_t) ( cursor_squads.count ) ); + for ( int32_t elem_i_squads = 0; elem_i_squads < cursor_squads.count; elem_i_squads++ ) + { + { + const int64_t elem_len_squads = SquadMeasureBody( ctx, numbering, ids, cursor_squads[elem_i_squads] ); + if ( elem_len_squads < 0 ) return false; + w.putleb( (uint64_t) elem_len_squads ); + if ( !SquadSaveBody( ctx, numbering, w, ids, cursor_squads[elem_i_squads] ) ) return false; + } + } + } + } + if ( value.after != 0 ) + { + w.putleb( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ); w.put8( 4 ); // after + w.put32( uint32_t( value.after ) ); + } + return !w.overflow; +} + +template +inline bool ArmySaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Army & value ) +{ + if ( !ArmySaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool ArmyLoadBody( TableReader & r, const TableNodeMap & nodes, Army & value ) +{ + ArmyReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x7848019b0c02a926ull: // squads + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.squads, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Squad * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + uint64_t elem_len_squads = 0; + if ( !sub.getleb( elem_len_squads ) || !sub.room( elem_len_squads ) ) { r.report->malformed = true; break; } + { + TableReader elem_squads( sub.buffer + sub.offset, (int64_t) elem_len_squads, r.report, r.ids ); + SquadLoadBody( elem_squads, nodes, ( *slot ) ); + } + sub.offset += (int64_t) elem_len_squads; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xbf82010f6f71eae9ull: // after + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.after = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +template +inline int64_t DeckMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Deck & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.hands_count < 0 || value.hands_count > 3 ) { return -1; } // storage invariant + if ( value.hands_count > 0 ) + { + const uint64_t ref_hands = ids.ref( 0x81b46a69304ee2c9ull, 9 ); + int64_t body_hands = 0; + body_hands += 1 + TableLebBytes( (uint64_t) ( value.hands_count ) ); // the element kind byte and the count + for ( int32_t elem_i = 0; elem_i < value.hands_count; elem_i++ ) + { + const int64_t elem_bytes = RowMeasureBody( ctx, numbering, ids, value.hands[elem_i] ); + if ( elem_bytes < 0 ) { return -1; } + body_hands += TableLebBytes( (uint64_t) ( elem_bytes ) ) + ( elem_bytes ); + } + bytes += TableLebBytes( ref_hands ) + 1 + TableLebBytes( (uint64_t) ( body_hands ) ) + ( body_hands ); // hands + } + if ( value.after != 0 ) { bytes += TableLebBytes( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ) + 1 + 4; } // after + return bytes; +} + +template +inline bool DeckSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Deck & value ) +{ + if ( value.hands_count < 0 || value.hands_count > 3 ) { return false; } // storage invariant + if ( value.hands_count > 0 ) + { + const uint64_t ref_hands = ids.ref( 0x81b46a69304ee2c9ull, 9 ); + int64_t body_hands = 0; + body_hands += 1 + TableLebBytes( (uint64_t) ( value.hands_count ) ); // the element kind byte and the count + for ( int32_t elem_i = 0; elem_i < value.hands_count; elem_i++ ) + { + const int64_t elem_bytes = RowMeasureBody( ctx, numbering, ids, value.hands[elem_i] ); + if ( elem_bytes < 0 ) { return false; } + body_hands += TableLebBytes( (uint64_t) ( elem_bytes ) ) + ( elem_bytes ); + } + w.putleb( ref_hands ); w.put8( 14 ); w.putleb( (uint64_t) body_hands ); // hands + w.put8( 13 ); w.putleb( (uint64_t) ( value.hands_count ) ); + for ( int32_t elem_i = 0; elem_i < value.hands_count; elem_i++ ) + { + { + const int64_t elem_len = RowMeasureBody( ctx, numbering, ids, value.hands[elem_i] ); + if ( elem_len < 0 ) return false; + w.putleb( (uint64_t) elem_len ); + if ( !RowSaveBody( ctx, numbering, w, ids, value.hands[elem_i] ) ) return false; + } + } + } + if ( value.after != 0 ) + { + w.putleb( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ); w.put8( 4 ); // after + w.put32( uint32_t( value.after ) ); + } + return !w.overflow; +} + +template +inline bool DeckSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Deck & value ) +{ + if ( !DeckSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool DeckLoadBody( TableReader & r, const TableNodeMap & nodes, Deck & value ) +{ + DeckReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x81b46a69304ee2c9ull: // hands + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER — the element kind byte and the + // count, so fewer than two bytes — is INERT (§4): the field keeps the + // value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + // A DAMAGED COUNT stops the elements and nothing else: the field + // RODE, so an optional is still PRESENT (§2.3) — only a foreign + // ELEMENT KIND says the payload is not this array's at all. + if ( !counted_ok ) { r.report->malformed = true; } + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + uint64_t keep = count; + if ( keep > 3 ) { keep = 3; r.report->clamped++; } + // elements are BOUNDED by the field body: a count the length + // cannot cover keeps the decoded prefix, flags malformed, and + // the parent continues at the next field — following fields' + // bytes are never fabricated into elements + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + uint64_t decoded = 0; + for ( uint64_t i = 0; i < keep; i++ ) + { + uint64_t elem_len = 0; + if ( !sub.getleb( elem_len ) || !sub.room( elem_len ) ) { r.report->malformed = true; break; } + { + TableReader elem( sub.buffer + sub.offset, (int64_t) elem_len, r.report, r.ids ); + RowLoadBody( elem, nodes, value.hands[(int32_t) i] ); + } + sub.offset += (int64_t) elem_len; + decoded = i + 1; + } + value.hands_count = (int32_t) decoded; + } + } + r.offset = body_end; // excess elements and slack skip via the length + break; + } + case 0xbf82010f6f71eae9ull: // after + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.after = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// RowWireExtent: the extent Row's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool RowWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x3e7884bf4f412c6full && field_kind == 14 ) // items: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Sample ), (int64_t) alignof( Sample ), 13, 2, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// RowExtentAt: the node extent Row's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as RowExtentPack advances it (§2.8, §2.9). +template +inline bool RowExtentAt( const Ctx & ctx, const Row & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.items ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Sample ) - 1 ) & ~( (int64_t) alignof( Sample ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Sample ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t RowExtent( const Ctx & ctx, const Row & value ) +{ + int64_t at = 0; + if ( !RowExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// RowExtentPack: carve Row's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset RowExtentAt advances (§2.8, §2.9). +template +inline bool RowExtentPack( const Ctx & ctx, const Row & src, Row & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.items ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Sample ) - 1 ) & ~( (int64_t) alignof( Sample ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Sample ); + if ( at + bytes > capacity ) { return false; } + Sample * placed = (Sample *) ( extent + at ); + at += bytes; + dst.items.count = cursor.count; + dst.items.padding = 0; + dst.items.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.items.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Sample ) ); // trivially copyable, by construction + } + } + return true; +} + +// SheetWireExtent: the extent Sheet's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool SheetWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0xa3a7061ff10a8138ull && field_kind == 14 ) // rows: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Row ), (int64_t) alignof( Row ), 13, 2, &RowWireExtent, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// SheetExtentAt: the node extent Sheet's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as SheetExtentPack advances it (§2.8, §2.9). +template +inline bool SheetExtentAt( const Ctx & ctx, const Sheet & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.rows ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Row ) - 1 ) & ~( (int64_t) alignof( Row ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Row ); // the whole array FIRST + for ( int32_t i = 0; i < cursor.count; i++ ) // then, element by element in index order + { + if ( !RowExtentAt( ctx, cursor[i], at ) ) { return false; } + } + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t SheetExtent( const Ctx & ctx, const Sheet & value ) +{ + int64_t at = 0; + if ( !SheetExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// SheetExtentPack: carve Sheet's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset SheetExtentAt advances (§2.8, §2.9). +template +inline bool SheetExtentPack( const Ctx & ctx, const Sheet & src, Sheet & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.rows ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Row ) - 1 ) & ~( (int64_t) alignof( Row ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Row ); + if ( at + bytes > capacity ) { return false; } + Row * placed = (Row *) ( extent + at ); + at += bytes; + dst.rows.count = cursor.count; + dst.rows.padding = 0; + dst.rows.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.rows.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Row ) ); // trivially copyable, by construction + } + for ( int32_t i = 0; i < cursor.count; i++ ) + { + if ( !RowExtentPack( ctx, cursor[i], placed[i], extent, at, capacity ) ) { return false; } + } + } + return true; +} + +// SquadWireExtent: the extent Squad's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool SquadWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x1c84390d304f4f42ull && field_kind == 14 ) // roster + { + uint64_t map_len = 0; + if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } + const uint8_t * map_body = r.buffer + r.offset; + r.offset += (int64_t) map_len; + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( SquadRosterEntry ), (int64_t) alignof( SquadRosterEntry ), NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// SquadExtentAt: the node extent Squad's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as SquadExtentPack advances it (§2.8, §2.9). +template +inline bool SquadExtentAt( const Ctx & ctx, const Squad & value, int64_t & at ) +{ + { + TableMapCursor cursor = TableMapOrder( ctx, value.roster ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( SquadRosterEntry ) + at += (int64_t) cursor.count * (int64_t) sizeof( SquadRosterEntry ); // the whole array FIRST + TableMapRelease( cursor ); + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t SquadExtent( const Ctx & ctx, const Squad & value ) +{ + int64_t at = 0; + if ( !SquadExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// SquadExtentPack: carve Squad's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset SquadExtentAt advances (§2.8, §2.9). +template +inline bool SquadExtentPack( const Ctx & ctx, const Squad & src, Squad & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableMapCursor cursor = TableMapOrder( ctx, src.roster ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( SquadRosterEntry ); + if ( at + bytes > capacity ) { TableMapRelease( cursor ); return false; } + SquadRosterEntry * placed = (SquadRosterEntry *) ( extent + at ); + at += bytes; + dst.roster.count = cursor.count; + dst.roster.padding = 0; + dst.roster.entries.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.roster.entries ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) + { + memcpy( (void *) ( placed + i ), (const void *) cursor[i], sizeof( SquadRosterEntry ) ); // trivially copyable, by construction + } + TableMapRelease( cursor ); + } + return true; +} + +// ArmyWireExtent: the extent Army's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool ArmyWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x7848019b0c02a926ull && field_kind == 14 ) // squads: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Squad ), (int64_t) alignof( Squad ), 13, 2, &SquadWireExtent, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// ArmyExtentAt: the node extent Army's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as ArmyExtentPack advances it (§2.8, §2.9). +template +inline bool ArmyExtentAt( const Ctx & ctx, const Army & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.squads ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Squad ) - 1 ) & ~( (int64_t) alignof( Squad ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Squad ); // the whole array FIRST + for ( int32_t i = 0; i < cursor.count; i++ ) // then, element by element in index order + { + if ( !SquadExtentAt( ctx, cursor[i], at ) ) { return false; } + } + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t ArmyExtent( const Ctx & ctx, const Army & value ) +{ + int64_t at = 0; + if ( !ArmyExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// ArmyExtentPack: carve Army's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset ArmyExtentAt advances (§2.8, §2.9). +template +inline bool ArmyExtentPack( const Ctx & ctx, const Army & src, Army & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.squads ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Squad ) - 1 ) & ~( (int64_t) alignof( Squad ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Squad ); + if ( at + bytes > capacity ) { return false; } + Squad * placed = (Squad *) ( extent + at ); + at += bytes; + dst.squads.count = cursor.count; + dst.squads.padding = 0; + dst.squads.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.squads.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Squad ) ); // trivially copyable, by construction + } + for ( int32_t i = 0; i < cursor.count; i++ ) + { + if ( !SquadExtentPack( ctx, cursor[i], placed[i], extent, at, capacity ) ) { return false; } + } + } + return true; +} + +// DeckWireExtent: the extent Deck's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool DeckWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x81b46a69304ee2c9ull && field_kind == 14 ) // hands: a nesting that holds a list or a map + { + uint64_t nested_len = 0; + if ( !r.getleb( nested_len ) || !r.room( nested_len ) ) { return true; } + const uint8_t * nested_body = r.buffer + r.offset; + r.offset += (int64_t) nested_len; + if ( !TableWireExtentElements( nested_body, (int64_t) nested_len, at, &RowWireExtent, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// DeckExtentAt: the node extent Deck's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as DeckExtentPack advances it (§2.8, §2.9). +template +inline bool DeckExtentAt( const Ctx & ctx, const Deck & value, int64_t & at ) +{ + for ( int32_t i = 0; i < value.hands_count && i < 3; i++ ) // hands + { + if ( !RowExtentAt( ctx, value.hands[i], at ) ) { return false; } + } + for ( int32_t i = value.hands_count; i < 3; i++ ) // hands: the slots the walk does not reach (§7.6) + { + if ( !TableExtentUnreachedEmpty( RowExtent( ctx, value.hands[i] ) ) ) { return false; } + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t DeckExtent( const Ctx & ctx, const Deck & value ) +{ + int64_t at = 0; + if ( !DeckExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// DeckExtentPack: carve Deck's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset DeckExtentAt advances (§2.8, §2.9). +template +inline bool DeckExtentPack( const Ctx & ctx, const Deck & src, Deck & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + for ( int32_t i = 0; i < src.hands_count && i < 3; i++ ) // hands + { + if ( !RowExtentPack( ctx, src.hands[i], dst.hands[i], extent, at, capacity ) ) { return false; } + } + for ( int32_t i = src.hands_count; i < 3; i++ ) // hands: the slots the walk does not reach (§7.6) + { + if ( !TableExtentUnreachedEmpty( RowExtent( ctx, src.hands[i] ) ) ) { return false; } + } + return true; +} + +// ---- Squad.roster: the builder's five and the side index (§2.8) ---- + +// INSERT: the key is copied, the value is handed back at its defaults to +// fill. A DUPLICATE key REPLACES — the value is reset and the same entry +// handed back, key and address unchanged — so a caller that wants to know +// writes Find first. NULL is NOT INSERTED: a key longer than the bound, +// because a truncated key would be a merged entry, and an arena that +// cannot carve another segment, alike. +inline Item * SquadRosterInsert( TableWorker & worker, TableMap & map, uint8_t key ) +{ + if ( worker.arena == NULL ) { return NULL; } + SquadRosterEntry * found = TableMapScan( *worker.arena, map, key ); // one LINEAR SCAN of the live entries + if ( found != NULL ) + { + TableResetMapValue( *found ); // a duplicate REPLACES: the value goes back to its defaults + return TableEntryValue( found ); + } + TableMapHead * head = TableMapReach( worker, map ); + if ( head == NULL ) { return NULL; } + SquadRosterEntry * entry = TableMapAppend( worker, head, map ); // APPENDS; nothing ever moves (§6.4) + if ( entry == NULL ) { return NULL; } + TableReset( *entry ); + TableEntrySetKey( *entry, key ); + return TableEntryValue( entry ); +} + +// FIND on the builder: the same linear scan, O( n ) key compares over the +// segments in insertion order. NULL when absent. The builder builds NO +// INDEX, and that is a rule — the sort happens once, at Lock, Save or +// Cook, and every lookup that matters runs over the sorted region. +inline Item * SquadRosterFind( TableArena & arena, TableMap & map, uint8_t key ) +{ + SquadRosterEntry * found = TableMapScan( arena, map, key ); + return found != NULL ? TableEntryValue( found ) : NULL; +} + +// ERASE: marks the entry DEAD, one bit in the segment's slot and not in the +// entry table. False when absent. Its storage is held until the builder +// resets and never reused mid-build, because reusing a slot would make "an +// entry's address is stable" false for exactly one case. +inline bool SquadRosterErase( TableArena & arena, TableMap & map, uint8_t key ) +{ + return TableMapErase( arena, map, key ); +} + +// EACH on the builder: INSERTION order, live entries only. +inline TableMapEach SquadRosterEach( const TableArena & arena, const TableMap & map ) +{ + return TableMapEachOf( arena, map ); +} + +// ---- the OPTIONAL INDEX: caller-owned, built at load, never stored ---- +// +// Open addressing with linear probing over the sorted array, for a map large +// enough that log n compares over a cold array cost more than one hash and a +// probe. ITS HASH AND ITS LOAD FACTOR ARE NOT A CROSS-PORT CONTRACT: the +// index is never stored, so no golden, no cook-check rule and no +// build-version line ever names either. What a port is held to is the +// CONTRACT of the lookup — the same value the sorted array's Find returns +// for the same key, and no allocation past the storage the caller handed in. +inline int64_t SquadRosterIndexMeasure( const TableMap & map ) +{ + return (int64_t) TableMapIndexSlots( map.count ) * (int64_t) sizeof( int32_t ); +} + +inline TableMapIndex SquadRosterIndex( const TableMap & map, void * storage, int64_t bytes ) +{ + TableMapIndex index; + const int32_t slots = TableMapIndexSlots( map.count ); + if ( storage == NULL || bytes < (int64_t) slots * (int64_t) sizeof( int32_t ) ) { return index; } + index.slots = (int32_t *) storage; + index.capacity = slots; + for ( int32_t i = 0; i < slots; i++ ) { index.slots[i] = 0; } + const SquadRosterEntry * entries = map.Entries(); + for ( int32_t i = 0; i < map.count; i++ ) // ONE PASS over the sorted array + { + int32_t at = (int32_t) ( TableMapHash( (uint64_t) entries[i].key ) & (uint64_t) ( slots - 1 ) ); + while ( index.slots[at] != 0 ) { at = ( at + 1 ) & ( slots - 1 ); } + index.slots[at] = i + 1; // slots are ENTRY INDICES; 0 is an empty slot + } + index.good = true; + return index; +} + +inline const Item * SquadRosterIndexFind( const TableMapIndex & index, const TableMap & map, uint8_t key ) +{ + if ( !index.good ) { return map.Find( key ); } // an index that did not build is not a wrong answer + const SquadRosterEntry * entries = map.Entries(); + int32_t at = (int32_t) ( TableMapHash( (uint64_t) key ) & (uint64_t) ( index.capacity - 1 ) ); + for ( int32_t probe = 0; probe < index.capacity; probe++ ) + { + const int32_t slot = index.slots[at]; + if ( slot == 0 ) { return NULL; } + if ( TableEntryOrder( entries[slot - 1], key ) == 0 ) { return TableEntryFound( entries + slot - 1 ); } + at = ( at + 1 ) & ( index.capacity - 1 ); + } + return NULL; +} + +// ---- Row.items: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Sample * RowItemsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool RowItemsErase( TableArena & arena, TableList & list, const Sample * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach RowItemsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Sheet.rows: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Row * SheetRowsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool SheetRowsErase( TableArena & arena, TableList & list, const Row * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach SheetRowsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Army.squads: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Squad * ArmySquadsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool ArmySquadsErase( TableArena & arena, TableList & list, const Squad * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach ArmySquadsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// RowNumber: number everything Row POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool RowNumber( const Ctx & ctx, TableNumbering & numbering, const Row & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// RowPackMeasure: the packed region bytes of everything Row POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t RowPackMeasure( const Ctx & ctx, TablePackMap & seen, const Row & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// RowPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool RowPackEdges( const Ctx & ctx, TablePackMap & seen, const Row & src, Row & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool RowPack( const Ctx & ctx, TablePackMap & seen, const Row & src, Row & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Row ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Row ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !RowExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return RowPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool RowPackEdges( const Ctx & ctx, TablePackMap & seen, const Row & src, Row & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// SheetNumber: number everything Sheet POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool SheetNumber( const Ctx & ctx, TableNumbering & numbering, const Sheet & value ) +{ + { // rows: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_rows = TableListElements( ctx, value.rows ); + if ( !cursor_rows.ok ) { return false; } + for ( int32_t i = 0; i < cursor_rows.count; i++ ) + { + if ( !RowNumber( ctx, numbering, cursor_rows[i] ) ) { return false; } + } + } + { + const Row * pointee = RowAt( ctx, value.pinned ); // pinned + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( numbering.seen, (const void *) pointee, + (int64_t) ( numbering.count + 2 ), taken, slot ); // its index, if this is its first visit + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + } + else + { + TableNodeEntry node; + node.node = (const void *) pointee; + node.type_id = 0xa013e119fec906fbull; // fnv1a64( "Row" ) + node.type_slot = 55; // its slot in the unit's vocabulary (§3.3) + node.measure = &TableNodeMeasureThunk; + node.save = &TableNodeSaveThunk; + if ( !TableNumberingAppend( numbering, node ) ) { return false; } + if ( !RowNumber( ctx, numbering, *pointee ) ) { return false; } + TablePackMapClose( numbering.seen, (const void *) pointee, slot ); + } + } + } + return true; +} + +// SheetPackMeasure: the packed region bytes of everything Sheet POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t SheetPackMeasure( const Ctx & ctx, TablePackMap & seen, const Sheet & value ) +{ + int64_t bytes = 0; + { // rows: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_rows = TableListElements( ctx, value.rows ); + if ( !cursor_rows.ok ) { return -1; } + for ( int32_t i = 0; i < cursor_rows.count; i++ ) + { + int64_t inner = RowPackMeasure( ctx, seen, cursor_rows[i] ); + if ( inner < 0 ) { return -1; } + bytes += inner; + } + } + { + const Row * pointee = RowAt( ctx, value.pinned ); // pinned + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, 0, taken, slot ); + if ( entry == NULL ) { return -1; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return -1; } // a data cycle + } + else + { + int64_t inner = RowPackMeasure( ctx, seen, *pointee ); + if ( inner < 0 ) { return -1; } + TablePackMapClose( seen, (const void *) pointee, slot ); + int64_t node_extent = RowExtent( ctx, *pointee ); + if ( node_extent < 0 ) { return -1; } + bytes += TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + node_extent ) + inner; + } + } + } + return bytes; +} + +// SheetPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool SheetPackEdges( const Ctx & ctx, TablePackMap & seen, const Sheet & src, Sheet & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool SheetPack( const Ctx & ctx, TablePackMap & seen, const Sheet & src, Sheet & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Sheet ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Sheet ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !SheetExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return SheetPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool SheetPackEdges( const Ctx & ctx, TablePackMap & seen, const Sheet & src, Sheet & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + { // rows: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_rows = TableListElements( ctx, src.rows ); + if ( !cursor_rows.ok ) { return false; } + Row * placed_rows = (Row *) ( dst.rows.elements.value != 0 ? ( (uint8_t *) &dst.rows.elements + dst.rows.elements.value ) : NULL ); + for ( int32_t i = 0; i < cursor_rows.count; i++ ) + { + if ( !RowPackEdges( ctx, seen, cursor_rows[i], placed_rows[i], base, capacity, used ) ) { return false; } + } + } + { + dst.pinned.value = 0; // pinned + const Row * pointee = RowAt( ctx, src.pinned ); + if ( pointee != NULL ) + { + int64_t at = TableAlignUp64( used ); // where it WOULD land, if this is its first visit + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, at, taken, slot ); + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + dst.pinned.value = (int64_t) ( ( base + entry->offset ) - (const uint8_t *) &dst.pinned ); // the one body it already has + } + else + { + int64_t node_extent = RowExtent( ctx, *pointee ); + if ( node_extent < 0 ) { return false; } + const int64_t node_bytes = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + node_extent ); + if ( at + node_bytes > capacity ) { return false; } + used = at + node_bytes; + Row * child = new ( base + at ) Row; // lifetime only: the Pack below memcpy's the whole node over it + dst.pinned.value = (int64_t) ( ( base + at ) - (const uint8_t *) &dst.pinned ); + if ( !RowPack( ctx, seen, *pointee, *child, base, capacity, used ) ) { return false; } + TablePackMapClose( seen, (const void *) pointee, slot ); + } + } + } + return true; +} + +// SquadNumber: number everything Squad POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool SquadNumber( const Ctx & ctx, TableNumbering & numbering, const Squad & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// SquadPackMeasure: the packed region bytes of everything Squad POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t SquadPackMeasure( const Ctx & ctx, TablePackMap & seen, const Squad & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// SquadPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool SquadPackEdges( const Ctx & ctx, TablePackMap & seen, const Squad & src, Squad & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool SquadPack( const Ctx & ctx, TablePackMap & seen, const Squad & src, Squad & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Squad ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Squad ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !SquadExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return SquadPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool SquadPackEdges( const Ctx & ctx, TablePackMap & seen, const Squad & src, Squad & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// ArmyNumber: number everything Army POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool ArmyNumber( const Ctx & ctx, TableNumbering & numbering, const Army & value ) +{ + { // squads: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_squads = TableListElements( ctx, value.squads ); + if ( !cursor_squads.ok ) { return false; } + for ( int32_t i = 0; i < cursor_squads.count; i++ ) + { + if ( !SquadNumber( ctx, numbering, cursor_squads[i] ) ) { return false; } + } + } + return true; +} + +// ArmyPackMeasure: the packed region bytes of everything Army POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t ArmyPackMeasure( const Ctx & ctx, TablePackMap & seen, const Army & value ) +{ + int64_t bytes = 0; + { // squads: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_squads = TableListElements( ctx, value.squads ); + if ( !cursor_squads.ok ) { return -1; } + for ( int32_t i = 0; i < cursor_squads.count; i++ ) + { + int64_t inner = SquadPackMeasure( ctx, seen, cursor_squads[i] ); + if ( inner < 0 ) { return -1; } + bytes += inner; + } + } + return bytes; +} + +// ArmyPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool ArmyPackEdges( const Ctx & ctx, TablePackMap & seen, const Army & src, Army & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool ArmyPack( const Ctx & ctx, TablePackMap & seen, const Army & src, Army & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Army ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Army ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !ArmyExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return ArmyPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool ArmyPackEdges( const Ctx & ctx, TablePackMap & seen, const Army & src, Army & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + { // squads: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_squads = TableListElements( ctx, src.squads ); + if ( !cursor_squads.ok ) { return false; } + Squad * placed_squads = (Squad *) ( dst.squads.elements.value != 0 ? ( (uint8_t *) &dst.squads.elements + dst.squads.elements.value ) : NULL ); + for ( int32_t i = 0; i < cursor_squads.count; i++ ) + { + if ( !SquadPackEdges( ctx, seen, cursor_squads[i], placed_squads[i], base, capacity, used ) ) { return false; } + } + } + return true; +} + +// DeckNumber: number everything Deck POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool DeckNumber( const Ctx & ctx, TableNumbering & numbering, const Deck & value ) +{ + for ( int32_t i = 0; i < value.hands_count && i < 3; i++ ) // hands + { + if ( !RowNumber( ctx, numbering, value.hands[i] ) ) { return false; } + } + return true; +} + +// DeckPackMeasure: the packed region bytes of everything Deck POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t DeckPackMeasure( const Ctx & ctx, TablePackMap & seen, const Deck & value ) +{ + int64_t bytes = 0; + for ( int32_t i = 0; i < value.hands_count && i < 3; i++ ) // hands + { + int64_t inner = RowPackMeasure( ctx, seen, value.hands[i] ); + if ( inner < 0 ) { return -1; } + bytes += inner; + } + return bytes; +} + +// DeckPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool DeckPackEdges( const Ctx & ctx, TablePackMap & seen, const Deck & src, Deck & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool DeckPack( const Ctx & ctx, TablePackMap & seen, const Deck & src, Deck & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Deck ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Deck ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !DeckExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return DeckPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool DeckPackEdges( const Ctx & ctx, TablePackMap & seen, const Deck & src, Deck & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + for ( int32_t i = 0; i < src.hands_count && i < 3; i++ ) // hands + { + if ( !RowPackEdges( ctx, seen, src.hands[i], dst.hands[i], base, capacity, used ) ) { return false; } + } + return true; +} + +// ---- Row: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: RowBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Row is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct RowBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + RowBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~RowBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + RowBuilder( const RowBuilder & ) = delete; + RowBuilder & operator=( const RowBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Row * GetRoot() { return arena.locked ? NULL : (Row *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Row * AsConst() const { return (const Row *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool RowBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Row & root = *(const Row *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = RowPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = RowExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + Row * destination = new ( packed ) Row; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !RowPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Row on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// RowNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t RowNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// RowNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void RowNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// RowNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t RowNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// RowNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t RowNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// RowNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void RowNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = RowNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? RowNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool RowNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Row & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return RowNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t RowMeasureWire( const Ctx & ctx, const Row & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( RowNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = RowMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t RowSaveWire( const Ctx & ctx, const Row & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !RowNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = RowSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == RowMeasure( root ) +} + +inline int64_t RowMeasure( const Row * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return RowMeasureWire( ctx, *root, allocator ); +} + +inline int64_t RowSave( const Row * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return RowSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t RowMeasure( const RowBuilder & builder ) +{ + if ( builder.region != NULL ) { return RowMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return RowMeasureWire( ctx, *(const Row *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t RowSave( const RowBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return RowSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return RowSaveWire( ctx, *(const Row *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t RowMeasureMessage( const Row * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return RowMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t RowSaveMessage( const Row * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return RowSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t RowMeasureMessage( const RowBuilder & builder ) +{ + if ( builder.region != NULL ) { return RowMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return RowMeasureWire( ctx, *(const Row *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t RowSaveMessage( const RowBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return RowSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return RowSaveWire( ctx, *(const Row *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// RowLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t RowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = RowNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// RowLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Row * RowLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Row ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = RowNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xa013e119fec906fbull; + Row * root = new ( region ) Row; // lifetime only: LoadBody's first act is RowReset + RowReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = RowNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + RowNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + RowNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Row ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + RowLoadBody( r, nodes, *root ); + return root; +} + +// RowLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t RowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = RowNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// RowLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Row * RowLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Row ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = RowNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xa013e119fec906fbull; + Row * root = new ( region ) Row; // lifetime only: LoadBody's first act is RowReset + RowReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = RowNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + RowNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + RowNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Row ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + RowLoadBody( r, nodes, *root ); + return root; +} + +// RowLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool RowLoadBuilder( RowBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Row * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0xa013e119fec906fbull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = RowNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + RowNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = RowLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Sheet: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: SheetBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Sheet is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct SheetBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + SheetBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~SheetBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + SheetBuilder( const SheetBuilder & ) = delete; + SheetBuilder & operator=( const SheetBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Sheet * GetRoot() { return arena.locked ? NULL : (Sheet *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Sheet * AsConst() const { return (const Sheet *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool SheetBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Sheet & root = *(const Sheet *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = SheetPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = SheetExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + Sheet * destination = new ( packed ) Sheet; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !SheetPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Sheet on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// SheetNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t SheetNodeStorage( uint64_t type_id, const uint8_t * body, int64_t length, const TableIdTable * ids, TableRefuseReason & reason ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + case 0xa013e119fec906fbull: // Row + { + int64_t extent = 0; + if ( !RowWireExtent( body, length, extent, ids, reason ) ) { return kTableNodeRefused; } + return TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + extent ); + } + default: break; + } + return -1; +} + +// SheetNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void SheetNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + switch ( type_id ) + { + case 0xa013e119fec906fbull: { Row * node = new ( at ) Row; RowReset( *node ); break; } // Row + default: break; + } +} + +// SheetNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t SheetNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + case 0xa013e119fec906fbull: return TableAlignUp64( (int64_t) sizeof( Row ) ); // Row + default: break; + } + return 0; +} + +// SheetNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t SheetNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + switch ( type_id ) + { + case 0xa013e119fec906fbull: return (uint32_t) worker.Alloc().ref.value; // Row + default: break; + } + return 0; +} + +// SheetNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void SheetNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + TableRefuseReason reason = count_over_length; // pass one already refused what this could refuse + const int64_t storage = SheetNodeStorage( type_id, r.buffer, r.size, r.ids, reason ); + const int64_t record = storage > 0 ? SheetNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + switch ( type_id ) + { + case 0xa013e119fec906fbull: RowLoadBody( r, nodes, *(Row *) at ); break; // Row + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool SheetNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Sheet & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return SheetNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t SheetMeasureWire( const Ctx & ctx, const Sheet & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( SheetNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = SheetMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t SheetSaveWire( const Ctx & ctx, const Sheet & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !SheetNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = SheetSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == SheetMeasure( root ) +} + +inline int64_t SheetMeasure( const Sheet * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SheetMeasureWire( ctx, *root, allocator ); +} + +inline int64_t SheetSave( const Sheet * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SheetSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t SheetMeasure( const SheetBuilder & builder ) +{ + if ( builder.region != NULL ) { return SheetMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SheetMeasureWire( ctx, *(const Sheet *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t SheetSave( const SheetBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return SheetSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SheetSaveWire( ctx, *(const Sheet *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t SheetMeasureMessage( const Sheet * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SheetMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t SheetSaveMessage( const Sheet * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SheetSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t SheetMeasureMessage( const SheetBuilder & builder ) +{ + if ( builder.region != NULL ) { return SheetMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return SheetMeasureWire( ctx, *(const Sheet *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t SheetSaveMessage( const SheetBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return SheetSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return SheetSaveWire( ctx, *(const Sheet *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// SheetLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t SheetLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !SheetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SheetNodeStorage( type_id, body, length, &ids_table, reason ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// SheetLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Sheet * SheetLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Sheet ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !SheetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SheetNodeStorage( type_id, body, length, &ids_table, reason ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x0cc9e0af9a85fbc8ull; + Sheet * root = new ( region ) Sheet; // lifetime only: LoadBody's first act is SheetReset + SheetReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = SheetNodeStorage( type_id, body, length, &ids_table, reason ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + SheetNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SheetNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Sheet ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + SheetLoadBody( r, nodes, *root ); + return root; +} + +// SheetLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t SheetLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !SheetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SheetNodeStorage( type_id, body, length, &ids_table, reason ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// SheetLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Sheet * SheetLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Sheet ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !SheetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SheetNodeStorage( type_id, body, length, &ids_table, reason ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x0cc9e0af9a85fbc8ull; + Sheet * root = new ( region ) Sheet; // lifetime only: LoadBody's first act is SheetReset + SheetReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = SheetNodeStorage( type_id, body, length, &ids_table, reason ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + SheetNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SheetNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Sheet ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + SheetLoadBody( r, nodes, *root ); + return root; +} + +// SheetLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool SheetLoadBuilder( SheetBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Sheet * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0x0cc9e0af9a85fbc8ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = SheetNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SheetNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = SheetLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Squad: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: SquadBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Squad is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct SquadBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + SquadBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~SquadBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + SquadBuilder( const SquadBuilder & ) = delete; + SquadBuilder & operator=( const SquadBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Squad * GetRoot() { return arena.locked ? NULL : (Squad *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Squad * AsConst() const { return (const Squad *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool SquadBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Squad & root = *(const Squad *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = SquadPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = SquadExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + Squad * destination = new ( packed ) Squad; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !SquadPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Squad on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// SquadNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t SquadNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// SquadNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void SquadNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// SquadNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t SquadNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// SquadNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t SquadNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// SquadNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void SquadNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = SquadNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? SquadNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool SquadNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Squad & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return SquadNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t SquadMeasureWire( const Ctx & ctx, const Squad & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( SquadNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = SquadMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t SquadSaveWire( const Ctx & ctx, const Squad & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !SquadNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = SquadSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == SquadMeasure( root ) +} + +inline int64_t SquadMeasure( const Squad * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SquadMeasureWire( ctx, *root, allocator ); +} + +inline int64_t SquadSave( const Squad * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SquadSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t SquadMeasure( const SquadBuilder & builder ) +{ + if ( builder.region != NULL ) { return SquadMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SquadMeasureWire( ctx, *(const Squad *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t SquadSave( const SquadBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return SquadSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SquadSaveWire( ctx, *(const Squad *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t SquadMeasureMessage( const Squad * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SquadMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t SquadSaveMessage( const Squad * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SquadSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t SquadMeasureMessage( const SquadBuilder & builder ) +{ + if ( builder.region != NULL ) { return SquadMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return SquadMeasureWire( ctx, *(const Squad *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t SquadSaveMessage( const SquadBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return SquadSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return SquadSaveWire( ctx, *(const Squad *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// SquadLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t SquadLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SquadNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// SquadLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Squad * SquadLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Squad ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SquadNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xec07a2f760550a91ull; + Squad * root = new ( region ) Squad; // lifetime only: LoadBody's first act is SquadReset + SquadReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = SquadNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + SquadNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SquadNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Squad ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + SquadLoadBody( r, nodes, *root ); + return root; +} + +// SquadLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t SquadLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SquadNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// SquadLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Squad * SquadLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Squad ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SquadNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xec07a2f760550a91ull; + Squad * root = new ( region ) Squad; // lifetime only: LoadBody's first act is SquadReset + SquadReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = SquadNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + SquadNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SquadNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Squad ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + SquadLoadBody( r, nodes, *root ); + return root; +} + +// SquadLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool SquadLoadBuilder( SquadBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Squad * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0xec07a2f760550a91ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = SquadNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SquadNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = SquadLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Army: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: ArmyBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Army is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct ArmyBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + ArmyBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~ArmyBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + ArmyBuilder( const ArmyBuilder & ) = delete; + ArmyBuilder & operator=( const ArmyBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Army * GetRoot() { return arena.locked ? NULL : (Army *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Army * AsConst() const { return (const Army *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool ArmyBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Army & root = *(const Army *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = ArmyPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = ArmyExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + Army * destination = new ( packed ) Army; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !ArmyPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Army on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// ArmyNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t ArmyNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// ArmyNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void ArmyNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// ArmyNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t ArmyNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// ArmyNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t ArmyNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// ArmyNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void ArmyNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = ArmyNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? ArmyNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool ArmyNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Army & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return ArmyNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t ArmyMeasureWire( const Ctx & ctx, const Army & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( ArmyNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = ArmyMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t ArmySaveWire( const Ctx & ctx, const Army & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !ArmyNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = ArmySaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == ArmyMeasure( root ) +} + +inline int64_t ArmyMeasure( const Army * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return ArmyMeasureWire( ctx, *root, allocator ); +} + +inline int64_t ArmySave( const Army * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return ArmySaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t ArmyMeasure( const ArmyBuilder & builder ) +{ + if ( builder.region != NULL ) { return ArmyMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return ArmyMeasureWire( ctx, *(const Army *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t ArmySave( const ArmyBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return ArmySave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return ArmySaveWire( ctx, *(const Army *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t ArmyMeasureMessage( const Army * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return ArmyMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t ArmySaveMessage( const Army * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return ArmySaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t ArmyMeasureMessage( const ArmyBuilder & builder ) +{ + if ( builder.region != NULL ) { return ArmyMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return ArmyMeasureWire( ctx, *(const Army *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t ArmySaveMessage( const ArmyBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return ArmySaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return ArmySaveWire( ctx, *(const Army *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// ArmyLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t ArmyLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !ArmyWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = ArmyNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// ArmyLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Army * ArmyLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Army ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !ArmyWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = ArmyNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x06e2378b553a9e84ull; + Army * root = new ( region ) Army; // lifetime only: LoadBody's first act is ArmyReset + ArmyReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = ArmyNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + ArmyNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + ArmyNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Army ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + ArmyLoadBody( r, nodes, *root ); + return root; +} + +// ArmyLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t ArmyLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !ArmyWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = ArmyNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// ArmyLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Army * ArmyLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Army ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !ArmyWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = ArmyNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x06e2378b553a9e84ull; + Army * root = new ( region ) Army; // lifetime only: LoadBody's first act is ArmyReset + ArmyReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = ArmyNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + ArmyNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + ArmyNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Army ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + ArmyLoadBody( r, nodes, *root ); + return root; +} + +// ArmyLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool ArmyLoadBuilder( ArmyBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Army * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0x06e2378b553a9e84ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = ArmyNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + ArmyNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = ArmyLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Deck: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: DeckBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Deck is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct DeckBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + DeckBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~DeckBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + DeckBuilder( const DeckBuilder & ) = delete; + DeckBuilder & operator=( const DeckBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Deck * GetRoot() { return arena.locked ? NULL : (Deck *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Deck * AsConst() const { return (const Deck *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool DeckBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Deck & root = *(const Deck *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = DeckPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = DeckExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + Deck * destination = new ( packed ) Deck; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !DeckPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Deck on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// DeckNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t DeckNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// DeckNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void DeckNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// DeckNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t DeckNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// DeckNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t DeckNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// DeckNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void DeckNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = DeckNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? DeckNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool DeckNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Deck & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return DeckNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t DeckMeasureWire( const Ctx & ctx, const Deck & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( DeckNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = DeckMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t DeckSaveWire( const Ctx & ctx, const Deck & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !DeckNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = DeckSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == DeckMeasure( root ) +} + +inline int64_t DeckMeasure( const Deck * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return DeckMeasureWire( ctx, *root, allocator ); +} + +inline int64_t DeckSave( const Deck * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return DeckSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t DeckMeasure( const DeckBuilder & builder ) +{ + if ( builder.region != NULL ) { return DeckMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return DeckMeasureWire( ctx, *(const Deck *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t DeckSave( const DeckBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return DeckSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return DeckSaveWire( ctx, *(const Deck *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t DeckMeasureMessage( const Deck * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return DeckMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t DeckSaveMessage( const Deck * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return DeckSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t DeckMeasureMessage( const DeckBuilder & builder ) +{ + if ( builder.region != NULL ) { return DeckMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return DeckMeasureWire( ctx, *(const Deck *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t DeckSaveMessage( const DeckBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return DeckSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return DeckSaveWire( ctx, *(const Deck *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// DeckLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t DeckLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !DeckWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = DeckNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// DeckLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Deck * DeckLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Deck ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !DeckWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = DeckNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xd043187343bfcfe8ull; + Deck * root = new ( region ) Deck; // lifetime only: LoadBody's first act is DeckReset + DeckReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = DeckNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + DeckNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + DeckNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Deck ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + DeckLoadBody( r, nodes, *root ); + return root; +} + +// DeckLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t DeckLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !DeckWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = DeckNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// DeckLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Deck * DeckLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Deck ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !DeckWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = DeckNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xd043187343bfcfe8ull; + Deck * root = new ( region ) Deck; // lifetime only: LoadBody's first act is DeckReset + DeckReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = DeckNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + DeckNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + DeckNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Deck ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + DeckLoadBody( r, nodes, *root ); + return root; +} + +// DeckLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool DeckLoadBuilder( DeckBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Deck * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0xd043187343bfcfe8ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = DeckNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + DeckNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = DeckLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- the cooked form: point at a cook (docs/SPEC-TABLES.md §7) ---- + +// SampleOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Sample IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Sample * SampleOpen( const void * bytes, uint64_t length ) +{ + return (const Sample *) TableCookOpen( bytes, length, (uint64_t) sizeof( Sample ), (uint64_t) alignof( Sample ) ); +} + +// RowOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH RowAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Row * RowOpen( const void * bytes, uint64_t length ) +{ + return (const Row *) TableCookOpen( bytes, length, (uint64_t) sizeof( Row ), (uint64_t) alignof( Row ) ); +} + +// SheetOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH SheetAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Sheet * SheetOpen( const void * bytes, uint64_t length ) +{ + return (const Sheet *) TableCookOpen( bytes, length, (uint64_t) sizeof( Sheet ), (uint64_t) alignof( Sheet ) ); +} + +// ItemOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Item IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Item * ItemOpen( const void * bytes, uint64_t length ) +{ + return (const Item *) TableCookOpen( bytes, length, (uint64_t) sizeof( Item ), (uint64_t) alignof( Item ) ); +} + +// SquadOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH SquadAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Squad * SquadOpen( const void * bytes, uint64_t length ) +{ + return (const Squad *) TableCookOpen( bytes, length, (uint64_t) sizeof( Squad ), (uint64_t) alignof( Squad ) ); +} + +// ArmyOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH ArmyAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Army * ArmyOpen( const void * bytes, uint64_t length ) +{ + return (const Army *) TableCookOpen( bytes, length, (uint64_t) sizeof( Army ), (uint64_t) alignof( Army ) ); +} + +// DeckOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH DeckAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Deck * DeckOpen( const void * bytes, uint64_t length ) +{ + return (const Deck *) TableCookOpen( bytes, length, (uint64_t) sizeof( Deck ), (uint64_t) alignof( Deck ) ); +} + +// ---- the cooked form: WRITE a cook (docs/SPEC-TABLES.md §7.6) ---- +// +// The bytes are `schema cook`'s, and the tool stays the reference: the two +// writers are held to one file, byte for byte, in both byte orders. A cook is +// content-addressed by (asset hash, build version), so two writers of one +// instance produce ONE artifact or the pair means nothing. + +inline void SampleCookBody( uint8_t * at, const Sample & value, TableByteOrder order ); +template inline bool RowCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Row & value, TableByteOrder order ); +template inline bool SheetCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Sheet & value, TableByteOrder order ); +inline void ItemCookBody( uint8_t * at, const Item & value, TableByteOrder order ); +inline void SquadRosterEntryCookBody( uint8_t * at, const SquadRosterEntry & value, TableByteOrder order ); +template inline bool SquadCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Squad & value, TableByteOrder order ); +template inline bool ArmyCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Army & value, TableByteOrder order ); +template inline bool DeckCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Deck & value, TableByteOrder order ); + +inline void SampleCookBody( uint8_t * at, const Sample & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.v, 4, order ); +} + +template inline bool RowCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Row & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // items: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.label, 4, order ); + return true; +} + +template inline bool SheetCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Sheet & value, TableByteOrder order ) +{ + table_cook_put( at + 0, 0, 8, order ); // rows: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + if ( !table_cook_ref( region, at + 16, (const void *) RowAt( ctx, value.pinned ), order ) ) { return false; } // pinned + return true; +} + +inline void ItemCookBody( uint8_t * at, const Item & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.count, 4, order ); +} + +inline void SquadRosterEntryCookBody( uint8_t * at, const SquadRosterEntry & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.key, 1, order ); + ItemCookBody( at + 4, value.value, order ); +} + +template inline bool SquadCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Squad & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // roster: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.name, 4, order ); + return true; +} + +template inline bool ArmyCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Army & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // squads: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.after, 4, order ); + return true; +} + +template inline bool DeckCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Deck & value, TableByteOrder order ) +{ + // all 3 slots: the storage is allocate-max, and a slot past the count rides as it lies (§7.2) + for ( int32_t i = 0; i < 3; i++ ) + { + if ( !RowCookBody( ctx, region, at + 0 + i * 24, value.hands[ i ], order ) ) { return false; } + } + table_cook_put( at + 72, (uint64_t) (uint32_t) value.hands_count, 4, order ); + table_cook_put( at + 76, (uint64_t) value.after, 4, order ); + return true; +} + +template inline bool SampleCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Sample & value, TableByteOrder order ); +template inline bool RowCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Row & value, TableByteOrder order ); +template inline bool SheetCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Sheet & value, TableByteOrder order ); +template inline bool ItemCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Item & value, TableByteOrder order ); +template inline bool SquadRosterEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const SquadRosterEntry & value, TableByteOrder order ); +template inline bool SquadCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Squad & value, TableByteOrder order ); +template inline bool ArmyCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Army & value, TableByteOrder order ); +template inline bool DeckCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Deck & value, TableByteOrder order ); + +// SampleCookExtent: Sample's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SampleCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Sample & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// RowCookExtent: Row's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool RowCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Row & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // items: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.items ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( Sample ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 4; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + SampleCookBody( array + i * 4, cursor[i], order ); + } + } + return true; +} + +// SheetCookExtent: Sheet's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SheetCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Sheet & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // rows: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.rows ); + if ( !cursor.ok ) { return false; } + at = ( at + 7 ) & ~(int64_t) 7; // at alignof( Row ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 24; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + if ( !RowCookBody( ctx, region, array + i * 24, cursor[i], order ) ) { return false; } + } + for ( int32_t i = 0; i < cursor.count; i++ ) // then, element by element in index order + { + if ( !RowCookExtent( ctx, region, extent, at, array + i * 24, cursor[i], order ) ) { return false; } + } + } + return true; +} + +// ItemCookExtent: Item's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool ItemCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Item & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// SquadRosterEntryCookExtent: SquadRosterEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SquadRosterEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const SquadRosterEntry & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// SquadCookExtent: Squad's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SquadCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Squad & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // roster + TableMapCursor cursor = TableMapOrder( ctx, value.roster ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( SquadRosterEntry ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 8; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) + { + SquadRosterEntryCookBody( array + i * 8, *cursor[i], order ); + } + TableMapRelease( cursor ); + } + return true; +} + +// ArmyCookExtent: Army's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool ArmyCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Army & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // squads: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.squads ); + if ( !cursor.ok ) { return false; } + at = ( at + 7 ) & ~(int64_t) 7; // at alignof( Squad ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 24; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + if ( !SquadCookBody( ctx, region, array + i * 24, cursor[i], order ) ) { return false; } + } + for ( int32_t i = 0; i < cursor.count; i++ ) // then, element by element in index order + { + if ( !SquadCookExtent( ctx, region, extent, at, array + i * 24, cursor[i], order ) ) { return false; } + } + } + return true; +} + +// DeckCookExtent: Deck's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool DeckCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Deck & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + for ( int32_t i = 0; i < ( value.hands_count < 3 ? value.hands_count : 3 ); i++ ) // hands + { + if ( !RowCookExtent( ctx, region, extent, at, record + 0 + i * 24, value.hands[i], order ) ) { return false; } + } + return true; +} + +// SampleCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool SampleCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Sample & value, TableByteOrder order ) +{ + SampleCookBody( at, value, order ); + int64_t extent_at = 0; + return SampleCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// RowCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool RowCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Row & value, TableByteOrder order ) +{ + if ( !RowCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return RowCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// SheetCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool SheetCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Sheet & value, TableByteOrder order ) +{ + if ( !SheetCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return SheetCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// ItemCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool ItemCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Item & value, TableByteOrder order ) +{ + ItemCookBody( at, value, order ); + int64_t extent_at = 0; + return ItemCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// SquadRosterEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool SquadRosterEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const SquadRosterEntry & value, TableByteOrder order ) +{ + SquadRosterEntryCookBody( at, value, order ); + int64_t extent_at = 0; + return SquadRosterEntryCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// SquadCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool SquadCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Squad & value, TableByteOrder order ) +{ + if ( !SquadCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return SquadCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// ArmyCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool ArmyCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Army & value, TableByteOrder order ) +{ + if ( !ArmyCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return ArmyCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// DeckCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool DeckCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Deck & value, TableByteOrder order ) +{ + if ( !DeckCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return DeckCookExtent( ctx, region, at + 80, extent_at, at, value, order ); +} + +// SampleCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Sample IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t SampleCookMeasure( const Sample & value ) +{ + (void) value; + return 88; // 64 header + 8 data + 16 attribution +} + +// SampleCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract SampleMeasure/SampleSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool SampleCook( const Sample & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) SampleCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 8, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + SampleCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 72, 0, 8, order ); + table_cook_put( raw + 80, 0xdc40d61254c70aa7ull, 8, order ); + return true; +} + +// RowCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool RowCookLayout( const Ctx & ctx, const Row & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = RowExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// RowCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t RowCookMeasureFrom( const Ctx & ctx, const Row & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( RowNumberFrom( ctx, numbering, root ) && RowCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// RowCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool RowCookFrom( const Ctx & ctx, const Row & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = RowNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && RowCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = RowCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0xa013e119fec906fbull, 8, order ); // the root: fnv1a64( "Row" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// RowCookMeasure / RowCook over a REGION root — a locked builder's AsConst, a +// region RowLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t RowCookMeasure( const Row * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return RowCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool RowCook( const Row * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return RowCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t RowCookMeasure( const RowBuilder & builder ) +{ + if ( builder.region != NULL ) { return RowCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return RowCookMeasureFrom( ctx, *(const Row *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool RowCook( const RowBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return RowCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return RowCookFrom( ctx, *(const Row *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// SheetCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool SheetCookLayout( const Ctx & ctx, const Sheet & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = SheetExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + case 0xa013e119fec906fbull: // Row + { + const int64_t extent = RowExtent( ctx, *(const Row *) numbering.entries[k].node ); + if ( extent < 0 ) { return false; } + size = 24 + extent; node_align = 8; + } + break; + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// SheetCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t SheetCookMeasureFrom( const Ctx & ctx, const Sheet & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( SheetNumberFrom( ctx, numbering, root ) && SheetCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// SheetCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool SheetCookFrom( const Ctx & ctx, const Sheet & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = SheetNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && SheetCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = SheetCookNode( ctx, region, region.base, root, order ); + for ( int64_t k = 0; ok && k < numbering.count; k++ ) + { + uint8_t * at = region.base + region.offsets[k + 1]; + const void * node = numbering.entries[k].node; + switch ( numbering.entries[k].type_id ) + { + case 0xa013e119fec906fbull: ok = RowCookNode( ctx, region, at, *(const Row *) node, order ); break; // Row + default: ok = false; break; + } + } + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0x0cc9e0af9a85fbc8ull, 8, order ); // the root: fnv1a64( "Sheet" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// SheetCookMeasure / SheetCook over a REGION root — a locked builder's AsConst, a +// region SheetLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t SheetCookMeasure( const Sheet * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SheetCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool SheetCook( const Sheet * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return SheetCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t SheetCookMeasure( const SheetBuilder & builder ) +{ + if ( builder.region != NULL ) { return SheetCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SheetCookMeasureFrom( ctx, *(const Sheet *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool SheetCook( const SheetBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return SheetCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SheetCookFrom( ctx, *(const Sheet *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ItemCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Item IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t ItemCookMeasure( const Item & value ) +{ + (void) value; + return 88; // 64 header + 8 data + 16 attribution +} + +// ItemCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract ItemMeasure/ItemSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool ItemCook( const Item & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) ItemCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 8, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + ItemCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 72, 0, 8, order ); + table_cook_put( raw + 80, 0x52cfa1d198476806ull, 8, order ); + return true; +} + +// SquadCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool SquadCookLayout( const Ctx & ctx, const Squad & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = SquadExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// SquadCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t SquadCookMeasureFrom( const Ctx & ctx, const Squad & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( SquadNumberFrom( ctx, numbering, root ) && SquadCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// SquadCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool SquadCookFrom( const Ctx & ctx, const Squad & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = SquadNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && SquadCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = SquadCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0xec07a2f760550a91ull, 8, order ); // the root: fnv1a64( "Squad" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// SquadCookMeasure / SquadCook over a REGION root — a locked builder's AsConst, a +// region SquadLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t SquadCookMeasure( const Squad * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SquadCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool SquadCook( const Squad * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return SquadCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t SquadCookMeasure( const SquadBuilder & builder ) +{ + if ( builder.region != NULL ) { return SquadCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SquadCookMeasureFrom( ctx, *(const Squad *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool SquadCook( const SquadBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return SquadCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SquadCookFrom( ctx, *(const Squad *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ArmyCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool ArmyCookLayout( const Ctx & ctx, const Army & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = ArmyExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// ArmyCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t ArmyCookMeasureFrom( const Ctx & ctx, const Army & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( ArmyNumberFrom( ctx, numbering, root ) && ArmyCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// ArmyCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool ArmyCookFrom( const Ctx & ctx, const Army & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = ArmyNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && ArmyCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = ArmyCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0x06e2378b553a9e84ull, 8, order ); // the root: fnv1a64( "Army" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// ArmyCookMeasure / ArmyCook over a REGION root — a locked builder's AsConst, a +// region ArmyLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t ArmyCookMeasure( const Army * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return ArmyCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool ArmyCook( const Army * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return ArmyCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t ArmyCookMeasure( const ArmyBuilder & builder ) +{ + if ( builder.region != NULL ) { return ArmyCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return ArmyCookMeasureFrom( ctx, *(const Army *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool ArmyCook( const ArmyBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return ArmyCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return ArmyCookFrom( ctx, *(const Army *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// DeckCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool DeckCookLayout( const Ctx & ctx, const Deck & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = DeckExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 80 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// DeckCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t DeckCookMeasureFrom( const Ctx & ctx, const Deck & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( DeckNumberFrom( ctx, numbering, root ) && DeckCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// DeckCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool DeckCookFrom( const Ctx & ctx, const Deck & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = DeckNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && DeckCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = DeckCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0xd043187343bfcfe8ull, 8, order ); // the root: fnv1a64( "Deck" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// DeckCookMeasure / DeckCook over a REGION root — a locked builder's AsConst, a +// region DeckLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t DeckCookMeasure( const Deck * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return DeckCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool DeckCook( const Deck * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return DeckCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t DeckCookMeasure( const DeckBuilder & builder ) +{ + if ( builder.region != NULL ) { return DeckCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return DeckCookMeasureFrom( ctx, *(const Deck *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool DeckCook( const DeckBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return DeckCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return DeckCookFrom( ctx, *(const Deck *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ---- relocatability, enforced: the wire is a pure length-prefixed +// stream AND the decoded storage is pointer-free — every closure type +// must stay trivially copyable and standard-layout, so instances can be +// memcpy'd, mmap'd, shared across processes, and walked through +// descriptor offsets. A failure here means a pointer, virtual or +// non-trivial member crept into generated storage. +// +// They ask the COMPILER ITSELF, which is what every C++ standard library +// answers the same two questions with — and it costs this header no +// include at all. +// A pointer FIELD is a TableRef — eight bytes and no address — so the +// property holds in BOTH forms: a fixed-size table is one relocatable +// struct, and a packed region is one relocatable block whose references +// are self-relative and therefore survive a plain memcpy. +static_assert( __is_trivially_copyable( Sample ), "Sample must stay relocatable" ); +static_assert( __is_standard_layout( Sample ), "Sample must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Row ), "Row must stay relocatable" ); +static_assert( __is_standard_layout( Row ), "Row must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Sheet ), "Sheet must stay relocatable" ); +static_assert( __is_standard_layout( Sheet ), "Sheet must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Item ), "Item must stay relocatable" ); +static_assert( __is_standard_layout( Item ), "Item must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( SquadRosterEntry ), "SquadRosterEntry must stay relocatable" ); +static_assert( __is_standard_layout( SquadRosterEntry ), "SquadRosterEntry must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Squad ), "Squad must stay relocatable" ); +static_assert( __is_standard_layout( Squad ), "Squad must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Army ), "Army must stay relocatable" ); +static_assert( __is_standard_layout( Army ), "Army must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Deck ), "Deck must stay relocatable" ); +static_assert( __is_standard_layout( Deck ), "Deck must stay standard-layout for offsetof" ); + +// ---- the cook's layout contract (docs/SPEC-TABLES.md §20.3) ---- +// +// The compiler derived every number below from the declaration and folded it +// into the BUILD VERSION; these asserts are this compiler saying whether it +// agrees. The model is not self-evidently right — on 32-bit System V +// alignof(uint64_t) is 4, not 8 — which is precisely why it is asserted +// rather than assumed. +static_assert( sizeof( Sample ) == 4, "Sample's sizeof moved: the build version was taken over 4, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Sample ) == 4, "Sample's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Sample, v ) == 0, "Sample's field v moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Row ) == 24, "Row's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Row ) == 8, "Row's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Row, items ) == 0, "Row's field items moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Row, label ) == 16, "Row's field label moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Sheet ) == 24, "Sheet's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Sheet ) == 8, "Sheet's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Sheet, rows ) == 0, "Sheet's field rows moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Sheet, pinned ) == 16, "Sheet's field pinned moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Item ) == 4, "Item's sizeof moved: the build version was taken over 4, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Item ) == 4, "Item's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Item, count ) == 0, "Item's field count moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( SquadRosterEntry ) == 8, "SquadRosterEntry's sizeof moved: the build version was taken over 8, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( SquadRosterEntry ) == 4, "SquadRosterEntry's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( SquadRosterEntry, key ) == 0, "SquadRosterEntry's field key moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( SquadRosterEntry, value ) == 4, "SquadRosterEntry's field value moved: the build version was taken over offset 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Squad ) == 24, "Squad's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Squad ) == 8, "Squad's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Squad, roster ) == 0, "Squad's field roster moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Squad, name ) == 16, "Squad's field name moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Army ) == 24, "Army's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Army ) == 8, "Army's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Army, squads ) == 0, "Army's field squads moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Army, after ) == 16, "Army's field after moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Deck ) == 80, "Deck's sizeof moved: the build version was taken over 80, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Deck ) == 8, "Deck's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Deck, hands ) == 0, "Deck's field hands moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Deck, after ) == 76, "Deck's field after moved: the build version was taken over offset 76 (docs/SPEC-TABLES.md §20.3)" ); + +static_assert( alignof( Sample ) <= kTableAlign, "Row.items: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( Row ) <= kTableAlign, "Sheet.rows: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( Squad ) <= kTableAlign, "Army.squads: an unbounded array's element alignment must fit the arena's" ); + +// ---- reflection descriptors (tables only, docs/SPEC-TABLES.md) ---- + +inline const TableTypeInfo * SampleTableType(); +inline const TableTypeInfo * RowTableType(); +inline const TableTypeInfo * SheetTableType(); +inline const TableTypeInfo * ItemTableType(); +inline const TableTypeInfo * SquadRosterEntryTableType(); +inline const TableTypeInfo * SquadTableType(); +inline const TableTypeInfo * ArmyTableType(); +inline const TableTypeInfo * DeckTableType(); +// The descriptors are CONSTANT-INITIALISED data, and a field's target is +// the ADDRESS of another descriptor. These declarations are what let a +// self- or mutually-referential graph — Node naming itself through *Node — +// be expressed as constant data instead of a lazy link, which could not +// have been written race-free OR recursion-safe. The whole reflection +// surface is therefore immutable: read it from any thread, any time. +extern const TableTypeInfo SampleTableInfo; +extern const TableTypeInfo RowTableInfo; +extern const TableTypeInfo SheetTableInfo; +extern const TableTypeInfo ItemTableInfo; +extern const TableTypeInfo SquadRosterEntryTableInfo; +extern const TableTypeInfo SquadTableInfo; +extern const TableTypeInfo ArmyTableInfo; +extern const TableTypeInfo DeckTableInfo; + +inline const TableFieldInfo SampleTableFields[] = { + { "v", "v", "int32", 0xaf63eb4c86020609ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Sample, v ), (uint32_t) sizeof( Sample::v ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo SampleTableInfo = { "Sample", (uint32_t) sizeof( Sample ), 1, SampleTableFields, +[]( void * p ) { SampleReset( *(Sample *) p ); }, false }; +inline const TableTypeInfo * SampleTableType() { return &SampleTableInfo; } + +inline const TableFieldInfo RowTableFields[] = { + { "items", "items", "Sample", 0x3e7884bf4f412c6full, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Row, items ), (uint32_t) sizeof( Sample ), (uint32_t) offsetof( Row, items.count ), 0xffffffffu, &SampleTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "label", "label", "int32", 0x39f7fcec8fcb623dull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Row, label ), (uint32_t) sizeof( Row::label ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo RowTableInfo = { "Row", (uint32_t) sizeof( Row ), 2, RowTableFields, +[]( void * p ) { RowReset( *(Row *) p ); }, true }; +inline const TableTypeInfo * RowTableType() { return &RowTableInfo; } + +inline const TableFieldInfo SheetTableFields[] = { + { "rows", "rows", "Row", 0xa3a7061ff10a8138ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Sheet, rows ), (uint32_t) sizeof( Row ), (uint32_t) offsetof( Sheet, rows.count ), 0xffffffffu, &RowTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "pinned", "pinned", "Row", 0x5f82477707ad620full, 17, false, true, []( const void * slot ) -> const void * { return (const void *) RowAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) RowEmplace( worker, *(TableRef *) slot ); }, false, false, 0, (uint32_t) offsetof( Sheet, pinned ), (uint32_t) sizeof( TableRef ), 0xffffffffu, 0xffffffffu, &RowTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo SheetTableInfo = { "Sheet", (uint32_t) sizeof( Sheet ), 2, SheetTableFields, +[]( void * p ) { SheetReset( *(Sheet *) p ); }, true }; +inline const TableTypeInfo * SheetTableType() { return &SheetTableInfo; } + +inline const TableFieldInfo ItemTableFields[] = { + { "count", "count", "int32", 0xb1e5e28e4479a274ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Item, count ), (uint32_t) sizeof( Item::count ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo ItemTableInfo = { "Item", (uint32_t) sizeof( Item ), 1, ItemTableFields, +[]( void * p ) { ItemReset( *(Item *) p ); }, false }; +inline const TableTypeInfo * ItemTableType() { return &ItemTableInfo; } + +inline const TableFieldInfo SquadRosterEntryTableFields[] = { + { "key", "key", "uint8", 0x3dc94a19365b10ecull, 6, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( SquadRosterEntry, key ), (uint32_t) sizeof( SquadRosterEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( SquadRosterEntry, value ), (uint32_t) sizeof( SquadRosterEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo SquadRosterEntryTableInfo = { "SquadRosterEntry", (uint32_t) sizeof( SquadRosterEntry ), 2, SquadRosterEntryTableFields, +[]( void * p ) { SquadRosterEntryReset( *(SquadRosterEntry *) p ); }, false }; +inline const TableTypeInfo * SquadRosterEntryTableType() { return &SquadRosterEntryTableInfo; } + +inline const TableFieldInfo SquadTableFields[] = { + { "roster", "roster", "map[uint8]Item", 0x1c84390d304f4f42ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Squad, roster ), (uint32_t) sizeof( SquadRosterEntry ), (uint32_t) offsetof( Squad, roster.count ), 0xffffffffu, &SquadRosterEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { SquadRosterEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint8_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint8_t) key_value ); } return (void *) placed; }, "" }, + { "name", "name", "int32", 0xc4bcadba8e631b86ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Squad, name ), (uint32_t) sizeof( Squad::name ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo SquadTableInfo = { "Squad", (uint32_t) sizeof( Squad ), 2, SquadTableFields, +[]( void * p ) { SquadReset( *(Squad *) p ); }, true }; +inline const TableTypeInfo * SquadTableType() { return &SquadTableInfo; } + +inline const TableFieldInfo ArmyTableFields[] = { + { "squads", "squads", "Squad", 0x7848019b0c02a926ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Army, squads ), (uint32_t) sizeof( Squad ), (uint32_t) offsetof( Army, squads.count ), 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Army, after ), (uint32_t) sizeof( Army::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo ArmyTableInfo = { "Army", (uint32_t) sizeof( Army ), 2, ArmyTableFields, +[]( void * p ) { ArmyReset( *(Army *) p ); }, true }; +inline const TableTypeInfo * ArmyTableType() { return &ArmyTableInfo; } + +inline const TableFieldInfo DeckTableFields[] = { + { "hands", "hands", "Row", 0x81b46a69304ee2c9ull, 13, true, false, NULL, NULL, true, false, 3, (uint32_t) offsetof( Deck, hands ), (uint32_t) sizeof( Deck::hands[0] ), (uint32_t) offsetof( Deck, hands_count ), 0xffffffffu, &RowTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Deck, after ), (uint32_t) sizeof( Deck::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo DeckTableInfo = { "Deck", (uint32_t) sizeof( Deck ), 2, DeckTableFields, +[]( void * p ) { DeckReset( *(Deck *) p ); }, true }; +inline const TableTypeInfo * DeckTableType() { return &DeckTableInfo; } + +// ---- the text form (docs/SPEC-TABLES.md §16) ---- + +// Sample in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// HoldersTable.cpp; link it to use them. +bool SampleFromJson( Sample & value, const char * text, int64_t bytes, TableReport * report ); +int64_t SampleToJsonMeasure( const Sample & value ); +int64_t SampleToJson( const Sample & value, char * buffer, int64_t capacity ); + +// Row in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in HoldersTable.cpp; link it to use them. +bool RowFromJson( RowBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t RowToJsonMeasure( const Row * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t RowToJson( const Row * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Sheet in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in HoldersTable.cpp; link it to use them. +bool SheetFromJson( SheetBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t SheetToJsonMeasure( const Sheet * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t SheetToJson( const Sheet * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Item in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// HoldersTable.cpp; link it to use them. +bool ItemFromJson( Item & value, const char * text, int64_t bytes, TableReport * report ); +int64_t ItemToJsonMeasure( const Item & value ); +int64_t ItemToJson( const Item & value, char * buffer, int64_t capacity ); + +// Squad in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in HoldersTable.cpp; link it to use them. +bool SquadFromJson( SquadBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t SquadToJsonMeasure( const Squad * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t SquadToJson( const Squad * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Army in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in HoldersTable.cpp; link it to use them. +bool ArmyFromJson( ArmyBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t ArmyToJsonMeasure( const Army * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t ArmyToJson( const Army * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Deck in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in HoldersTable.cpp; link it to use them. +bool DeckFromJson( DeckBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t DeckToJsonMeasure( const Deck * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t DeckToJson( const Deck * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/MigrateTable.cpp b/testdata/golden/tables/lists/MigrateTable.cpp new file mode 100644 index 000000000..bd509a04b --- /dev/null +++ b/testdata/golden/tables/lists/MigrateTable.cpp @@ -0,0 +1,3149 @@ +// Code generated by the schema compiler from Migrate.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — the TABLE wire's text form (docs/SPEC-TABLES.md §16). +// Compile this file to use FromJson / ToJson; a project that +// never reads or writes a text does not compile it and pays nothing. + +#include "MigrateTable.h" + +#include // the text form: number formatting +#include // the text form: exact number conversion +#include // the text form: the runtime's decimal point + +// The guard is not vestigial. Several listdemo Table.cpp files may be +// concatenated into ONE translation unit — a unity build — and without it +// each would redefine the walk. It is also why the walk's functions may be +// weak (vague linkage) across separate objects: ODR requires their +// definitions to be token-identical, and the generic-walk gate is what +// proves that, byte for byte, across every generated .cpp. +#ifndef LISTDEMO_SCHEMA_TABLE_JSON +#define LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +// ---- the pointer adapters (docs/SPEC-TABLES.md §16.7) ---- +// +// The walk below is ONE walk, byte-identical in every generated .cpp, and a +// pointer is the one kind it cannot walk alone: reading one needs the +// builder's arena and writing one needs a region's deref, and neither exists +// in a unit that declares no pointer. So the walk calls these three and does +// not define them. A unit with no pointer defines them as stubs no field ever +// reaches; a pointered unit defines them in the graph half that follows the +// walk. + +struct TableJsonIn; +struct TableJsonOut; + +// a pointer field's object, or the `&node` reference standing in for it, into +// the slot; the cursor is on the opening brace +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// the node a pointer slot names, in place — or as `&node` when it is shared +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// the FIRST key of an object the walk is skipping begins with `&`: the cursor is +// on its value. A dropped definition still takes its label (§16.7); a fixed reader +// skips the value whole, as it skips everything else it does not place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); + +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- +// +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map +inline bool TableJsonIsMap( const TableFieldInfo * f ); +// the map as a plain JSON object keyed by the KEY, in ASCENDING key order +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that object back into the slot, in whatever order the text gives it +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); + +// ---- json walk: begin ---- +// +// The TEXT form (docs/SPEC-TABLES.md §16): one table, one text, one walk over the +// reflection descriptors (§8). Reading fills ONE caller-owned instance and +// allocates nothing beyond it; writing targets a caller buffer with the +// wire's measure/write symmetry. Everything AROUND this — which file goes +// with which instance, what key an instance is filed under, how instances +// link into a root table's collections — is a packer's opinion and stays +// with the tool that holds it. +// +// The dialect: trailing commas are accepted on read (the authoring files +// this exists for carry them) and never written; comments are not JSON and +// are refused; unknown keys are skipped and counted; a duplicate key is +// last-wins and counted; a key present with the wrong JSON type is skipped +// and counted, never coerced. + +static const int32_t kTableJsonMaxDepth = 128; + +// A key longer than this cannot name a field, so it is skipped as unknown. +static const int32_t kTableJsonMaxKey = 256; + +// The longest numeric token the walk will convert. Anything longer is a +// value no field can hold and counts as a kind mismatch. +static const int32_t kTableJsonMaxNumber = 512; + +// The decimal point the C runtime is CURRENTLY using. Number conversion is +// the one locale-sensitive corner of the grammar — JSON's point is always +// '.', the runtime's is whatever the program set — so every number crosses +// this one character on the way out and on the way back in. Nothing else in +// the walk consults the locale. +inline char TableJsonDecimalPoint() +{ + const struct lconv * conv = localeconv(); + if ( conv != NULL && conv->decimal_point != NULL && conv->decimal_point[0] != 0 ) + { + return conv->decimal_point[0]; + } + return '.'; +} + +// ---- storage access: the descriptors give an offset and a width, and the +// ---- storage is the HOST's, so every load and store goes through a width +// ---- switch rather than a memcpy into the low bytes of a wider word + +// finite: not a NaN, not an infinity. Written without — the walk's +// runtime surface stays the handful of functions it already names. +// A vocabulary entry the descriptor could not spell. The generated name +// functions answer "???" for a value outside the declared set, and that is +// not a name — writing it would put a spelling in the text that the reader +// then counts as unknown, turning a refusal into a silent loss. +inline bool TableJsonNamed( const char * name ) +{ + return name != NULL && strcmp( name, "???" ) != 0; +} + +inline bool TableJsonFinite( double v ) +{ + return v == v && v <= 1.7976931348623157e308 && v >= -1.7976931348623157e308; +} + +inline uint64_t TableJsonGetRaw( const void * storage, uint32_t width ) +{ + switch ( width ) + { + case 1: { uint8_t v = 0; memcpy( &v, storage, 1 ); return v; } + case 2: { uint16_t v = 0; memcpy( &v, storage, 2 ); return v; } + case 4: { uint32_t v = 0; memcpy( &v, storage, 4 ); return v; } + case 8: { uint64_t v = 0; memcpy( &v, storage, 8 ); return v; } + } + return 0; +} + +inline void TableJsonSetRaw( void * storage, uint32_t width, uint64_t value ) +{ + switch ( width ) + { + case 1: { uint8_t v = (uint8_t) value; memcpy( storage, &v, 1 ); break; } + case 2: { uint16_t v = (uint16_t) value; memcpy( storage, &v, 2 ); break; } + case 4: { uint32_t v = (uint32_t) value; memcpy( storage, &v, 4 ); break; } + case 8: { uint64_t v = value; memcpy( storage, &v, 8 ); break; } + } +} + +inline int64_t TableJsonGetSigned( const void * storage, uint32_t width ) +{ + uint64_t raw = TableJsonGetRaw( storage, width ); + if ( width < 8 ) + { + uint64_t sign = uint64_t( 1 ) << ( width * 8 - 1 ); + if ( ( raw & sign ) != 0 ) + { + raw |= ~( ( sign << 1 ) - 1 ); + } + } + return (int64_t) raw; +} + +// ---- the WIDE kinds (docs/SPEC-TABLES.md §3, §16.2) ---- +// +// The 128-bit integers and the fixed-point family convert EXACTLY, over two +// 64-bit lanes: a 128-bit integer is a decimal integer, a fixed value a +// decimal in WHOLE UNITS (1.0, -0.25, 3.0000152587890625) and nothing +// on either path passes through a double. Nothing here needs a 128-bit type +// either, which is what keeps this walk one text for every unit. +struct TableJsonWide +{ + uint64_t lo; + uint64_t hi; +}; + +inline bool TableJsonKindWide( uint8_t kind ) { return kind >= 18 && kind <= 29; } +inline bool TableJsonKindWideSigned( uint8_t kind ) { return kind == 18 || ( kind >= 20 && kind <= 24 ); } +inline bool TableJsonKindFixed( uint8_t kind ) { return kind >= 20 && kind <= 29; } + +inline bool TableJsonWideZero( TableJsonWide v ) { return v.lo == 0 && v.hi == 0; } +inline bool TableJsonWideNegative( TableJsonWide v ) { return ( v.hi >> 63 ) != 0; } + +inline int TableJsonWideCompare( TableJsonWide a, TableJsonWide b, bool is_signed ) +{ + if ( is_signed && TableJsonWideNegative( a ) != TableJsonWideNegative( b ) ) { return TableJsonWideNegative( a ) ? -1 : 1; } + if ( a.hi != b.hi ) { return a.hi < b.hi ? -1 : 1; } + if ( a.lo != b.lo ) { return a.lo < b.lo ? -1 : 1; } + return 0; +} + +inline TableJsonWide TableJsonWideShl( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.hi = v.lo << ( n - 64 ); return r; } + r.hi = ( v.hi << n ) | ( v.lo >> ( 64 - n ) ); + r.lo = v.lo << n; + return r; +} + +inline TableJsonWide TableJsonWideShr( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.lo = v.hi >> ( n - 64 ); return r; } + r.lo = ( v.lo >> n ) | ( v.hi << ( 64 - n ) ); + r.hi = v.hi >> n; + return r; +} + +inline TableJsonWide TableJsonWideNeg( TableJsonWide v ) +{ + TableJsonWide r; + r.lo = ~v.lo + 1; + r.hi = ~v.hi + ( r.lo == 0 ? 1 : 0 ); + return r; +} + +// v = v * m + a; the return is the carry out of 128 bits +inline uint32_t TableJsonWideMulAdd( TableJsonWide * v, uint32_t m, uint32_t a ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t carry = a; + for ( int i = 0; i < 4; i++ ) + { + uint64_t p = limb[i] * m + carry; + limb[i] = p & 0xffffffffull; + carry = p >> 32; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) carry; +} + +// v = v / d; the return is the remainder +inline uint32_t TableJsonWideDiv( TableJsonWide * v, uint32_t d ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t rem = 0; + for ( int i = 3; i >= 0; i-- ) + { + uint64_t cur = ( rem << 32 ) | limb[i]; + limb[i] = cur / d; + rem = cur % d; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) rem; +} + +// The storage of a wide kind, as lanes. A sixteen-byte storage is serialize's +// pair — native __int128 in the host's byte order, or the emulated struct with +// its low lane first — so the lanes are read in the host's order; a narrower +// storage is one lane, sign-extended for a signed kind. +inline TableJsonWide TableJsonWideLoad( const void * storage, uint32_t width, bool is_signed ) +{ + TableJsonWide v = { 0, 0 }; + if ( width == 16 ) + { + uint64_t half[2]; + memcpy( half, storage, 16 ); + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + v.lo = little ? half[0] : half[1]; + v.hi = little ? half[1] : half[0]; + return v; + } + v.lo = is_signed ? (uint64_t) TableJsonGetSigned( storage, width ) : TableJsonGetRaw( storage, width ); + v.hi = ( is_signed && ( v.lo >> 63 ) != 0 ) ? ~uint64_t( 0 ) : 0; + return v; +} + +inline void TableJsonWideStore( void * storage, uint32_t width, TableJsonWide v ) +{ + if ( width == 16 ) + { + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + uint64_t half[2]; + half[0] = little ? v.lo : v.hi; + half[1] = little ? v.hi : v.lo; + memcpy( storage, half, 16 ); + return; + } + TableJsonSetRaw( storage, width, v.lo ); +} + +// a counted field's companion: a string's length, a bytes' length, a counted +// array's count. Bounded by the declared extent on the way out, so a storage +// invariant a caller broke cannot walk off the end of the array. +inline int32_t TableJsonCount( const void * base, const TableFieldInfo * f ) +{ + if ( !f->counted ) + { + return f->array_bound; + } + int32_t count = 0; + memcpy( &count, (const uint8_t *) base + f->count_offset, sizeof( count ) ); + if ( count < 0 ) { count = 0; } + if ( count > f->array_bound ) { count = f->array_bound; } + return count; +} + +inline void TableJsonSetCount( void * base, const TableFieldInfo * f, int32_t count ) +{ + if ( f->counted ) + { + memcpy( (uint8_t *) base + f->count_offset, &count, sizeof( count ) ); + } +} + +// ---- what a field's kind expects to see in the text ---- +// +// One classifier, consulted by both directions, so a reader and a writer can +// never disagree about a kind's JSON form. 'o' object, 'a' array, 's' +// string, 'n' number, 'b' boolean. +// +// A vocabulary field is spelled by NAME: an enum is one name, a flags mask +// is the array of the names of its set bits. The two are told apart by the +// id column — an enum variant rides under a wire id, a flags BIT never does +// (docs/SPEC-TABLES.md §4), so a name function with no id function is flags. +// +// bytes(N) is the one kind whose element kind does not decide its form: it +// shares u8 with a plain array of u8, and rides as base64. The schema type +// name settles it, and "bytes" is a keyword no declaration can claim. +inline bool TableJsonIsBytes( const TableFieldInfo * f ) +{ + return f->is_array && f->kind == 6 && strcmp( f->type_name, "bytes" ) == 0; +} + +// An ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): its JSON form is an OBJECT +// keyed by variant name, not a positional array, because that is what the +// storage is — one slot per variant, addressed by the variant. +inline bool TableJsonIsKeyed( const TableFieldInfo * f ) +{ + return f->key_name != NULL; +} + +// THE KEY A STORAGE SLOT HOLDS (§2.4, §8): the storage shifts left, so slot i +// holds the key i + 1 and nothing is stored for None. This is the ONE place +// the walker spells the shift. +inline uint64_t TableJsonKeyedSlotKey( int64_t slot ) +{ + return (uint64_t) ( slot + 1 ); +} + +// A slot whose key names a variant of the keying enum. Every slot in +// [0, array_bound) does, unless the enum carries max-headroom variants outside +// a table closure, where a reserved value names nothing and its key id is 0 — +// the reserved id no declared name can fold to (§5). +inline bool TableJsonKeyedSlotValid( const TableFieldInfo * f, int64_t slot ) +{ + return f->key_id( TableJsonKeyedSlotKey( slot ) ) != 0; +} + +inline bool TableJsonIsFlags( const TableFieldInfo * f ) +{ + return f->enum_name != NULL && f->variant_id == NULL; +} + +inline bool TableJsonIsEnum( const TableFieldInfo * f ) +{ + return f->variant_id != NULL && f->arms == NULL; +} + +inline char TableJsonShape( const TableFieldInfo * f ) +{ + if ( TableJsonIsMap( f ) ) return 'o'; // a MAP: an object keyed by the KEY (§2.8) + if ( f->kind == 12 ) return 's'; // string + if ( TableJsonIsBytes( f ) ) return 's'; // bytes: base64 + if ( TableJsonIsKeyed( f ) ) return 'o'; // an object keyed by variant NAME + if ( f->is_array ) return 'a'; + if ( f->arms != NULL ) return 'o'; // union: an object with ONE key + if ( f->kind == 13 ) return 'o'; // nested table or type + if ( f->kind == 17 ) return f->table != NULL ? 'o' : 's'; // a pointer: the pointee's object in place, or null (§16.7); a byte buffer's string (§2.5) + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// the ELEMENT shape of an array field — the same classifier one level down +inline char TableJsonElementShape( const TableFieldInfo * f ) +{ + if ( f->arms != NULL ) return 'o'; // an element of an array of unions: one key, the arm (§2.6) + if ( f->kind == 13 ) return 'o'; + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// A guarded group rides only when its guard reads true — the wire's own +// elision (§4), carried into the text so a text and a wire written from one +// instance say the same thing. The guard is spelled as its branch condition +// over bool fields of the SAME type ("at_rest", "!at_rest", +// "active && has_target"), so evaluating it is a walk of the same +// descriptor. Nothing is inferred in the other direction: reading places +// every key it can name, and the guard is a plain bool key (§16.2). +inline bool TableJsonGuardHolds( const void * base, const TableTypeInfo * info, const char * guard ) +{ + const char * p = guard; + for ( ;; ) + { + while ( *p == ' ' || *p == '&' ) { p++; } + if ( *p == 0 ) { return true; } + bool want = true; + if ( *p == '!' ) { want = false; p++; } + const char * start = p; + while ( *p != 0 && *p != ' ' && *p != '&' ) { p++; } + size_t length = (size_t) ( p - start ); + bool value = false; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( strlen( f->name ) == length && strncmp( f->name, start, length ) == 0 ) + { + value = TableJsonGetRaw( (const uint8_t *) base + f->offset, f->elem_size ) != 0; + break; + } + } + if ( value != want ) { return false; } + } +} + +// ---- writing ---- + +// The writer sink MEASURES when the buffer is NULL and WRITES when it is +// not, over one code path — so measure and write agree byte for byte, the +// wire's invariant (§9) carried across. +struct TableJsonOut +{ + char * buffer; + int64_t capacity; + int64_t offset; + bool overflow; + void * graph; // the pointered write's identity map (§16.7); NULL for a fixed table + + void raw( const char * data, int64_t count ) + { + if ( buffer != NULL ) + { + if ( offset + count > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) count ); + } + offset += count; + } + void put( char c ) { raw( &c, 1 ); } + void text( const char * s ) { raw( s, (int64_t) strlen( s ) ); } + void line( int32_t depth ) + { + put( '\n' ); + for ( int32_t i = 0; i < depth; i++ ) { raw( " ", 2 ); } + } +}; + +inline const char * TableJsonBase64Alphabet() +{ + return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; +} + +inline void TableJsonWriteBase64( TableJsonOut & out, const uint8_t * data, int32_t length ) +{ + const char * alphabet = TableJsonBase64Alphabet(); + out.put( '"' ); + int32_t i = 0; + for ( ; i + 3 <= length; i += 3 ) + { + uint32_t triple = ( uint32_t( data[i] ) << 16 ) | ( uint32_t( data[i+1] ) << 8 ) | uint32_t( data[i+2] ); + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], + alphabet[ ( triple >> 6 ) & 0x3f ], alphabet[ triple & 0x3f ] }; + out.raw( quad, 4 ); + } + if ( i < length ) + { + int32_t left = length - i; + uint32_t triple = uint32_t( data[i] ) << 16; + if ( left == 2 ) { triple |= uint32_t( data[i+1] ) << 8; } + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], '=', '=' }; + if ( left == 2 ) { quad[2] = alphabet[ ( triple >> 6 ) & 0x3f ]; } + out.raw( quad, 4 ); + } + out.put( '"' ); +} + +// One UTF-8 sequence at s, or -1 when the bytes there are not one. Rejects +// the lot: a stray continuation, an overlong form, a surrogate half, and +// anything past U+10FFFF. +inline int32_t TableJsonUtf8( const char * s, int32_t remaining, int32_t * width ) +{ + unsigned char lead = (unsigned char) s[0]; + int32_t want = 0; + int32_t code = 0; + if ( lead < 0x80 ) { *width = 1; return lead; } + else if ( lead >= 0xc2 && lead <= 0xdf ) { want = 2; code = lead & 0x1f; } + else if ( lead >= 0xe0 && lead <= 0xef ) { want = 3; code = lead & 0x0f; } + else if ( lead >= 0xf0 && lead <= 0xf4 ) { want = 4; code = lead & 0x07; } + else { return -1; } + if ( remaining < want ) { return -1; } + for ( int32_t i = 1; i < want; i++ ) + { + unsigned char next = (unsigned char) s[i]; + if ( ( next & 0xc0 ) != 0x80 ) { return -1; } + code = ( code << 6 ) | ( next & 0x3f ); + } + if ( want == 3 && code < 0x800 ) { return -1; } // overlong + if ( want == 4 && code < 0x10000 ) { return -1; } // overlong + if ( code >= 0xd800 && code <= 0xdfff ) { return -1; } // a surrogate half + if ( code > 0x10ffff ) { return -1; } + *width = want; + return code; +} + +// A JSON text MUST be valid UTF-8 (RFC 8259 §8.1). The read path is +// byte-transparent — the wire imposes no encoding (§3) and a string may hold +// anything — so the WRITER is where that obligation is met: a byte that is +// not part of a well-formed sequence is written as U+FFFD, one per bad byte, +// and never raw. A text this walk writes is therefore readable by any +// conforming parser, which a raw byte would not be. The cost is stated +// plainly: for a string holding invalid UTF-8, the round trip is NOT +// byte-identical, because the alternative is emitting a text that is not +// JSON. +inline void TableJsonWriteString( TableJsonOut & out, const char * s, int32_t length ) +{ + static const char hex[] = "0123456789abcdef"; + out.put( '"' ); + for ( int32_t i = 0; i < length; i++ ) + { + unsigned char c = (unsigned char) s[i]; + switch ( c ) + { + case '"': out.raw( "\\\"", 2 ); break; + case '\\': out.raw( "\\\\", 2 ); break; + case '\b': out.raw( "\\b", 2 ); break; + case '\f': out.raw( "\\f", 2 ); break; + case '\n': out.raw( "\\n", 2 ); break; + case '\r': out.raw( "\\r", 2 ); break; + case '\t': out.raw( "\\t", 2 ); break; + default: + if ( c < 0x20 ) + { + char escape[6] = { '\\', 'u', '0', '0', hex[ c >> 4 ], hex[ c & 0xf ] }; + out.raw( escape, 6 ); + } + else if ( c < 0x80 ) + { + out.put( (char) c ); + } + else + { + int32_t width = 0; + if ( TableJsonUtf8( s + i, length - i, &width ) < 0 ) + { + out.raw( "\xef\xbf\xbd", 3 ); // U+FFFD, one per bad byte + } + else + { + out.raw( s + i, width ); + i += width - 1; + } + } + break; + } + } + out.put( '"' ); +} + +inline void TableJsonWriteUnsigned( TableJsonOut & out, uint64_t value ) +{ + char digits[24]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) ( value % 10 ) ); + value /= 10; + } while ( value != 0 ); + char text[24]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); +} + +inline void TableJsonWriteSigned( TableJsonOut & out, int64_t value ) +{ + if ( value < 0 ) + { + out.put( '-' ); + TableJsonWriteUnsigned( out, uint64_t( 0 ) - (uint64_t) value ); + return; + } + TableJsonWriteUnsigned( out, (uint64_t) value ); +} + +// A wide kind writes its raw storage as §16.2's text: a 128-bit integer as a +// decimal integer; a fixed value in WHOLE UNITS as the shortest exact decimal +// with at least one fractional digit (1.0, -0.25), the spelling the schema text +// gives a fixed default. The fraction terminates because a dyadic fraction has +// a finite decimal expansion — at most F digits. +inline void TableJsonWriteWide( TableJsonOut & out, const void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + TableJsonWide v = TableJsonWideLoad( storage, f->elem_size, is_signed ); + if ( is_signed && TableJsonWideNegative( v ) ) + { + out.put( '-' ); + v = TableJsonWideNeg( v ); + } + int frac = f->frac_bits; + TableJsonWide whole = TableJsonWideShr( v, frac ); + char digits[40]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) TableJsonWideDiv( &whole, 10 ) ); + } while ( !TableJsonWideZero( whole ) ); + char text[40]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); + if ( !TableJsonKindFixed( f->kind ) ) { return; } + out.put( '.' ); + // the fraction bits alone: v with everything at and above bit F cleared + TableJsonWide fraction = v; + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + if ( frac == 0 ) { fraction.lo = 0; } + if ( TableJsonWideZero( fraction ) ) + { + out.put( '0' ); + return; + } + while ( !TableJsonWideZero( fraction ) ) + { + // ×10: the digit is what lands at and above bit F, including the + // carry out of 128 bits when F leaves no room for it below + uint32_t carry = TableJsonWideMulAdd( &fraction, 10, 0 ); + uint64_t digit = TableJsonWideShr( fraction, frac ).lo; + if ( frac > 64 ) { digit |= uint64_t( carry ) << ( 128 - frac ); } + out.put( (char) ( '0' + (int) digit ) ); + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + } +} + +// A float writes at the SHORTEST precision that reads back as the same value +// at the field's own width, so a round trip is exact and a text stays +// readable. Non-finite values have no JSON spelling at all, and the writer +// REFUSES rather than losing one silently — the same rule measure and save +// already apply to an enum value no variant names (§5). +inline bool TableJsonWriteFloat( TableJsonOut & out, double value, bool single ) +{ + if ( !TableJsonFinite( value ) ) { return false; } + char text[64]; + int low = single ? 6 : 15; + int high = single ? 9 : 17; + int length = 0; + for ( int digits = low; ; digits++ ) + { + length = snprintf( text, sizeof( text ), "%.*g", digits, value ); + if ( length <= 0 || length >= (int) sizeof( text ) ) { return false; } + if ( digits >= high ) { break; } + // the round-trip check runs BEFORE the decimal point is normalised: + // the token still carries whatever point snprintf just produced + if ( single ) + { + if ( (double) strtof( text, NULL ) == value ) { break; } + } + else + { + if ( strtod( text, NULL ) == value ) { break; } + } + } + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int i = 0; i < length; i++ ) + { + if ( text[i] == point ) { text[i] = '.'; } + } + } + out.raw( text, length ); + return true; +} + +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration writes through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ); + +// one scalar, at one storage address: a nested object, a union, a +// vocabulary, or a number +inline bool TableJsonWriteScalar( TableJsonOut & out, const void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; None is {} + const TableUnionInfo * arms = f->arms(); + uint64_t tag = TableJsonGetRaw( (const uint8_t *) storage + arms->tag_offset, arms->tag_size ); + if ( tag == 0 ) + { + out.raw( "{}", 2 ); + return true; + } + if ( (int64_t) tag > f->enum_max ) + { + return false; // a tag no arm names, exactly as measure refuses it + } + const char * arm = f->enum_name( tag ); + // and refuse on the NAME, not merely on the bound: §16.2 says a value + // no variant NAMES is refused, so the check is the name. Writing + // whatever came back would emit "???", a spelling the reader counts + // as unknown — a silent round-trip loss in place of a refusal. + if ( !TableJsonNamed( arm ) ) { return false; } + out.put( '{' ); + out.line( depth + 1 ); + TableJsonWriteString( out, arm, (int32_t) strlen( arm ) ); + out.raw( ": ", 2 ); + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2): an arm that names + // no declaration carries the FIELD descriptor a field of its type + // would carry, offsets taken inside the union storage (§2.6), so the + // value walks through the field writer one key down. + if ( arms->arms[tag].field != NULL ) + { + if ( !TableJsonWriteField( out, storage, arms->arms[tag].field, depth + 1 ) ) + { + return false; + } + } + else if ( arms->arms[tag].table == NULL ) + { + out.raw( "null", 4 ); // a payload-free arm: the name selects it (§2.6) + } + else if ( !TableJsonWriteValue( out, (const uint8_t *) storage + arms->arms[tag].offset, arms->arms[tag].table, depth + 1 ) ) + { + return false; + } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->kind == 13 ) + { + return TableJsonWriteValue( out, storage, f->table, depth ); + } + if ( TableJsonIsEnum( f ) ) + { + uint64_t value = TableJsonGetRaw( storage, f->elem_size ); + // a value no variant names has no text spelling, exactly as it has no + // wire identity: the writer REFUSES rather than writing None over it, + // the rule measure and save already apply (docs/SPEC-TABLES.md §5) + if ( (int64_t) value > f->enum_max ) { return false; } + if ( value != 0 && f->variant_id( value ) == 0 ) { return false; } + const char * name = f->enum_name( value ); + if ( !TableJsonNamed( name ) ) { return false; } + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + return true; + } + if ( TableJsonIsFlags( f ) ) + { + uint64_t bits = TableJsonGetRaw( storage, f->elem_size ); + if ( bits == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + bool first = true; + for ( int64_t bit = 0; bit < 64; bit++ ) + { + if ( ( bits & ( uint64_t( 1 ) << bit ) ) == 0 ) { continue; } + if ( bit > f->enum_max ) + { + return false; // a bit no variant names has no text spelling + } + const char * name = f->enum_name( (uint64_t) bit ); + if ( !TableJsonNamed( name ) ) { return false; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + } + out.line( depth ); + out.put( ']' ); + return true; + } + switch ( f->kind ) + { + case 1: + out.text( TableJsonGetRaw( storage, f->elem_size ) != 0 ? "true" : "false" ); + return true; + case 10: + { + float v = 0.0f; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, (double) v, true ); + } + case 11: + { + double v = 0.0; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, v, false ); + } + case 2: case 3: case 4: case 5: + TableJsonWriteSigned( out, TableJsonGetSigned( storage, f->elem_size ) ); + return true; + default: + if ( TableJsonKindWide( f->kind ) ) + { + TableJsonWriteWide( out, storage, f ); + return true; + } + TableJsonWriteUnsigned( out, TableJsonGetRaw( storage, f->elem_size ) ); + return true; + } +} + +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ) +{ + const uint8_t * storage = (const uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonWriteMap( out, (const void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } + if ( f->kind == 17 && !f->is_array ) + { + return TableJsonWritePointer( out, storage, f, depth ); + } + if ( f->kind == 17 ) + { + // an ARRAY OF POINTERS (§2.1): the pointer row per element — the + // pointee's object in place, null, or `&node` for a shared one (§16.7) + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWritePointer( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; + } + if ( f->kind == 12 ) + { + TableJsonWriteString( out, (const char *) storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + TableJsonWriteBase64( out, storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + // one entry per SLOT, keyed by the variant that owns it, so inserting + // a variant next season moves nothing in the text either. Slot i holds + // the key i + 1: nothing is stored for None, so nothing is written for it. + out.put( '{' ); + bool first = true; + for ( int64_t slot = 0; slot < f->array_bound; slot++ ) + { + if ( !TableJsonKeyedSlotValid( f, slot ) ) { continue; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + const char * key = f->key_name( TableJsonKeyedSlotKey( slot ) ); + TableJsonWriteString( out, key, (int32_t) strlen( key ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteScalar( out, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + if ( first ) { out.raw( "}", 1 ); return true; } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->is_array ) + { + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWriteScalar( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + out.line( depth ); + out.put( ']' ); + return true; + } + return TableJsonWriteScalar( out, storage, f, depth ); +} + +// One instance's fields, in DECLARATION ORDER, defaults included — a text is +// for people and tools, and a text that elides is a text a reader has to know +// the schema to complete. `any` says whether the object is already open on +// entry — a shared node's `&node` opens it before the fields (§16.7) — and +// whether it is open on return. +inline bool TableJsonWriteFields( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth, bool & any ) +{ + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( f->guard[0] != 0 && !TableJsonGuardHolds( base, info, f->guard ) ) { continue; } + // an ABSENT optional writes no key: presence of the key IS the + // presence (§16.2), so an absent field is an absent key and nothing + // else would read back as absent + if ( f->optional && + TableJsonGetRaw( (const uint8_t *) base + f->present_offset, 1 ) == 0 ) + { + continue; + } + if ( !any ) { out.put( '{' ); } + else { out.put( ',' ); } + any = true; + out.line( depth + 1 ); + TableJsonWriteString( out, f->json, (int32_t) strlen( f->json ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, base, f, depth + 1 ) ) { return false; } + } + return true; +} + +// One instance as one object. The writer carries the reader's depth cap +// (§16.2): a pointer chain nests as deep as it is long (§16.7), and a text the +// writer produced past the cap would be a text the reader refuses. +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { return false; } + bool any = false; + if ( !TableJsonWriteFields( out, base, info, depth, any ) ) { return false; } + if ( !any ) + { + out.raw( "{}", 2 ); + return true; + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- reading ---- + +struct TableJsonIn +{ + const char * text; + int64_t size; + int64_t pos; + TableReport * report; + bool bad; // the text is not JSON: the walk stops and keeps what it placed + void * graph; // the pointered read's builder and label map (§16.7); NULL for a fixed table +}; + +inline void TableJsonSpace( TableJsonIn & in ) +{ + while ( in.pos < in.size ) + { + char c = in.text[in.pos]; + if ( c == ' ' || c == '\t' || c == '\n' || c == '\r' ) { in.pos++; continue; } + // comments are not JSON, and a walk that guessed at one would be + // reading a dialect nobody wrote down + if ( c == '/' ) { in.bad = true; } + return; + } +} + +inline char TableJsonPeek( TableJsonIn & in ) +{ + TableJsonSpace( in ); + return in.pos < in.size ? in.text[in.pos] : 0; +} + +// the shape of the value sitting at the cursor, without consuming it +inline char TableJsonValueShape( TableJsonIn & in ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return 'o'; + case '[': return 'a'; + case '"': return 's'; + case 't': case 'f': return 'b'; + case 'n': return 'z'; + case 0: return 0; + default: return 'n'; + } +} + +inline bool TableJsonLiteral( TableJsonIn & in, const char * word ) +{ + int64_t length = (int64_t) strlen( word ); + if ( in.pos + length > in.size || memcmp( in.text + in.pos, word, (size_t) length ) != 0 ) + { + in.bad = true; + return false; + } + in.pos += length; + return true; +} + +// one \uXXXX escape body; -1 when the four hex digits are not there +inline int TableJsonHex4( TableJsonIn & in ) +{ + if ( in.pos + 4 > in.size ) { return -1; } + int value = 0; + for ( int i = 0; i < 4; i++ ) + { + char c = in.text[in.pos + i]; + int digit; + if ( c >= '0' && c <= '9' ) { digit = c - '0'; } + else if ( c >= 'a' && c <= 'f' ) { digit = c - 'a' + 10; } + else if ( c >= 'A' && c <= 'F' ) { digit = c - 'A' + 10; } + else { return -1; } + value = ( value << 4 ) | digit; + } + in.pos += 4; + return value; +} + +inline int32_t TableJsonEncodeUtf8( uint32_t code, char * unit ) +{ + if ( code < 0x80 ) { unit[0] = (char) code; return 1; } + if ( code < 0x800 ) + { + unit[0] = (char) ( 0xc0 | ( code >> 6 ) ); + unit[1] = (char) ( 0x80 | ( code & 0x3f ) ); + return 2; + } + if ( code < 0x10000 ) + { + unit[0] = (char) ( 0xe0 | ( code >> 12 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( code & 0x3f ) ); + return 3; + } + unit[0] = (char) ( 0xf0 | ( code >> 18 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 12 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[3] = (char) ( 0x80 | ( code & 0x3f ) ); + return 4; +} + +// Scan one JSON string into a caller buffer. Bytes are appended ONE CODE +// POINT AT A TIME — an escape's encoding, or a UTF-8 sequence read whole — +// so a string longer than the field is clamped AT A CODE POINT BOUNDARY and +// never cut through a multi-byte character. Clamping is counted, never +// fatal, exactly as it is on the wire (§4). A NULL destination scans past a +// string without keeping it. +inline bool TableJsonScanString( TableJsonIn & in, char * out, int32_t capacity, int32_t * length ) +{ + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + int32_t placed = 0; + bool clamped = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos]; + if ( c == '"' ) { in.pos++; break; } + char unit[4]; + int32_t unit_length = 0; + if ( c == '\\' ) + { + in.pos++; + if ( in.pos >= in.size ) { in.bad = true; return false; } + char escape = in.text[in.pos++]; + switch ( escape ) + { + case '"': unit[0] = '"'; unit_length = 1; break; + case '\\': unit[0] = '\\'; unit_length = 1; break; + case '/': unit[0] = '/'; unit_length = 1; break; + case 'b': unit[0] = '\b'; unit_length = 1; break; + case 'f': unit[0] = '\f'; unit_length = 1; break; + case 'n': unit[0] = '\n'; unit_length = 1; break; + case 'r': unit[0] = '\r'; unit_length = 1; break; + case 't': unit[0] = '\t'; unit_length = 1; break; + case 'u': + { + int high = TableJsonHex4( in ); + if ( high < 0 ) { in.bad = true; return false; } + uint32_t code = (uint32_t) high; + if ( high >= 0xd800 && high <= 0xdbff && in.pos + 2 <= in.size && + in.text[in.pos] == '\\' && in.text[in.pos + 1] == 'u' ) + { + int64_t mark = in.pos; + in.pos += 2; + int low = TableJsonHex4( in ); + if ( low >= 0xdc00 && low <= 0xdfff ) + { + code = 0x10000 + ( ( (uint32_t) high - 0xd800 ) << 10 ) + ( (uint32_t) low - 0xdc00 ); + } + else + { + in.pos = mark; // a lone lead surrogate rides as itself + } + } + // a surrogate half that never found its partner has no + // UTF-8 encoding: encoding it anyway would manufacture + // CESU-8 — invalid UTF-8 — out of input that was valid + // JSON, so it reads as the replacement character + if ( code >= 0xd800 && code <= 0xdfff ) { code = 0xfffd; } + unit_length = TableJsonEncodeUtf8( code, unit ); + break; + } + default: in.bad = true; return false; + } + } + else if ( (unsigned char) c < 0x20 ) + { + in.bad = true; // a raw control character is not a JSON string body + return false; + } + else + { + // a UTF-8 sequence read WHOLE, so the clamp below can only land + // between code points. Only bytes that ACTUALLY look like + // continuations are taken: the wire imposes no encoding (§3), so + // a string may legitimately hold a stray lead byte, and one at + // the end of a text must not swallow the closing quote. + unsigned char lead = (unsigned char) c; + int32_t want = 1; + if ( ( lead & 0xe0 ) == 0xc0 ) { want = 2; } + else if ( ( lead & 0xf0 ) == 0xe0 ) { want = 3; } + else if ( ( lead & 0xf8 ) == 0xf0 ) { want = 4; } + unit[0] = c; + in.pos++; + unit_length = 1; + while ( unit_length < want && in.pos < in.size && + ( (unsigned char) in.text[in.pos] & 0xc0 ) == 0x80 ) + { + unit[unit_length++] = in.text[in.pos++]; + } + } + if ( out == NULL ) + { + placed += unit_length; // measured and not kept: a byte buffer's read sizes its node this way (§2.5) + } + else if ( placed + unit_length <= capacity ) + { + memcpy( out + placed, unit, (size_t) unit_length ); + placed += unit_length; + } + else + { + clamped = true; + } + } + if ( clamped ) { in.report->clamped++; } + if ( length != NULL ) { *length = placed; } + return true; +} + +// the numeric token at the cursor, copied out whole; false = not a number +// Scan one number, to JSON's OWN grammar (RFC 8259 §6) and not to a run of +// number-ish characters: +// +// number = [ "-" ] int [ frac ] [ exp ] +// int = "0" / ( digit1-9 *digit ) +// frac = "." 1*digit +// exp = ( "e" / "E" ) [ "-" / "+" ] 1*digit +// +// Scanning the production is what makes a typo in an authoring file a +// DIAGNOSTIC rather than a value: "1-2" scans as 1 and leaves "-2" where the +// object expects a comma, so the text is malformed — which is what §16.2 +// already promises. A permissive scan would hand "1-2" to a digit loop and +// report a clamp, and a config pipeline would never hear about it. Leading +// "+", leading zeros, ".5" and "3." are not JSON either. +inline bool TableJsonWalkNumber( TableJsonIn & in, bool * integral ) +{ + TableJsonSpace( in ); + bool whole = true; + if ( in.pos < in.size && in.text[in.pos] == '-' ) { in.pos++; } + // int: a lone zero, or a non-zero digit and any digits after it + if ( in.pos >= in.size ) { return false; } + if ( in.text[in.pos] == '0' ) + { + in.pos++; + } + else if ( in.text[in.pos] >= '1' && in.text[in.pos] <= '9' ) + { + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + } + else + { + return false; + } + // frac + if ( in.pos < in.size && in.text[in.pos] == '.' ) + { + in.pos++; + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + // exp + if ( in.pos < in.size && ( in.text[in.pos] == 'e' || in.text[in.pos] == 'E' ) ) + { + in.pos++; + if ( in.pos < in.size && ( in.text[in.pos] == '-' || in.text[in.pos] == '+' ) ) { in.pos++; } + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + *integral = whole; + return true; +} + +// the same production, with the token kept for conversion +inline bool TableJsonScanNumber( TableJsonIn & in, char * token, int32_t capacity, int32_t * length, bool * integral ) +{ + TableJsonSpace( in ); + int64_t start = in.pos; + if ( !TableJsonWalkNumber( in, integral ) ) { return false; } + int64_t count = in.pos - start; + if ( count <= 0 || count >= capacity ) { return false; } + memcpy( token, in.text + start, (size_t) count ); + token[count] = 0; + *length = (int32_t) count; + return true; +} + +// the token's exact double, through the runtime's own converter — which +// speaks the LOCALE's decimal point, so the token crosses back over that +// character on its way in +inline double TableJsonTokenDouble( const char * token, int32_t length, bool single ) +{ + char work[kTableJsonMaxNumber]; + memcpy( work, token, (size_t) length ); + work[length] = 0; + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int32_t i = 0; i < length; i++ ) + { + if ( work[i] == '.' ) { work[i] = point; } + } + } + if ( single ) { return (double) strtof( work, NULL ); } + return strtod( work, NULL ); +} + +// the token's exact integer, parsed digit by digit so no width and no +// locale can move it. Saturation is reported as a clamp, the wire's rule for +// a value outside what the reader can hold (§4). +inline int64_t TableJsonTokenInteger( const char * token, int32_t length, bool is_signed, bool * saturated ) +{ + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) + { + negative = token[i] == '-'; + i++; + } + uint64_t magnitude = 0; + bool over = false; + for ( ; i < length; i++ ) + { + uint64_t digit = (uint64_t) ( token[i] - '0' ); + if ( magnitude > ( UINT64_MAX - digit ) / 10 ) { over = true; break; } + magnitude = magnitude * 10 + digit; + } + if ( !is_signed ) + { + // -0 IS zero, and clamping it would report an event that did not + // happen; only a real negative magnitude is out of range here + if ( negative ) { *saturated = magnitude != 0; return 0; } + if ( over ) { *saturated = true; return (int64_t) UINT64_MAX; } + *saturated = false; + return (int64_t) magnitude; + } + if ( negative ) + { + if ( over || magnitude > ( uint64_t( 1 ) << 63 ) ) { *saturated = true; return INT64_MIN; } + *saturated = false; + if ( magnitude == ( uint64_t( 1 ) << 63 ) ) { return INT64_MIN; } + return -(int64_t) magnitude; + } + if ( over || magnitude > (uint64_t) INT64_MAX ) { *saturated = true; return INT64_MAX; } + *saturated = false; + return (int64_t) magnitude; +} + +// A number token into a wide kind's raw storage (docs/SPEC-TABLES.md §16.2). A +// 128-bit integer takes any token whose VALUE is integral; a fixed field any +// token whose value is EXACTLY representable in its Q I.F — a finer fraction +// is the wrong shape for the field, counted as a kind mismatch and never +// rounded, the rule SPEC.md §4.6 gives a fixed default. A magnitude past 128 +// bits saturates and counts as a clamp, as an int64 field saturates at +// INT64_MAX; the declared range clamps after it, on the RAW scale, as it does +// for every bounded scalar. +// +// The token is normalized to its digits with the decimal point after "point" +// of them. An integer part past 40 digits is above 2^128 whatever the digits +// are, and a value below 10^-40 is finer than 2^-127, the finest fraction any +// F can spell — so outside that band the answer is known without the +// arithmetic, and a token spelling 1e999999999 costs nothing to refuse. +inline bool TableJsonReadWide( TableJsonIn & in, const char * token, int32_t length, void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + int frac = f->frac_bits; + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { negative = token[i] == '-'; i++; } + const char * int_digits = token + i; + int32_t int_len = 0; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { int_len++; i++; } + const char * frac_digits = token + i; + int32_t frac_len = 0; + if ( i < length && token[i] == '.' ) + { + i++; + frac_digits = token + i; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { frac_len++; i++; } + } + int64_t exp = 0; + if ( i < length && ( token[i] == 'e' || token[i] == 'E' ) ) + { + i++; + bool exp_negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { exp_negative = token[i] == '-'; i++; } + while ( i < length && token[i] >= '0' && token[i] <= '9' ) + { + if ( exp < 100000 ) { exp = exp * 10 + ( token[i] - '0' ); } + i++; + } + if ( exp_negative ) { exp = -exp; } + } + // the digits, with the point after "point" of them; leading and trailing + // zeros stripped. digit( k ) reads the k-th of the int and frac runs. + int32_t start = 0, end = int_len + frac_len; + int64_t point = int_len + exp; + while ( start < end && ( start < int_len ? int_digits[start] : frac_digits[start - int_len] ) == '0' ) { start++; point--; } + while ( end > start && ( end - 1 < int_len ? int_digits[end - 1] : frac_digits[end - 1 - int_len] ) == '0' ) { end--; } + + TableJsonWide raw = { 0, 0 }; + bool saturated = false; + TableJsonWide signed_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) >> 1 }; + TableJsonWide signed_min = { 0, uint64_t( 1 ) << 63 }; + TableJsonWide unsigned_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) }; + if ( start == end ) + { + // zero, and -0 IS zero + } + else if ( point > 40 ) + { + saturated = true; + if ( !negative ) { raw = is_signed ? signed_max : unsigned_max; } + else if ( is_signed ) { raw = signed_min; } + } + else if ( point < -40 ) + { + in.report->kind_mismatch++; // finer than any F can spell + return true; + } + else + { + // the fraction FIRST, so an inexact value is the wrong shape whatever + // its magnitude: its digits, with the zeros a negative point puts in + // front, doubled F times; each doubling's carry is the next bit, and + // the value is exact iff nothing is left after the last one + char fd[kTableJsonMaxNumber + 48]; + int32_t fn = 0; + for ( int64_t z = point; z < 0; z++ ) { fd[fn++] = 0; } + for ( int32_t k = (int32_t) ( point > 0 ? point : 0 ) + start; k < end; k++ ) + { + fd[fn++] = (char) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ); + } + TableJsonWide fraction = { 0, 0 }; + for ( int b = 0; b < frac; b++ ) + { + int carry = 0; + for ( int32_t k = fn - 1; k >= 0; k-- ) + { + int d = fd[k] * 2 + carry; + fd[k] = (char) ( d % 10 ); + carry = d / 10; + } + fraction = TableJsonWideShl( fraction, 1 ); + fraction.lo |= (uint64_t) carry; + } + for ( int32_t k = 0; k < fn; k++ ) + { + if ( fd[k] != 0 ) + { + in.report->kind_mismatch++; + return true; + } + } + // then the whole part, saturating past 128 bits + TableJsonWide whole = { 0, 0 }; + for ( int64_t k = start; k < start + point && !saturated; k++ ) + { + uint32_t digit = k < end ? (uint32_t) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ) : 0; + if ( TableJsonWideMulAdd( &whole, 10, digit ) != 0 ) { saturated = true; } + } + if ( !saturated && frac > 0 && !TableJsonWideZero( TableJsonWideShr( whole, 128 - frac ) ) ) { saturated = true; } + if ( !saturated ) + { + raw = TableJsonWideShl( whole, frac ); + raw.lo |= fraction.lo; + raw.hi |= fraction.hi; + } + if ( is_signed ) + { + if ( !saturated && !negative && TableJsonWideNegative( raw ) ) { saturated = true; } + if ( !saturated && negative && TableJsonWideCompare( raw, signed_min, false ) > 0 ) { saturated = true; } + if ( saturated ) { raw = negative ? signed_min : signed_max; } + else if ( negative ) { raw = TableJsonWideNeg( raw ); } + } + else + { + if ( saturated ) { raw = unsigned_max; } + if ( negative && !TableJsonWideZero( raw ) ) { raw.lo = 0; raw.hi = 0; saturated = true; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->wide != NULL ) + { + TableJsonWide lo = { f->wide->lo[0], f->wide->lo[1] }; + TableJsonWide hi = { f->wide->hi[0], f->wide->hi[1] }; + if ( TableJsonWideCompare( raw, lo, is_signed ) < 0 ) { raw = lo; in.report->clamped++; } + else if ( TableJsonWideCompare( raw, hi, is_signed ) > 0 ) { raw = hi; in.report->clamped++; } + } + TableJsonWideStore( storage, f->elem_size, raw ); + return true; +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ); + +inline bool TableJsonSkipContainer( TableJsonIn & in, char close, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; // the opening bracket + bool first = true; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == close ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + if ( close == '}' ) + { + // the key is kept, because a skipped OBJECT may still be a + // pointer's: an `&node` opening it names a node the storage could + // not hold, and the numbering has to survive the drop (§16.7). + // Anywhere but first, the prefix is the reserved key out of place + // — in a pointered unit; a fixed unit skips the value whole. + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( key[0] == '&' && in.graph != NULL ) + { + if ( !first ) { in.report->malformed = true; in.bad = true; return false; } + if ( !TableJsonSkippedAmpersand( in, key, depth ) ) { return false; } + first = false; + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } + } + first = false; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return TableJsonSkipContainer( in, '}', depth ); + case '[': return TableJsonSkipContainer( in, ']', depth ); + case '"': return TableJsonScanString( in, NULL, 0, NULL ); + case 't': return TableJsonLiteral( in, "true" ); + case 'f': return TableJsonLiteral( in, "false" ); + case 'n': return TableJsonLiteral( in, "null" ); + case 0: in.bad = true; return false; + default: + { + // consumed, never converted: skipping needs no buffer, and this + // is the one walk a hostile text drives to the depth cap. It is + // the SAME production the value path scans, so an unknown key + // cannot smuggle past a number a named key would refuse. + bool integral = false; + if ( !TableJsonWalkNumber( in, &integral ) ) { in.bad = true; return false; } + return true; + } + } +} + +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration reads through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ); + +// place one scalar at one storage address +inline bool TableJsonReadScalar( TableJsonIn & in, void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; {} is None, and + // two keys is a text this walk will not guess at + const TableUnionInfo * arms = f->arms(); + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, 0 ); + if ( TableJsonPeek( in ) == '}' ) { in.pos++; return true; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t tag = 0; + for ( int64_t t = 1; t <= f->enum_max; t++ ) + { + if ( strcmp( f->enum_name( (uint64_t) t ), key ) == 0 ) { tag = t; break; } + } + if ( tag == 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + void * payload = (uint8_t *) storage + arms->arms[tag].offset; + const TableFieldInfo * arm = arms->arms[tag].field; + bool placed = true; + if ( arm != NULL ) + { + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2). A value of + // the wrong shape for that row is a KIND MISMATCH: the union + // reads None, the event is counted, and the enclosing object + // continues — the rule a FIELD's value lives under, one key + // down. A pointer arm's null is a null pointer, not a shape + // error, exactly as a pointer field's is (§16.7). + char got = TableJsonValueShape( in ); + if ( arm->kind == 17 && !arm->is_array && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + memset( payload, 0, (size_t) arms->arms[tag].size ); + } + else if ( got != TableJsonShape( arm ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( arm->kind == 17 && !arm->is_array ) + { + // A POINTER ARM'S VALUE IS THE POINTEE IN PLACE, or a + // node reference to one (§16.7) — the read a pointer + // FIELD takes, which is not the scalar walk + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadPointer( in, payload, arm, depth + 1 ) ) { return false; } + } + else + { + // SELECTION ZERO-ESTABLISHES THE ARM (SPEC §5): an arm + // takes no specified default, so zero is the establish + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadField( in, storage, arm, depth + 1 ) ) { return false; } + } + } + else if ( arms->arms[tag].table != NULL ) + { + if ( TableJsonValueShape( in ) != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else + { + arms->arms[tag].table->reset( payload ); + if ( !TableJsonReadTable( in, payload, arms->arms[tag].table, depth + 1 ) ) { return false; } + } + } + else + { + // A PAYLOAD-FREE ARM'S VALUE IS null (§2.6): the arm name + // selects it and there is nothing to place + if ( TableJsonValueShape( in ) != 'z' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( !TableJsonLiteral( in, "null" ) ) + { + return false; + } + } + if ( placed ) + { + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, (uint64_t) tag ); + } + } + char c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; // a second key: a one-of with two arms is not a value + return false; + } + if ( f->kind == 13 ) + { + f->table->reset( storage ); + return TableJsonReadTable( in, storage, f->table, depth + 1 ); + } + if ( TableJsonIsEnum( f ) ) + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + for ( int64_t v = 0; v <= f->enum_max; v++ ) + { + if ( strcmp( f->enum_name( (uint64_t) v ), name ) == 0 ) + { + TableJsonSetRaw( storage, f->elem_size, (uint64_t) v ); + return true; + } + } + // a name this build cannot name reads as None and counts as unknown, + // exactly as an unknown variant id does on the wire (§4) + TableJsonSetRaw( storage, f->elem_size, 0 ); + in.report->unknown++; + return true; + } + if ( TableJsonIsFlags( f ) ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + uint64_t bits = 0; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( c != '"' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + bool found = false; + for ( int64_t bit = 0; bit <= f->enum_max; bit++ ) + { + if ( strcmp( f->enum_name( (uint64_t) bit ), name ) == 0 ) + { + bits |= uint64_t( 1 ) << bit; + found = true; + break; + } + } + if ( !found ) { in.report->unknown++; } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + TableJsonSetRaw( storage, f->elem_size, bits ); + return true; + } + if ( f->kind == 1 ) + { + char c = TableJsonPeek( in ); + if ( c == 't' ) { if ( !TableJsonLiteral( in, "true" ) ) { return false; } TableJsonSetRaw( storage, f->elem_size, 1 ); return true; } + if ( !TableJsonLiteral( in, "false" ) ) { return false; } + TableJsonSetRaw( storage, f->elem_size, 0 ); + return true; + } + char token[kTableJsonMaxNumber]; + int32_t length = 0; + bool integral = false; + if ( !TableJsonScanNumber( in, token, kTableJsonMaxNumber, &length, &integral ) ) + { + in.bad = true; + return false; + } + if ( TableJsonKindWide( f->kind ) ) + { + return TableJsonReadWide( in, token, length, storage, f ); + } + if ( f->kind == 10 || f->kind == 11 ) + { + bool single = f->kind == 10; + double value = TableJsonTokenDouble( token, length, single ); + // A magnitude the field's format cannot hold is the WRONG SHAPE for + // the kind, and it never reaches storage: 1e400 is not a float64 and + // 1e300 is not a float32. Storing the infinity the conversion + // produced would leave an instance this walk called CLEAN that + // ToJsonMeasure then refuses forever (a non-finite float has no JSON + // spelling), and §16.1's one invariant is that a text which reads + // clean writes back. + if ( !TableJsonFinite( value ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( f->has_range ) + { + if ( value < f->range_min ) { value = f->range_min; in.report->clamped++; } + else if ( value > f->range_max ) { value = f->range_max; in.report->clamped++; } + } + if ( single ) + { + float narrow = (float) value; + if ( !TableJsonFinite( (double) narrow ) ) + { + in.report->kind_mismatch++; + return true; + } + memcpy( storage, &narrow, sizeof( narrow ) ); + } + else + { + memcpy( storage, &value, sizeof( value ) ); + } + return true; + } + // JSON HAS ONE NUMBER TYPE. 2.0 IS the integer 2 and 1e3 IS 1000, and a + // library that round-trips numbers through a double emits them that way — + // this walker's own float writer emits 1e+21. So an integer field takes + // any number whose VALUE is integral, however it was spelled; only a + // genuinely fractional value is the wrong shape for it. + bool is_signed = f->kind >= 2 && f->kind <= 5; + bool saturated = false; + int64_t value = 0; + if ( integral ) + { + value = TableJsonTokenInteger( token, length, is_signed, &saturated ); + } + else + { + double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( is_signed ) + { + if ( d >= 9223372036854775808.0 ) { value = INT64_MAX; saturated = true; } + else if ( d < -9223372036854775808.0 ) { value = INT64_MIN; saturated = true; } + else if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) d; } + } + else + { + if ( d < 0.0 ) + { + // a negative for an unsigned field clamps to zero, as the + // exact digit path already does + if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + value = 0; + saturated = true; + } + else if ( d >= 18446744073709551616.0 ) { value = (int64_t) UINT64_MAX; saturated = true; } + else if ( d != (double) (uint64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) (uint64_t) d; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->has_range ) + { + if ( (double) value < f->range_min ) { value = (int64_t) f->range_min; in.report->clamped++; } + else if ( (double) value > f->range_max ) { value = (int64_t) f->range_max; in.report->clamped++; } + } + // the field's own storage width is the last bound: a value past it + // clamps rather than wrapping, which is what the wire does too + if ( f->elem_size < 8 ) + { + if ( is_signed ) + { + int64_t high = ( int64_t( 1 ) << ( f->elem_size * 8 - 1 ) ) - 1; + int64_t low = -high - 1; + if ( value > high ) { value = high; in.report->clamped++; } + else if ( value < low ) { value = low; in.report->clamped++; } + } + else + { + uint64_t high = ( uint64_t( 1 ) << ( f->elem_size * 8 ) ) - 1; + if ( value < 0 ) { value = 0; in.report->clamped++; } + else if ( (uint64_t) value > high ) { value = (int64_t) high; in.report->clamped++; } + } + } + // at eight bytes the storage IS the parser's width, and an unsigned value + // past INT64_MAX rides here as a negative int64 by design — the token + // parser already turned a NEGATIVE token for an unsigned field into a + // clamped zero, so there is nothing left to bound. + TableJsonSetRaw( storage, f->elem_size, (uint64_t) value ); + return true; +} + +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ) +{ + uint8_t * storage = (uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonReadMap( in, (void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + + if ( f->kind == 12 ) + { + int32_t length = 0; + if ( !TableJsonScanString( in, (char *) storage, f->array_bound, &length ) ) { return false; } + storage[length] = 0; + TableJsonSetCount( base, f, length ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + // base64 decodes STRAIGHT INTO the field's storage, six bits at a + // time — no window, no temporary, so a bytes(N) of any declared + // extent reads the same way. A base64 body carries no escapes, so a + // backslash in one is simply not an alphabet character. + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + memset( storage, 0, (size_t) f->array_bound ); + TableJsonSetCount( base, f, 0 ); + const char * alphabet = TableJsonBase64Alphabet(); + int32_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + bool clamped = false; + bool malformed = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + const char * at = c != 0 ? strchr( alphabet, c ) : NULL; + if ( at == NULL ) { malformed = true; continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( at - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < f->array_bound ) + { + storage[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); + } + else + { + clamped = true; + } + } + } + if ( malformed ) + { + // a body that is not base64 is the wrong shape for the kind: the + // field keeps its default and the event is counted + in.report->kind_mismatch++; + return true; + } + if ( clamped ) { in.report->clamped++; } + TableJsonSetCount( base, f, placed ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + // every slot back to its declared defaults first, so a key the text + // omits keeps them and a repeated field key cannot leave an earlier + // occurrence's slots standing + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + void * slot = storage + (int64_t) i * f->elem_size; + if ( f->kind == 13 ) { f->table->reset( slot ); } + else { memset( slot, 0, (size_t) f->elem_size ); } + } + char shape = TableJsonElementShape( f ); + // A KEYED OBJECT'S KEYS ARE KEYS: a variant named twice is a duplicate + // key like any other, last-wins and counted (§16.2). Tracked the way + // a table's own field keys are — a bounded, allocation-free bitmask; + // a vocabulary wider than this still reads, its repeats simply stop + // being counted. + uint64_t seen[8] = {}; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t slot = -1; + for ( int64_t v = 0; v < f->array_bound; v++ ) + { + // nothing is stored for None, so "None" finds no slot and is + // an unknown key like any other name this reader cannot place + if ( !TableJsonKeyedSlotValid( f, v ) ) { continue; } + if ( strcmp( f->key_name( TableJsonKeyedSlotKey( v ) ), key ) == 0 ) { slot = v; break; } + } + if ( slot >= 0 && slot < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( slot & 63 ); + if ( ( seen[slot >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[slot >> 6] |= bit; + } + if ( slot < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; + } + if ( f->is_array ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + // LAST WINS has to be true of a repeated ARRAY key too, and it is + // wire-visible: a fixed array writes every slot, so a second, shorter + // occurrence overlaying a prefix would leave the first occurrence's + // tail standing. The field goes back to its declared defaults before + // this occurrence's elements are placed — the re-establishment a nested + // table and a union arm already get. A table element's defaults are + // its own (the reset hook); every other element kind's storage + // default is zero, which is what the generated array declares. + if ( f->kind == 13 ) + { + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + f->table->reset( storage + (int64_t) i * f->elem_size ); + } + } + else + { + memset( storage, 0, (size_t) f->array_bound * (size_t) f->elem_size ); + } + TableJsonSetCount( base, f, 0 ); + int32_t placed = 0; + char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( placed >= f->array_bound ) + { + // more elements than the reader's bound: the bounded prefix + // is kept and the excess counts, the wire's rule (§4) + in.report->clamped++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( f->kind == 17 ) + { + // an element of an ARRAY OF POINTERS (§2.1): null is a null slot, an + // object is the pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( storage + (int64_t) placed * f->elem_size, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed++; + } + else + { + if ( !TableJsonReadScalar( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + // a fixed array's tail keeps the defaults the prefill left there, + // exactly as a short wire count does + TableJsonSetCount( base, f, placed ); + return true; + } + return TableJsonReadScalar( in, storage, f, depth ); +} + +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ); + +// ONE table object: keys are field keys, unknown ones are skipped and +// counted, a repeated key is last-wins and counted. The instance is already +// at its declared defaults when this is entered, so a key the text never +// mentions keeps the default an absent field takes on the wire (§4). +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + return TableJsonReadTableKeys( in, base, info, depth, NULL ); +} + +// The keys of an object whose brace is already consumed. A pointer's object +// opens the same way a table's does, but its FIRST key may be `&node` (§16.7) +// and the adapter that reads it has to scan the key to know — so it hands the +// key it scanned in as `first_key`, with the colon consumed, and this places +// it before scanning the rest. +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ) +{ + // duplicate tracking, bounded and allocation-free: a table with more + // fields than this still reads, its repeats simply stop being counted + uint64_t seen[8] = {}; + for ( ;; ) + { + char key[kTableJsonMaxKey]; + char c = 0; + if ( first_key != NULL ) + { + memcpy( key, first_key, strlen( first_key ) + 1 ); // scanned into a buffer this size by the caller + first_key = NULL; + } + else + { + c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + } + int32_t index = -1; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + if ( strcmp( info->fields[i].json, key ) == 0 ) { index = i; break; } + } + if ( key[0] == '&' ) + { + // THE AMPERSAND PREFIX IS RESERVED TO THE FORM (docs/SPEC-TABLES.md + // §16.7). No declaration may take a key beginning with it, so this + // is never a field this build lacks — it is the sharing construct + // somewhere it cannot stand: `&node` is the FIRST key of a pointer's + // object and nothing else, and the adapter that reads a pointer + // has consumed it before these keys are read. MALFORMED, refused + // and counted; never counted as unknown, never skipped. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( index < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + const TableFieldInfo * f = &info->fields[index]; + if ( index < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( index & 63 ); + if ( ( seen[index >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[index >> 6] |= bit; + } + // PRESENCE OF THE KEY IS THE PRESENCE (§16.2): reaching this line + // is the key being present, so an optional is set present + // whatever its value — with one exception the page names: a JSON + // null, which reads as ABSENT rather than as a value. + char got = TableJsonValueShape( in ); + if ( f->kind == 17 && !f->is_array ) + { + // a pointer: null is a null pointer, an object is the pointee + // in place or an `&node` reference to one (§16.7), a string is + // a BYTE BUFFER's bytes (§2.5), and anything else is the wrong + // shape for the kind + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) base + f->offset, f->elem_size, 0 ); + } + else if ( got != TableJsonShape( f ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) base + f->offset, f, depth ) ) + { + return false; + } + } + else if ( f->optional && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + // absent, and back at its defaults: a repeated key whose last + // occurrence is null must not leave an earlier value standing + if ( f->table != NULL ) { f->table->reset( (uint8_t *) base + f->offset ); } + else { memset( (uint8_t *) base + f->offset, 0, (size_t) f->elem_size ); } + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 0 ); + } + else + { + if ( got != TableJsonShape( f ) ) + { + // the wrong JSON type for the kind: skipped, never coerced + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, base, f, depth ) ) + { + return false; + } + if ( f->optional ) + { + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 1 ); + } + } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +// ---- the two entry points the per-table wrappers name ---- + +inline bool TableJsonRead( void * value, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = NULL; + info->reset( value ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, value, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +inline int64_t TableJsonWrite( const void * value, const TableTypeInfo * info, char * buffer, int64_t capacity ) +{ + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = NULL; + if ( !TableJsonWriteValue( out, value, info, 0 ) ) { return -1; } + // THE CANONICAL TEXT ENDS WITH EXACTLY ONE NEWLINE (docs/SPEC-TABLES.md + // §16.1). Every writer emits it — this walk, the C# walk and + // "schema unpack" — and every reader accepts a text with or without one, + // because the trailing whitespace a read already skips is what makes the + // two the same text. It is a byte of the FORM rather than a file + // convention: a text that is written to a file, pasted into a diff and + // handed back through a pipe has to be one text in all three places, and a + // buffer whose last byte is a closing brace is the one shape that is not. + out.put( '\n' ); + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json walk: end ---- + +// ---- json graph walk: begin ---- +// +// THE VARIABLE CLASS's half of the text form (docs/SPEC-TABLES.md §16.7). The +// walk above places every kind but one; this defines the three adapters it +// calls for that one, and the two entry points a pointered table's wrappers +// name. The text is the fixed class's — a pointee is an object in place — and a +// node named more than once carries `&node`: defined once, with its fields, +// and referenced after by `{ "&node": N }` alone. + +// ---- the identity map ---- +// +// ONE map shape serves both directions. Writing keys it by a node's ADDRESS and +// counts the slots that name the node, so the second pass knows at a node's +// first occurrence whether it will be named again; reading keys it by the +// text's own label and answers the node it defined. Open addressing, a +// multiply-shift hash and quadrupling growth — TablePackMap's shape (§6.2), on +// the same terms: proportional to nodes, never to bytes, on the authoring +// side, and released before the call returns. + +struct TableJsonGraphEntry +{ + uint64_t key; // a node's address (write) or a label (read); 0 is an empty slot + int64_t count; // write: how many slots name this node + int64_t label; // write: the `&node` label assigned at its first write, 0 until then + uint8_t open; // the descent is still open: a reference here is a cycle (write), a self-reference (read) + uint32_t node; // read: the node's arena offset; 0 for a definition the reader dropped + const TableTypeInfo * type; // read: the node's table; NULL for a dropped one +}; + +struct TableJsonGraphMap +{ + TableJsonGraphEntry * entries; + int64_t capacity; // a power of two, or zero while empty + int64_t count; + TableAllocator allocator; // the caller's pair (§6.5): the builder's on read, the one handed to ToJson on write +}; + +inline void TableJsonGraphMapInit( TableJsonGraphMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TableJsonGraphMapShutdown( TableJsonGraphMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TableJsonGraphMapInit( map, map.allocator ); +} + +inline int64_t TableJsonGraphMapSlot( const TableJsonGraphMap & map, uint64_t key ) +{ + uint64_t hash = key * 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != 0 && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TableJsonGraphEntry * TableJsonGraphMapFind( TableJsonGraphMap & map, uint64_t key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +inline bool TableJsonGraphMapGrow( TableJsonGraphMap & map ) +{ + TableJsonGraphMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 64; + grown.count = 0; + grown.entries = (TableJsonGraphEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TableJsonGraphEntry ) ); // zeroed, by the pair's contract + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == 0 ) { continue; } + grown.entries[ TableJsonGraphMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// the entry for a key, made if it was not there; `taken` says which. NULL is the +// allocator refusing, and the walk refuses with it. +inline TableJsonGraphEntry * TableJsonGraphMapReach( TableJsonGraphMap & map, uint64_t key, bool & taken ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TableJsonGraphMapGrow( map ) ) { return NULL; } + } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + taken = entry->key != key; + if ( taken ) + { + entry->key = key; + map.count++; + } + return entry; +} + +// ---- reading: into a builder ---- + +struct TableJsonGraphIn +{ + TableWorker * worker; // where every node comes from + TableJsonGraphMap labels; // a label -> the node it defined +}; + +// `&node`'s value, the LABEL: a positive integer spelled as one — digits, no sign, no +// fraction, no exponent, no leading zero (§16.7). Anything else is malformed. +inline bool TableJsonScanLabel( TableJsonIn & in, uint64_t & label ) +{ + TableJsonSpace( in ); + if ( in.pos >= in.size || in.text[in.pos] < '1' || in.text[in.pos] > '9' ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + uint64_t value = 0; + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) + { + uint64_t digit = (uint64_t) ( in.text[in.pos] - '0' ); + if ( value > ( UINT64_MAX - digit ) / 10 ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + value = value * 10 + digit; + in.pos++; + } + label = value; + return true; +} + +// A BYTE BUFFER's text (docs/SPEC-TABLES.md §2.5, §16.2): a string. For a +// *string the string's bytes become the blob; for a *bytes the string is base64 +// and its decoded bytes do. The blob is allocated at EXACTLY the decoded +// length — the string is scanned once without keeping it to learn the length, +// and once into the node — so a blob of any size reads with no window and no +// bound to clamp against. A *bytes body that is not base64 is the wrong shape +// for the kind: the reference stays null and the event is counted. +inline bool TableJsonReadBlob( TableJsonIn & in, void * slot, const TableFieldInfo * f ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + TableRef * ref = (TableRef *) slot; + ref->value = 0; + if ( strcmp( f->type_name, "string" ) == 0 ) + { + const int64_t mark = in.pos; + int32_t length = 0; + if ( !TableJsonScanString( in, NULL, 0, &length ) ) { return false; } + in.pos = mark; + char * data = TableStringEmplace( *graph->worker, *ref, NULL, (int64_t) length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int32_t placed = 0; + return TableJsonScanString( in, data, length, &placed ); + } + // base64: the alphabet characters decide the length, six bits apiece + const char * alphabet = TableJsonBase64Alphabet(); + const int64_t mark = in.pos + 1; + int64_t symbols = 0; + bool malformed = false; + in.pos++; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + if ( c == 0 || strchr( alphabet, c ) == NULL ) { malformed = true; continue; } + symbols++; + } + if ( malformed ) + { + in.report->kind_mismatch++; + return true; + } + const int64_t length = ( symbols * 6 ) / 8; + uint8_t * data = TableBytesEmplace( *graph->worker, *ref, length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int64_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + for ( int64_t at = mark; ; at++ ) + { + char c = in.text[at]; + if ( c == '"' ) { break; } + const char * symbol = c != '=' ? strchr( alphabet, c ) : NULL; + if ( symbol == NULL ) { continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( symbol - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < length ) { data[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); } + } + } + return true; +} + +// A pointer's object. Its FIRST key decides what it is: `&node` naming a label not +// yet defined, with fields after it, is a DEFINITION; `&node` naming one already +// defined, alone, is a REFERENCE; any other key is a node named once, its +// object in place. The node comes from the +// builder's arena, and the slot holds its arena offset (§6.3). A pointer whose +// target is a BYTE BUFFER — no table — takes a string instead (§2.5). +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( f->table == NULL ) { return TableJsonReadBlob( in, slot, f ); } + // the pointee nests one level down, exactly as a by-value table does, and + // takes the same cap: a chain nests as deep as it is long (§16.7) + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + char c = TableJsonPeek( in ); + if ( c == '}' ) + { + // an empty object: a node at its defaults, named once + in.pos++; + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + return true; + } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( strcmp( key, "&node" ) != 0 ) + { + // a node named once: the pointee's object in place, and this key is + // its first field — unless it is the reserved prefix under a spelling + // this form does not have, which ReadTableKeys refuses + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return TableJsonReadTableKeys( in, node, f->table, depth + 1, key ); + } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->labels, label, taken ); + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + // ONE SPELLING, and what follows the label says which half it is: fields + // after a label the text has not defined DEFINE it, and a label alone that + // the text has defined REFERS to it. The other two are malformed — a label + // alone that the text never defined, which would otherwise read as a default + // node under a silent report, and a field after a label already defined, + // which would be a second definition. That is what keeps a typo loud. + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + bool bare = c == '}'; + if ( bare == taken ) { in.report->malformed = true; in.bad = true; return false; } + if ( bare ) + { + // A REFERENCE. A label is defined when its object CLOSES, so a + // reference met inside its own definition — at any depth of by-value + // nesting — names a node whose descent is still open: the cycle the + // wire refuses (§3.1), refused here where it is written. A definition + // the reader dropped names no node, so the slot stays null with + // nothing more counted — the drop was counted where it happened. A + // node of another table than the slot declares is a kind mismatch, as + // on the wire. + in.pos++; + if ( entry->open != 0 ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + if ( entry->type == NULL ) + { + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + if ( entry->type != f->table ) + { + memcpy( slot, &ref, sizeof( ref ) ); + in.report->kind_mismatch++; + return true; + } + ref.value = (int64_t) entry->node; + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + // A DEFINITION: the node is allocated, the label is its, and the keys after + // `&node` are its fields. The entry is OPEN until the object closes, so a + // reference to the label from inside the node's own fields is refused as + // the cycle it is; the node and its table are filled in at the close. + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + entry->open = 1; + if ( !TableJsonReadTableKeys( in, node, f->table, depth + 1, NULL ) ) { return false; } + entry = TableJsonGraphMapFind( graph->labels, label ); // the map may have grown under the descent + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + memcpy( &ref, slot, sizeof( ref ) ); + entry->node = (uint32_t) ref.value; + entry->type = f->table; + entry->open = 0; + return true; +} + +// An `&`-prefixed key opening an object the walk is SKIPPING — a value past an +// array's bound, an unknown key's value, a value of the wrong shape. A +// definition in there still takes its label, so the numbering survives whatever +// the storage could not hold (§16.7): the label is registered with no node, and a +// reference to it reads null. Any other prefixed key is the reserved prefix +// out of place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL || strcmp( key, "&node" ) != 0 ) { in.report->malformed = true; in.bad = true; return false; } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + if ( TableJsonGraphMapReach( graph->labels, label, taken ) == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return true; // a fresh entry is node 0, type NULL: a definition with no node +} + +// ---- writing: from a region's const root ---- + +struct TableJsonGraphOut +{ + TableJsonGraphMap nodes; // a node's address -> how many slots name it, and its `&node` once assigned + bool counting; // PASS ONE: count the references, refuse a cycle, emit nothing + int64_t next_label; +}; + +// The node a slot names: null as `null`, a node named once as its object in +// place, and a node named more than once under the construct. Which of the +// last two it is was learned in pass one; pass two spells it. +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphOut * graph = (TableJsonGraphOut *) out.graph; + if ( graph == NULL ) { return false; } + const void * node = f->resolve( slot ); + if ( node == NULL ) + { + out.raw( "null", 4 ); + return true; + } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->nodes, (uint64_t) (uintptr_t) node, taken ); + if ( entry == NULL ) { return false; } + if ( f->table == NULL ) + { + // A BYTE BUFFER (§2.5, §16.7): its text is a string, which has no + // first key to carry `&node`, so a blob named from more than one + // slot has no spelling this form can carry and the graph is refused — + // as a shared node with nothing to write is. A blob named once is its + // bytes in place: base64 for a *bytes, the string itself for a *string. + if ( graph->counting ) { entry->count++; return true; } + if ( entry->count > 1 ) { return false; } + const TableBlob * blob = (const TableBlob *) node; + if ( blob->length > (uint32_t) 0x7fffffff ) { return false; } + if ( strcmp( f->type_name, "string" ) == 0 ) { TableJsonWriteString( out, (const char *) ( blob + 1 ), (int32_t) blob->length ); } + else { TableJsonWriteBase64( out, (const uint8_t *) ( blob + 1 ), (int32_t) blob->length ); } + return true; + } + if ( graph->counting ) + { + // PASS ONE: one visit per node, every slot that names it counted, and + // a reference to a node whose descent is still open is a cycle — + // refused here as the wire refuses it (§3.1) + entry->count++; + if ( !taken ) { return entry->open == 0; } + entry->open = 1; + if ( !TableJsonWriteValue( out, node, f->table, depth ) ) { return false; } + entry = TableJsonGraphMapFind( graph->nodes, (uint64_t) (uintptr_t) node ); // the map may have grown under the descent + if ( entry == NULL ) { return false; } + entry->open = 0; + return true; + } + // PASS TWO: a node named once is its object in place; a node named more + // than once is DEFINED at its first occurrence — `&node` first, then its + // fields — and REFERENCED by `&node` alone after that, spelled the same way at + // every site. Labels run from 1 in first-write order and are the text's own, + // so a stray number in a hand-edited text is most often one never defined. + if ( entry->count <= 1 ) + { + return TableJsonWriteValue( out, node, f->table, depth ); + } + if ( depth > kTableJsonMaxDepth ) { return false; } + if ( entry->label != 0 ) + { + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + out.line( depth ); + out.put( '}' ); + return true; + } + entry->label = ++graph->next_label; + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + bool any = true; + int64_t before = out.offset; + if ( !TableJsonWriteFields( out, node, f->table, depth, any ) ) { return false; } + // a definition carries at least one field, because a label alone is a + // reference: a shared node with nothing to write has no definition this + // form can spell, and the writer refuses it as it refuses any value it + // cannot spell (§16.3) + if ( out.offset == before ) { return false; } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- the two entry points a pointered table's wrappers name ---- + +// The text into the builder's root. Every node the text names is allocated in +// the builder's arena through the field's own Emplace; the label map is the +// walk's, released before this returns. The root itself takes no label — nothing +// may name it (§16.7) — so an `&node` at the root is refused like any other key +// of the prefix. +inline bool TableJsonReadGraph( TableWorker & worker, void * root, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + if ( worker.arena == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } + TableJsonGraphIn graph; + graph.worker = &worker; + TableJsonGraphMapInit( graph.labels, worker.arena->allocator ); + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = &graph; + info->reset( root ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, root, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + TableJsonGraphMapShutdown( graph.labels ); + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +// The text of a region's const root: measured when the buffer is NULL, written +// when it is not, over one code path. Two passes over one walk — the first +// counts how many slots name each node and refuses a cycle, the second writes +// — so a node's first occurrence knows whether it will be named again. The +// ROOT's entry is open for the whole first pass, so a reference back at it is +// the cycle it is (§3.1), and it takes no label. +inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * info, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + if ( root == NULL ) { return -1; } + TableJsonGraphOut graph; + TableJsonGraphMapInit( graph.nodes, allocator ); + graph.counting = true; + graph.next_label = 0; + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph.nodes, (uint64_t) (uintptr_t) root, taken ); + if ( entry == NULL ) { TableJsonGraphMapShutdown( graph.nodes ); return -1; } + entry->open = 1; + TableJsonOut count; + count.buffer = NULL; + count.capacity = 0; + count.offset = 0; + count.overflow = false; + count.graph = &graph; + bool ok = TableJsonWriteValue( count, root, info, 0 ); + graph.counting = false; + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = &graph; + if ( ok ) { ok = TableJsonWriteValue( out, root, info, 0 ); } + TableJsonGraphMapShutdown( graph.nodes ); + if ( !ok ) { return -1; } + out.put( '\n' ); // the canonical text ends with exactly one newline (§16.1) + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json graph walk: end ---- + +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + +// ---- json map walk: begin ---- + +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} + +// the entry's two rows: fields[0] IS the key and fields[1] IS the value, which +// is what makes a user's own table of pairs the same bytes (§2.8) +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } + +inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } +inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } + +// AN INTEGER KEY IS THE INTEGER'S DECIMAL SPELLING, QUOTED, because a JSON +// object's keys are strings. Written digit by digit so no locale can move it. +inline void TableJsonWriteMapIntegerKey( TableJsonOut & out, const void * storage, const TableFieldInfo * key ) +{ + uint64_t magnitude = 0; + bool negative = false; + if ( TableJsonMapKeySigned( key ) ) + { + int64_t value = 0; + switch ( key->kind ) + { + case 2: value = (int64_t) *(const int8_t *) storage; break; + case 3: value = (int64_t) *(const int16_t *) storage; break; + case 4: value = (int64_t) *(const int32_t *) storage; break; + default: value = *(const int64_t *) storage; break; + } + negative = value < 0; + magnitude = negative ? ( ~(uint64_t) value ) + 1 : (uint64_t) value; + } + else + { + switch ( key->kind ) + { + case 6: magnitude = (uint64_t) *(const uint8_t *) storage; break; + case 7: magnitude = (uint64_t) *(const uint16_t *) storage; break; + case 8: magnitude = (uint64_t) *(const uint32_t *) storage; break; + default: magnitude = *(const uint64_t *) storage; break; + } + } + char digits[24]; + int32_t at = (int32_t) sizeof( digits ); + do { digits[--at] = (char) ( '0' + ( magnitude % 10 ) ); magnitude /= 10; } while ( magnitude != 0 ); + if ( negative ) { digits[--at] = '-'; } + TableJsonWriteString( out, digits + at, (int32_t) sizeof( digits ) - at ); +} + +inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const TableFieldInfo * key ) +{ + const uint8_t * storage = (const uint8_t *) entry + key->offset; + if ( TableJsonMapKeyIsString( key ) ) + { + // A STRING KEY IS THE STRING (§2.8): every JSON key of a map object is + // a KEY OF THE MAP and none is a field key, so the `&` prefix §16.7 + // reserves for field keys is ordinary data here. + TableJsonWriteString( out, (const char *) storage, *(const int32_t *) ( (const uint8_t *) entry + key->count_offset ) ); + return; + } + TableJsonWriteMapIntegerKey( out, (const void *) storage, key ); +} + +// ToJson WRITES ENTRIES IN ASCENDING KEY ORDER, so unpack then pack is +// byte-stable and a diff of two texts is a diff of two maps (§2.8, §17.2). +// A region holds them in that order already, so this is the array in place. +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "{}", 2 ); return true; } + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); + out.put( '{' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); + TableJsonWriteMapKey( out, entry, key ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// AN INTEGER KEY IS READ BY §16.2's INTEGER RULE AND BY NOTHING ELSE, so +// "2.0" and "1e3" are the integers 2 and 1000 and "-0" is zero. The token is +// walked as a JSON number over its own bytes; a token that rule calls +// malformed makes the KEY malformed, and a genuinely fractional value, or one +// outside the key kind's range, is kind_mismatch for that entry. +inline bool TableJsonMapKeyValue( const char * token, int32_t length, const TableFieldInfo * key, + int64_t & value, bool & fits ) +{ + fits = false; + TableReport scratch; + TableJsonIn probe = { token, (int64_t) length, 0, &scratch, false, NULL }; + bool integral = false; + if ( !TableJsonWalkNumber( probe, &integral ) ) { return false; } + if ( probe.pos != (int64_t) length ) { return false; } // trailing bytes: not a number + if ( !integral ) + { + const double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) { return true; } // a value no key kind holds + const double whole = d < 0 ? -d : d; + if ( whole != (double) (int64_t) whole ) { return true; } // genuinely fractional + } + bool saturated = false; + const bool is_signed = TableJsonMapKeySigned( key ); + value = integral ? TableJsonTokenInteger( token, length, is_signed, &saturated ) + : (int64_t) TableJsonTokenDouble( token, length, false ); + if ( saturated ) { return true; } // outside every width: kind_mismatch, never clamped + switch ( key->kind ) + { + case 2: fits = value >= -128 && value <= 127; break; + case 3: fits = value >= -32768 && value <= 32767; break; + case 4: fits = value >= -2147483647 - 1 && value <= 2147483647; break; + case 5: fits = true; break; + case 6: fits = value >= 0 && value <= 255; break; + case 7: fits = value >= 0 && value <= 65535; break; + case 8: fits = value >= 0 && (uint64_t) value <= 4294967295ull; break; + default: fits = integral; break; // uint64: the token's own magnitude + } + return true; +} + +// FromJson READS KEYS IN WHATEVER ORDER THE TEXT GIVES THEM. A repeated key is +// last-wins and counted duplicate, the object rule (§16.2) applied inside the +// map. An empty object is an empty map, and null is kind_mismatch. +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const char shape = TableJsonShape( value ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char token[kTableJsonMaxKey]; + int32_t token_length = 0; + if ( !TableJsonScanString( in, token, kTableJsonMaxKey - 1, &token_length ) ) { return false; } + token[token_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t key_value = 0; + bool place = true; + if ( !TableJsonMapKeyIsString( key ) ) + { + bool fits = false; + if ( !TableJsonMapKeyValue( token, token_length, key, key_value, fits ) ) + { + // A MALFORMED KEY STOPS THE READ where §16.1's rule stops it, + // with the instance holding what was placed before the stop. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( !fits ) { in.report->kind_mismatch++; place = false; } + } + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; + if ( place && entry == NULL ) + { + // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the + // wire's rule, because a clamped key is a merged entry (§2.8). + in.report->clamped++; + } + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) + { + in.report->duplicate++; // last-wins, the object rule inside the map + } + const char got = TableJsonValueShape( in ); + if ( entry == NULL ) + { + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( value->kind == 17 && !value->is_array ) + { + // A POINTER VALUE IS SHARED EXACTLY AS A POINTER FIELD IS (§2.8): + // null is a null slot, an object is the pointee in place or an + // &node reference to one (§16.7), anything else is the wrong shape — + // the same three the field-key loop gives a pointer field, because + // an entry's value IS a field line. + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) entry + value->offset, value->elem_size, 0 ); + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) entry + value->offset, value, depth + 1 ) ) + { + return false; + } + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, entry, value, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json map walk: end ---- + +// ---- json list walk: begin ---- + +// an unbounded array is the out-of-line array that is not a map (§8.1) +inline bool TableJsonIsList( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && !TableJsonIsMap( f ); +} + +// ToJson WRITES THE ELEMENTS IN INDEX ORDER, which is the only order there is, +// so unpack then pack is byte-stable without a rule of its own (§2.9, §17.2). +// A region holds the array in place, so this steps it at the descriptor's pitch. +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + const uint8_t * elements = TableJsonExtentElements( slot ); + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const uint8_t * element = elements + (int64_t) i * f->elem_size; + if ( f->kind == 17 ) + { + // a []*T's elements take the pointer row (§16.7): the pointee's + // object in place, null, or `&node` for a shared one + if ( !TableJsonWritePointer( out, element, f, depth + 1 ) ) { return false; } + } + else if ( !TableJsonWriteScalar( out, element, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; +} + +// FromJson READS EVERY ELEMENT THE TEXT CARRIES, appending each through the +// descriptor's place resolver: `[]` is an empty list, and null is +// kind_mismatch, the array row's own rule (§16.2). LAST WINS holds for a +// repeated key: the list goes back to EMPTY before this occurrence's elements +// land, the builder's storage being reclaimed at reset (§2.9). +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) slot, 8, 0 ); + TableJsonSetRaw( (uint8_t *) slot + 8, 4, 0 ); + const char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + void * element = f->place( *graph->worker, slot, NULL, 0, 0 ); + if ( element == NULL ) + { + // NOT ADDED: the arena could not carve another segment, or the + // count met the int32 cap. The text cannot be placed whole, and + // the read stops where §16.1's rule stops it. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( f->kind == 17 ) + { + // an element of a []*T (§2.9): null is a null slot, an object is the + // pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) element, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, element, f, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + // the wrong shape for the element kind: the slot keeps its + // defaults and the event counts, the array row's rule (§16.2) + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, element, f, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json list walk: end ---- + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +bool UnitFromJson( Unit & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, UnitTableType(), text, bytes, report ); +} + +int64_t UnitToJsonMeasure( const Unit & value ) +{ + return TableJsonWrite( &value, UnitTableType(), NULL, 0 ); +} + +int64_t UnitToJson( const Unit & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, UnitTableType(), buffer, capacity ); +} + +bool BoundedFromJson( Bounded & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, BoundedTableType(), text, bytes, report ); +} + +int64_t BoundedToJsonMeasure( const Bounded & value ) +{ + return TableJsonWrite( &value, BoundedTableType(), NULL, 0 ); +} + +int64_t BoundedToJson( const Bounded & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, BoundedTableType(), buffer, capacity ); +} + +bool UnboundedFromJson( UnboundedBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Unbounded * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, UnboundedTableType(), text, bytes, report ); +} + +int64_t UnboundedToJsonMeasure( const Unbounded * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, UnboundedTableType(), NULL, 0, allocator ); +} + +int64_t UnboundedToJson( const Unbounded * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, UnboundedTableType(), buffer, capacity, allocator ); +} + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/MigrateTable.h b/testdata/golden/tables/lists/MigrateTable.h new file mode 100644 index 000000000..b7c7fe283 --- /dev/null +++ b/testdata/golden/tables/lists/MigrateTable.h @@ -0,0 +1,5606 @@ +// Code generated by the schema compiler from Migrate.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — protocol id 0xa5fbe602c119cdd9 (packets only: tables version by field id, not by protocol id) +// The TABLE wire (evolution-tolerant, docs/SPEC-TABLES.md): no serialize +// dependency — includable from any TU. + +#pragma once + +#include +#include // the prefill's scalar-array fills +#include // offsetof, for the reflection descriptors + +// ---- the hooks (docs/USAGE.md, "the C++ table runtime's hooks") ---- +// +// schema_assert — the runtime's own assert, and the refusal a debugger reads. +// NDEBUG removes it, exactly as it removes assert. A caller who already routes +// serialize's asserts writes `#define schema_assert serialize_assert` before +// including this header and both halves land in one handler. +#ifndef schema_assert +#include +#define schema_assert assert +#endif // #ifndef schema_assert + +// schema_fatal — what stands after the assert on a path that cannot continue. +// NDEBUG does not remove it. Supply it and is never included. +#ifndef schema_fatal +#include // abort +#define schema_fatal abort +#endif // #ifndef schema_fatal + +// schema_allocate / schema_release — what "no allocator handed in" means for +// this program. schema_allocate hands back ZEROED bytes and NULL on failure: +// an arena segment is copied whole, padding included, so anything left +// uninitialized here would reach a packed region. Supply both and +// is never included; hand a TableAllocator to a builder to route one +// structure's allocations somewhere else again. +#ifndef schema_allocate +#include // calloc, free +#define schema_allocate( bytes ) calloc( (size_t) 1, (size_t) ( bytes ) ) +#define schema_release( pointer ) free( pointer ) +#endif // #ifndef schema_allocate +#include // a node's lifetime starts in arena storage (placement new) +#include // one atomic per slab: the arena is lock-free by ownership + +#include "Migrate.h" + +#ifndef LISTDEMO_SCHEMA_TABLE_PRIMITIVES +#define LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +// THE CODEC DOES NOT DEPEND ON THE COMPILER'S INLINING BUDGET. A table of a +// realistic field count emits one large body per type, and the cursor a body +// writes through lives in the caller's `TableWriter`: across a call boundary +// that cursor round-trips through memory, and a `uint8_t *` store may alias the +// writer itself, so every put reloads it. When a budget runs out mid-body the +// codec silently degrades to that shape. Forcing the primitives and the +// fixed-class bodies inline is what keeps the cursor in registers and lets +// adjacent constant framing bytes merge into one store. +#if defined( _MSC_VER ) +#define LISTDEMO_TABLE_INLINE __forceinline +#elif defined( __GNUC__ ) || defined( __clang__ ) +#define LISTDEMO_TABLE_INLINE inline __attribute__(( always_inline )) +#else +#define LISTDEMO_TABLE_INLINE inline +#endif + +namespace listdemo { + +// WHY A READ WAS REFUSED, by name (docs/SPEC-TABLES.md §3.3, §11). A REFUSAL +// is not one of §4's events: nothing is decoded, no counter moves and no +// damage is reported, so five zero counters and a false flag are what a clean +// read prints too and only the verdict tells them apart. The reason says which +// refusal it was. +// +// This is the MESSAGE PATH's vocabulary and not the cooked form's (§7.4): a +// caller meeting one of these has been refused a MESSAGE on a connection, +// which is a different recovery with a different owner than a file a header +// match turned down. +enum TableMessageReason +{ + newer_form, // a FORM BYTE this reader does not carry (§3) + no_vocabulary, // no table for this connection: the message arrived before the announcement, or after a refused one + second_announcement, // a second announcement on a connection: it sets nothing, amends nothing, and the connection closes + vocabulary_too_large, // an announcement above the receiver's declared bound, refused before an entry is touched + message_form_as_file // a form 2 wire where a FILE was expected: its table is somewhere else +}; + +// The table-wire read report — the permissive contract's ledger. Silence +// (all zero) means the data matched this reader's schema exactly. +struct TableReport +{ + int32_t unknown = 0; // unknown field ids skipped (newer data) + int32_t kind_mismatch = 0; // known id, changed type — skipped, never misdecoded + int32_t clamped = 0; // out-of-range values clamped to declared bounds + // a key the TEXT form saw twice: last wins, and the repeat is counted + // (docs/SPEC-TABLES.md §16.2). The wire never raises it — a body carrying an + // id twice is legal input whose last occurrence wins, silently (§3). + int32_t duplicate = 0; + bool malformed = false; // framing damage; decode stopped, partial result kept + // THE REFUSAL VERDICT, which is not one of §4's events and moves no counter + // (docs/SPEC-TABLES.md §3): a FORM BYTE this reader does not carry. Five + // zero counters and a false flag are what a clean read prints too, so the + // verdict is what tells the two apart. + bool refused = false; + // WHICH refusal, and it is read only when refused is set: a read that + // was not refused has no reason, and this member is the one the caller + // must not look at then (docs/SPEC-TABLES.md §3.3). + TableMessageReason reason = newer_form; +}; + + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; +// ---- reflection (tables only, docs/SPEC-TABLES.md) ---- +// +// Static field descriptors for every type in the table closure: name, wire +// id/kind, storage offset, bounds, ranges, enum names and branch guards — +// enough to walk, print, diff, edit or bind any table value at runtime with +// no RTTI and no schema files. TableType() returns X's descriptor. + +struct TableTypeInfo; + +// One arm of a union field: where its payload sits inside the union's storage +// and what its payload looks like. The arm's NAME and its table-wire id come +// from the field's enum_name/variant_id functions at the same tag, so nothing +// is spelled twice (docs/SPEC-TABLES.md §8). +struct TableFieldInfo; + +struct TableUnionArmInfo +{ + uint32_t offset; // offsetof the arm's payload within the union storage + const TableTypeInfo * table; // the arm payload's descriptor, or NULL + // AN ARM IS A FIELD LINE (docs/SPEC-TABLES.md §2.6): an arm that names no + // declared type or table carries the FIELD descriptor a field of that + // type would carry instead — offsets taken within the union storage — so + // a generic walk meets an arm's kind, width, bounds and companions where + // it meets a field's. Exactly one of the two is non-NULL on a set arm. + const TableFieldInfo * field; + uint32_t size; // the arm's whole storage, which selection zero-establishes +}; + +// A union field's shape: the tag, and the arms indexed by it. Arms run +// [0, enum_max]; index 0 is the EMPTY arm and carries no payload. +struct TableUnionInfo +{ + uint32_t tag_offset; // offsetof the tag within the union storage + uint32_t tag_size; // sizeof the tag + const TableUnionArmInfo * arms; +}; + +// The exact raw range of a wide-kind field (docs/SPEC-TABLES.md §8.2): two 128-bit +// values as 64-bit lanes, low lane first, two's complement for the signed kinds. +struct TableWideRange +{ + uint64_t lo[2]; + uint64_t hi[2]; +}; + +// the arena's allocation front, defined with the variable-length runtime +// below; a descriptor names it only through a pointer parameter. +struct TableWorker; + +struct TableFieldInfo +{ + const char * name; // schema field name, e.g. "health" + const char * json; // the TEXT form's key: the json = "key" attribute, else name (§16.3) + const char * type_name; // schema type name, e.g. "float32", "Grade" + uint64_t id; // table-wire field id: fnv1a64 of the name, of the was alias after a rename (§5) + uint8_t kind; // table-wire kind; for arrays/strings/bytes, the ELEMENT kind + bool is_array; // fixed or counted array (bytes included) + bool is_pointer; // a *T pointer field: storage is an 8-byte TableRef; the target is a table + // THE TWO THE TEXT FORM NEEDS (docs/SPEC-TABLES.md §16.7), and they + // are here for the same reason is_pointer is: the walk is ONE walk + // over descriptors and cannot spell a target's own At or + // Emplace. `resolve` reads a slot in a REGION and answers the + // node it names, or NULL; `emplace` allocates one in a BUILDER's + // arena and points the slot at it. NULL on every field that is not + // a pointer, and emitted only in a unit that declares one. + const void * (*resolve)( const void * slot ); + void * (*emplace)( TableWorker & worker, void * slot ); + bool counted; // a _count/_length int32 companion exists (counted arrays, strings, bytes) + bool optional; // a ?T field: a _present bool companion decides whether it rides + int32_t array_bound; // array capacity / string max length; 0 for plain scalars + uint32_t offset; // offsetof the storage member + uint32_t elem_size; // sizeof the member (element size for arrays) + uint32_t count_offset; // offsetof the _count/_length companion, or 0xffffffff + uint32_t present_offset; // offsetof the _present companion, or 0xffffffff + const TableTypeInfo * table; // nested table's descriptor, or NULL + bool has_range; // a declared [min, max] (int or float) + double range_min; // NOTE: int64 ranges beyond 2^53 lose precision here + double range_max; + // the WIDE kinds (18-29, docs/SPEC-TABLES.md §3, §8.2): frac_bits is a fixed + // field's F — its storage holds units × 2^F — and wide is the declared + // range on that RAW scale, exact, as two 128-bit two's-complement values + // in 64-bit lanes (low lane first). NULL where the declaration bounds + // nothing (a bare uint128) and for every other kind; frac_bits is 0 for + // every kind that is not fixed-point. range_min/range_max still carry + // the declared bounds as doubles — whole units for a fixed field — for + // a walker that only shows them. + uint8_t frac_bits; + const TableWideRange * wide; + int64_t enum_max; // enums: highest valid value (None = 0 always valid); + // unions: the arm count (tag range [0, enum_max]); + // flags: the highest declared BIT INDEX; else -1 + // the vocabulary's names, indexed the same way enum_max bounds: an enum's + // value -> name, a union's tag -> arm name, a FLAGS field's bit index -> + // variant name. NULL for every other kind. + const char * (*enum_name)( uint64_t value ); + // the TABLE-WIRE id of one variant (docs/SPEC-TABLES.md §5): for an enum, the + // hash of the variant's name; for a union, the hash of the arm's name. + // 0 is the reserved id — an enum's None, a union's empty. NULL for every + // other kind — a FLAGS field's variants have no per-variant wire id (§4), + // so a NULL here beside a non-NULL enum_name is what says "flags". + // Walk [0, enum_max] to enumerate a vocabulary and its ids. + uint64_t (*variant_id)( uint64_t value ); + // an ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): the array has one slot per + // variant of key_type_name, indexed by the variant's value, and its slots + // ride under variant ids rather than positions. key_name and key_id are + // the key's vocabulary — walk [0, array_bound) to print slots by name. + // NULL on every other field. + const char * key_type_name; + const char * (*key_name)( uint64_t value ); + uint64_t (*key_id)( uint64_t value ); + // union fields: the tag and its arms, behind a function so the whole + // descriptor stays CONSTANT-INITIALISED (a captureless lambda converts to + // a function pointer at compile time; the arms themselves are a static + // inside it). NULL for every other kind. + const TableUnionInfo * (*arms)(); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded +}; + +struct TableTypeInfo +{ + const char * name; // schema type name + uint32_t size; // sizeof the storage struct + int32_t num_fields; + const TableFieldInfo * fields; + // put one instance back at its declared defaults, in place. A generic + // walker that fills a value has to be able to establish the defaults an + // absent field takes, and it holds no type to spell — this is the one + // thing the descriptors could not express without it. Placement-new + // value-init, exactly what the wire's read path does, and no temporary. + void (*reset)( void * storage ); + // the DERIVED mode (docs/SPEC-TABLES.md): false = fixed-size, a plain + // relocatable struct; true = variable-length, built through a Builder + // and read through a region root. Nobody declares it; the compiler + // works it out. + bool variable; +}; + +struct TableWriter +{ + uint8_t * buffer; + int64_t capacity; + int64_t offset = 0; + bool overflow = false; + + // the parameters do not repeat the member names: a parameter that hides a + // member is a warning the estate's compilers disagree about (gcc's + // -Wshadow and cl's C4458 refuse it, clang's -Wshadow does not), and this + // is a header a consumer compiles under its OWN flags + TableWriter( uint8_t * to_buffer, int64_t to_capacity ) : buffer( to_buffer ), capacity( to_capacity ) {} + + LISTDEMO_TABLE_INLINE void raw( const void * data, int64_t bytes ) + { + if ( offset + bytes > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) bytes ); + offset += bytes; + } + LISTDEMO_TABLE_INLINE void put8( uint8_t v ) { raw( &v, 1 ); } + LISTDEMO_TABLE_INLINE void put16( uint16_t v ) { uint8_t b[2] = { uint8_t( v ), uint8_t( v >> 8 ) }; raw( b, 2 ); } + LISTDEMO_TABLE_INLINE void put32( uint32_t v ) { uint8_t b[4] = { uint8_t( v ), uint8_t( v >> 8 ), uint8_t( v >> 16 ), uint8_t( v >> 24 ) }; raw( b, 4 ); } + LISTDEMO_TABLE_INLINE void put64( uint64_t v ) { put32( uint32_t( v ) ); put32( uint32_t( v >> 32 ) ); } + // a 128-bit value as two lanes, the low half first (docs/SPEC-TABLES.md §3) + LISTDEMO_TABLE_INLINE void put128( uint64_t lo, uint64_t hi ) { put64( lo ); put64( hi ); } + // EVERY LENGTH, COUNT, INDEX AND ID REFERENCE IS ONE CANONICAL UNSIGNED + // LEB128 (docs/SPEC-TABLES.md §3): seven value bits a byte, the lowest + // group first, the high bit set on every byte but the last. One value has + // one spelling, so two conforming writers agree byte for byte. + LISTDEMO_TABLE_INLINE void putleb( uint64_t v ) + { + while ( v >= 0x80 ) { put8( uint8_t( v ) | 0x80 ); v >>= 7; } + put8( uint8_t( v ) ); + } +}; + +// TableLebBytes is one value's spelling length, which a MEASURE needs before +// the bytes exist — the length of a body has to be known before it is written, +// because a length whose own width moves cannot be patched in place. +inline int64_t TableLebBytes( uint64_t v ) +{ + int64_t n = 1; + while ( v >= 0x80 ) { v >>= 7; n++; } + return n; +} + +// THE ID TABLE, WRITER SIDE (docs/SPEC-TABLES.md §3). It holds every id the +// body used, once each, in FIRST-USE order over the whole wire, and the body +// names them by position: reference k is the kth entry, counted from 1, and +// reference 0 names NO ID. +// +// Its capacity is a COMPILE-TIME fact of the unit — the distinct names its +// table closure can spell — so a save allocates nothing: the table is a local +// of Measure and of Save. The bucket chain makes ref constant time and makes +// truncate constant time too, which is what an ELIDED field needs: a field +// that turns out not to ride costs nothing in the id table either, so the walk +// interns its id, builds the payload that decides, and undoes the entry when +// nothing rides. +struct TableIds +{ + static const int32_t kCapacity = 60; + static const int32_t kBuckets = 128; + + uint64_t ids[ kCapacity ]; + int32_t chain[ kCapacity ]; + int32_t head[ kBuckets ]; + int32_t count; + bool overflow; + // THE MESSAGE FORM'S SLOTS (docs/SPEC-TABLES.md §3.3). A form 2 wire + // names ids through the CONNECTION's table, which is the unit's whole + // vocabulary in a compiler-settled order — so every reference is known at + // compile time and rides at the header as a literal beside the id. This + // flag is what selects it: false interns the id in first-use order and + // writes a trailer, true answers the slot and writes none, and the walk + // that decides is one walk. + bool vocabulary; + + TableIds() : count( 0 ), overflow( false ), vocabulary( false ) + { + for ( int32_t i = 0; i < kBuckets; i++ ) { head[i] = -1; } + } + + static LISTDEMO_TABLE_INLINE uint32_t bucket_of( uint64_t id ) + { + return uint32_t( ( id * 0x9E3779B97F4A7C15ull ) >> 57 ) & uint32_t( kBuckets - 1 ); + } + + // the reference an id takes: its message-form SLOT under the connection's + // table, or the file's own first-use entry + LISTDEMO_TABLE_INLINE uint64_t ref( uint64_t id, uint64_t slot ) + { + if ( vocabulary ) { return slot; } + return intern( id ); + } + + // the FILE form's half, appending the id on first use + uint64_t intern( uint64_t id ) + { + const uint32_t b = bucket_of( id ); + for ( int32_t i = head[b]; i >= 0; i = chain[i] ) + { + if ( ids[i] == id ) { return uint64_t( i ) + 1; } + } + if ( count >= kCapacity ) { overflow = true; return 1; } + ids[count] = id; chain[count] = head[b]; head[b] = count; count++; + return uint64_t( count ); + } + + // undo every entry appended since mark. An entry removed is the most + // recent one in its bucket, so it sits at that bucket's head. + void truncate( int32_t mark ) + { + // a SLOT costs no entry, so an elided field has nothing to undo + if ( vocabulary ) { return; } + while ( count > mark ) + { + count--; + head[ bucket_of( ids[count] ) ] = chain[count]; + } + } +}; + +// TableIdsBytes is the trailer's own size: the entries, each a fixed +// little-endian u64, and the ENTRY COUNT, the one fixed-width number on the +// wire (docs/SPEC-TABLES.md §3). +inline int64_t TableIdsBytes( const TableIds & ids ) { return int64_t( ids.count ) * 8 + 8; } + +// TableIdsWrite puts the trailer where the walk ended: a writer never patches, +// because first-use order is known only when the walk ends. +inline void TableIdsWrite( TableWriter & w, const TableIds & ids ) +{ + for ( int32_t i = 0; i < ids.count; i++ ) { w.put64( ids.ids[i] ); } + w.put64( uint64_t( ids.count ) ); +} + +// THE ID TABLE, READER SIDE (docs/SPEC-TABLES.md §3). A reader locates it from +// the END of the wire and resolves it ONCE, at open: the entries are eight +// bytes each and a body names them by position, so every field dispatches +// through an index rather than through a search over hashes. +struct TableIdTable +{ + const uint8_t * entries = NULL; + int64_t count = 0; + + // the id a reference names. ref is 1-based and bounds-checked by the + // caller: a reference ABOVE the entry count is framing damage on the body + // that carries it, and 0 names no id at all. + uint64_t at( uint64_t ref ) const + { + const uint8_t * e = entries + ( ref - 1 ) * 8; + uint64_t lo = uint64_t( e[0] ) | uint64_t( e[1] ) << 8 | uint64_t( e[2] ) << 16 | uint64_t( e[3] ) << 24; + uint64_t hi = uint64_t( e[4] ) | uint64_t( e[5] ) << 8 | uint64_t( e[6] ) << 16 | uint64_t( e[7] ) << 24; + return lo | ( hi << 32 ); + } +}; + +struct TableReader +{ + const uint8_t * buffer; + int64_t size; + int64_t offset = 0; + TableReport * report; + const TableIdTable * ids = NULL; + // ONLY THE ROOT BODY CARRIES THE NODE TABLE (docs/SPEC-TABLES.md §3.1), so + // a body has to know which it is: the reserved id inside a NESTED body is + // malformed, because a second numbering cannot exist. Every reader made + // for a payload is nested; the two the wire surfaces make for a root say so. + bool nested = true; + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report ) + : buffer( from_buffer ), size( from_size ), report( to_report ) {} + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report, const TableIdTable * to_ids ) + : buffer( from_buffer ), size( from_size ), report( to_report ), ids( to_ids ) {} + + LISTDEMO_TABLE_INLINE bool has( int64_t bytes ) const { return offset + bytes <= size; } + // A LENGTH IS A 64-BIT NUMBER AND A BUFFER IS NOT (docs/SPEC-TABLES.md + // §3): every length, count and index on this wire has sixty-four bits of + // capability, so one past what remains must be compared UNSIGNED. Casting + // it to int64 first turns 0xFFFFFFFFFFFFFFFF into -1, and a negative + // length looks like room. + LISTDEMO_TABLE_INLINE bool room( uint64_t bytes ) const { return bytes <= (uint64_t) ( size - offset ); } + LISTDEMO_TABLE_INLINE uint8_t get8() { return buffer[offset++]; } + LISTDEMO_TABLE_INLINE uint16_t get16() { uint16_t v = uint16_t( buffer[offset] ) | uint16_t( buffer[offset+1] ) << 8; offset += 2; return v; } + LISTDEMO_TABLE_INLINE uint32_t get32() { uint32_t v = uint32_t( buffer[offset] ) | uint32_t( buffer[offset+1] ) << 8 | uint32_t( buffer[offset+2] ) << 16 | uint32_t( buffer[offset+3] ) << 24; offset += 4; return v; } + LISTDEMO_TABLE_INLINE uint64_t get64() { uint64_t lo = get32(); uint64_t hi = get32(); return lo | ( hi << 32 ); } + LISTDEMO_TABLE_INLINE void get128( uint64_t & lo, uint64_t & hi ) { lo = get64(); hi = get64(); } + + // ONE CANONICAL UNSIGNED LEB128 (docs/SPEC-TABLES.md §3), and a + // non-minimal spelling is MALFORMED: 0x80 0x00 and 0x00 both spell zero, + // and only the second is legal input. An encoding past ten bytes, or a + // tenth byte with a bit above the 64th value bit, is malformed on the same + // rule. false = framing damage on the body carrying it. + bool getleb( uint64_t & value ) + { + // A NUMBER THIS READER REFUSES LEAVES THE CURSOR WHERE IT WAS. The + // caller's next question is often "did this body end exactly at its + // L", and a rejected number that had moved the cursor would answer + // that question with the damage already stepped over. + const int64_t at = offset; + value = 0; + uint32_t shift = 0; + for ( int32_t i = 0; i < 10; i++ ) + { + if ( !has( 1 ) ) { offset = at; return false; } + const uint8_t b = get8(); + if ( i == 9 && b > 1 ) { offset = at; return false; } + value |= uint64_t( b & 0x7F ) << shift; + if ( ( b & 0x80 ) == 0 ) + { + if ( i > 0 && b == 0 ) { offset = at; return false; } // a redundant continuation + return true; + } + shift += 7; + } + offset = at; + return false; + } + + // resolve one id reference against the file's table. false = a reference + // ABOVE the entry count, or a 0 where an id is required, both of which + // are framing damage on the body that carries it. + bool getid( uint64_t & id ) + { + uint64_t ref = 0; + if ( !getleb( ref ) ) { return false; } + if ( ref == 0 || ids == NULL || ref > (uint64_t) ids->count ) { return false; } + id = ids->at( ref ); + return true; + } + + // skip one payload by kind; false = framing damage. FOUR RULES COVER THE + // SET (docs/SPEC-TABLES.md §3), and a kind outside it is not skippable — + // which is why the set is closed and why kind 31 exists. + bool skip( uint8_t kind ) + { + switch ( kind ) + { + // the fixed-width kinds, each by its width: 18-29 are the 128-bit integers and + // the fixed-point family at every storage width (docs/SPEC-TABLES.md §3) + case 1: case 2: case 6: case 20: case 25: return has( 1 ) ? ( offset += 1, true ) : false; + case 3: case 7: case 21: case 26: return has( 2 ) ? ( offset += 2, true ) : false; + case 4: case 8: case 10: case 22: case 27: return has( 4 ) ? ( offset += 4, true ) : false; + case 5: case 9: case 11: case 23: case 28: return has( 8 ) ? ( offset += 8, true ) : false; + case 18: case 19: case 24: case 29: return has( 16 ) ? ( offset += 16, true ) : false; + case 17: case 30: // a NODE INDEX (§3.1) and an ENUM's variant reference: one LEB128 and stop + { + uint64_t ignored = 0; + return getleb( ignored ); + } + case 12: case 13: case 14: case 16: case 31: case 32: // 31 is the ESCAPE, 32 the payload-free kind + { + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + case 15: // union: the arm id reference, then its kind, its L and its payload (reference 0 = empty) + { + uint64_t arm = 0; + if ( !getleb( arm ) ) return false; + if ( arm == 0 ) return true; + if ( !has( 1 ) ) return false; + offset += 1; // the arm's kind byte + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + } + return false; + } +}; + +// The RESERVED node-table id, the one id the language holds back +// (docs/SPEC-TABLES.md §3.1, §5). It rides in every unit, pointered or not, +// because every body has to know that a NESTED body claiming one is damaged. +static const uint64_t kTableNodeTableFieldId = 0xFFFFFFFFFFFFFFFFull; + +// TableWireForm is the FORM BYTE, and it is the whole header +// (docs/SPEC-TABLES.md §3). A reader that meets a byte it does not know +// refuses the wire by name and never reports damage. +const uint8_t kTableWireForm = 1; + +// TableOpen reads the form byte and the trailer, in that order, and hands back +// the ROOT BODY. It answers one of three verdicts, because five zero counters +// and a false flag are what a clean read prints too: +// +// TableOpenOk the form is known and the table read whole +// TableOpenRefused a FORM BYTE this reader does not carry: nothing is +// decoded, nothing is counted, and no damage is reported +// TableOpenDamaged a table that cannot be read whole — fewer than eight +// bytes, a count whose entries run past the front of the +// file, a count that leaves no room for the form byte, or +// ONE ID IN TWO ENTRIES. The whole wire is malformed, +// nothing is decoded, and one event is counted. +// TableOpenBodyStopped the form and the table were good and the ROOT BODY +// could not be walked to its own terminator. What it +// decoded before that is kept, as everywhere on this wire. +enum TableOpenVerdict { TableOpenOk, TableOpenRefused, TableOpenDamaged, TableOpenBodyStopped }; + +inline TableOpenVerdict TableOpen( const uint8_t * buffer, int64_t bytes, TableIdTable & table, int64_t & body_bytes ) +{ + if ( bytes < 1 ) { return TableOpenDamaged; } + if ( buffer[0] != kTableWireForm ) { return TableOpenRefused; } + if ( bytes < 9 ) { return TableOpenDamaged; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + uint64_t count = lo | ( hi << 32 ); + if ( count > (uint64_t) ( bytes / 8 ) ) { return TableOpenDamaged; } + const int64_t span = (int64_t) count * 8 + 8; + if ( span + 1 > bytes ) { return TableOpenDamaged; } + table.entries = buffer + bytes - span; + table.count = (int64_t) count; + // THE ENTRIES ARE DISTINCT: a table that carries one id twice is malformed + // for the whole wire, because no wire this schema writes carries a repeat + // and it would leave one more shape of table for a hostile writer to aim + // at (docs/SPEC-TABLES.md §3). + for ( int64_t i = 1; i < table.count; i++ ) + { + const uint64_t id = table.at( uint64_t( i ) + 1 ); + for ( int64_t j = 0; j < i; j++ ) + { + if ( table.at( uint64_t( j ) + 1 ) == id ) { return TableOpenDamaged; } + } + } + body_bytes = bytes - span - 1; + return TableOpenOk; +} + +// TableBodyExtent walks a body's framing to the zero reference that ends it, +// so a reader can tell a body that ENDED EARLY — leaving bytes no field claims +// — from one that is merely damaged. ANY BYTE BETWEEN THE ROOT'S TERMINATOR +// AND THE TABLE'S FIRST ENTRY IS MALFORMED, because no field claims it and the +// two ends of the file have met (docs/SPEC-TABLES.md §3). +inline bool TableBodyEndsEarly( const uint8_t * body, int64_t bytes, const TableIdTable & table ) +{ + TableReport ignored; + TableReader r( body, bytes, &ignored, &table ); + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { return false; } + if ( ref == 0 ) { return r.offset != bytes; } + if ( ref > (uint64_t) table.count ) { return false; } + if ( !r.has( 1 ) ) { return false; } + if ( !r.skip( r.get8() ) ) { return false; } + } +} + +// THE MESSAGE FORM (docs/SPEC-TABLES.md §3.3): a FILE carries its own id +// table and a MESSAGE STREAM announces one and then carries none. +// +// A form 2 wire is TWO PARTS, the form byte and the root body: the body ends +// at its own zero reference as it does in a file, there is no trailer, and the +// message's last byte is the body's terminator. Its references resolve against +// the CONNECTION's table, which is the unit's whole vocabulary in the order +// the compiler settled. +const uint8_t kTableWireMessageForm = 2; + +// The RESERVED build-version id, the second id the language holds back (§5, +// §11), beside the node table's. It is the announcement's one required field, +// and a reserved id in any body but the one whose transport it is, is +// malformed (§3.1). +static const uint64_t kTableBuildVersionFieldId = 0xFFFFFFFFFFFFFFFEull; + +// The reserved NODE-TABLE id's own slot in this unit's vocabulary (§3.3). A +// pointered message names the node table through it, exactly as every other +// field header names its id through a slot. +static const uint64_t kTableNodeTableFieldSlot = 38; + +// THE UNIT'S ANNOUNCEMENT, byte for byte: 61 entries and 508 bytes. It is an +// ordinary form 1 FILE — the form byte, a body carrying the BUILD VERSION +// under the reserved id at kind 9, and the trailer that IS the connection's +// table, slot 1 the reserved id and slots 2 and up the vocabulary under one +// numbering. +// +// The vocabulary is the unit's whole closure in the COOK PROJECTION's order +// (§20.2) — each record in the order the projection renders it and each +// record's fields in the order the projection renders them, then each enum's +// variants and each union's arms — followed by the tail the projection does +// not name: the reserved node-table id, the three blob type ids as bytes, +// string and wstring, and every table's own name id in the projection's sorted +// record order. The tail is UNCONDITIONAL, so an ordinary edit only ever grows +// it at its end and never moves a slot a generated field header carries as a +// literal. +static const int64_t kTableAnnounceBytes = 508; +static const uint8_t kTableAnnounce[ kTableAnnounceBytes ] = { + 0x01, 0x01, 0x09, 0xc7, 0x71, 0x45, 0xca, 0xda, 0x0e, 0x7c, 0x8d, 0x00, + 0xfe, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x30, 0xb1, 0x3a, 0xff, + 0x4a, 0xd9, 0xb1, 0x40, 0x20, 0xea, 0x4d, 0x40, 0x8e, 0xa7, 0x19, 0xaa, + 0x26, 0xa9, 0x02, 0x0c, 0x9b, 0x01, 0x48, 0x78, 0xe9, 0xea, 0x71, 0x6f, + 0x0f, 0x01, 0x82, 0xbf, 0x6f, 0x2c, 0x41, 0x4f, 0xbf, 0x84, 0x78, 0x3e, + 0xf3, 0xa4, 0x48, 0x44, 0x19, 0xab, 0xd7, 0x56, 0x05, 0x4a, 0xa3, 0x30, + 0x67, 0x55, 0x5b, 0x85, 0xc9, 0xe2, 0x4e, 0x30, 0x69, 0x6a, 0xb4, 0x81, + 0xfb, 0x67, 0x4d, 0x1a, 0xcf, 0x7b, 0x27, 0x21, 0x74, 0xa2, 0x79, 0x44, + 0x8e, 0xe2, 0xe5, 0xb1, 0x84, 0x76, 0xbc, 0x2e, 0xef, 0x83, 0x76, 0x1e, + 0xc5, 0x99, 0xf7, 0x82, 0x76, 0x4e, 0x0a, 0xd9, 0xa8, 0x2e, 0x86, 0x70, + 0x84, 0xed, 0xf2, 0x4a, 0xbb, 0xf0, 0x0c, 0x9b, 0xcc, 0xfb, 0x2d, 0x73, + 0x68, 0xb7, 0xf0, 0xae, 0x4c, 0x0c, 0xf6, 0x52, 0xbf, 0xe9, 0xd1, 0x2f, + 0x93, 0xcd, 0xda, 0xdb, 0x22, 0x72, 0x34, 0x7d, 0xf6, 0x0b, 0x72, 0x17, + 0x07, 0x17, 0x02, 0x86, 0x4c, 0xf5, 0x63, 0xaf, 0x54, 0x15, 0x02, 0x86, + 0x4c, 0xf4, 0x63, 0xaf, 0x3a, 0x70, 0x6e, 0x3e, 0x93, 0x43, 0xe5, 0x9d, + 0x3d, 0x62, 0xcb, 0x8f, 0xec, 0xfc, 0xf7, 0x39, 0x09, 0x06, 0x02, 0x86, + 0x4c, 0xeb, 0x63, 0xaf, 0x09, 0x4b, 0x4d, 0x57, 0xaa, 0x33, 0x47, 0xd2, + 0x31, 0x54, 0xaf, 0x1d, 0x19, 0x73, 0x50, 0x12, 0xb2, 0x0f, 0x40, 0x27, + 0x0b, 0x6b, 0x98, 0x01, 0x38, 0x81, 0x0a, 0xf1, 0x1f, 0x06, 0xa7, 0xa3, + 0x0f, 0x62, 0xad, 0x07, 0x77, 0x47, 0x82, 0x5f, 0x42, 0x4f, 0x4f, 0x30, + 0x0d, 0x39, 0x84, 0x1c, 0x86, 0x1b, 0x63, 0x8e, 0xba, 0xad, 0xbc, 0xc4, + 0xec, 0x10, 0x5b, 0x36, 0x19, 0x4a, 0xc9, 0x3d, 0xea, 0x0c, 0xe8, 0x30, + 0x94, 0xfd, 0xe4, 0x7c, 0xec, 0x22, 0x02, 0x86, 0x4c, 0xfc, 0x63, 0xaf, + 0x05, 0x28, 0x02, 0x86, 0x4c, 0xff, 0x63, 0xaf, 0x52, 0x26, 0x02, 0x86, + 0x4c, 0xfe, 0x63, 0xaf, 0xb1, 0x45, 0xc3, 0x44, 0x35, 0xab, 0xfe, 0x73, + 0xc0, 0x7f, 0xb3, 0x8a, 0xbe, 0x08, 0x63, 0x7f, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xe4, 0x4f, 0x1c, 0x4f, 0x47, 0xc0, 0x2e, 0x2f, + 0x58, 0xfc, 0xaf, 0xfa, 0xd8, 0xe0, 0x4b, 0x70, 0xc7, 0xd4, 0x7b, 0x26, + 0xb0, 0x9d, 0x29, 0x5f, 0xcc, 0x14, 0x15, 0x7f, 0xcb, 0xc2, 0x58, 0xd8, + 0x84, 0x9e, 0x3a, 0x55, 0x8b, 0x37, 0xe2, 0x06, 0x2a, 0x89, 0xf5, 0x12, + 0x09, 0xc1, 0xca, 0x0a, 0x44, 0xa2, 0x31, 0xc1, 0xad, 0xa7, 0xee, 0xee, + 0xe8, 0xcf, 0xbf, 0x43, 0x73, 0x18, 0x43, 0xd0, 0x42, 0xad, 0xf6, 0xf8, + 0x59, 0x86, 0x63, 0x91, 0xb7, 0xce, 0x00, 0x7c, 0xd1, 0xc5, 0x34, 0x20, + 0x06, 0x68, 0x47, 0x98, 0xd1, 0xa1, 0xcf, 0x52, 0x5f, 0x82, 0x58, 0xac, + 0x36, 0x15, 0x78, 0x5e, 0xb8, 0x8b, 0x59, 0x6f, 0xc9, 0xc6, 0x86, 0xbb, + 0xc3, 0x64, 0x89, 0x50, 0xd2, 0x8d, 0xa7, 0xf1, 0x80, 0xea, 0x3a, 0xb9, + 0xf1, 0x21, 0xf7, 0x41, 0x11, 0xed, 0xd9, 0xce, 0x96, 0x92, 0x43, 0x8a, + 0xfb, 0x06, 0xc9, 0xfe, 0x19, 0xe1, 0x13, 0xa0, 0xa7, 0x0a, 0xc7, 0x54, + 0x12, 0xd6, 0x40, 0xdc, 0x08, 0xf0, 0xf5, 0xc0, 0x24, 0x5f, 0xf8, 0x33, + 0xc8, 0xfb, 0x85, 0x9a, 0xaf, 0xe0, 0xc9, 0x0c, 0x91, 0x0a, 0x55, 0x60, + 0xf7, 0xa2, 0x07, 0xec, 0x8b, 0x6d, 0x02, 0x86, 0x43, 0xf3, 0xc2, 0x2e, + 0x87, 0x27, 0xcc, 0x86, 0xf0, 0xe0, 0x26, 0x8f, 0x3d, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, +}; + +// AnnounceMeasure is the announcement's byte count, which is a constant of the +// unit and not a walk. +inline int64_t AnnounceMeasure() { return kTableAnnounceBytes; } + +// Announce writes the announcement into the caller's buffer and answers the +// bytes written — exactly AnnounceMeasure's answer — or -1 when the buffer is +// too small. It allocates nothing and walks nothing. +inline int64_t Announce( uint8_t * buffer, int64_t capacity ) +{ + if ( buffer == NULL || capacity < kTableAnnounceBytes ) { return -1; } + memcpy( buffer, kTableAnnounce, (size_t) kTableAnnounceBytes ); + return kTableAnnounceBytes; +} + +// TableVocabulary is ONE DIRECTION of ONE CONNECTION's id table (§3.3): the +// entries an announcement carried, whole, under one numbering with slot 1 the +// reserved build-version id. +// +// A peer holds TWO of these for a connection, the one it writes with and the +// one it reads with, and neither is the other's. A restart opens a fresh +// connection with empty tables and nothing is cached across connections, so +// its whole life is one connection's. It BORROWS the announcement's bytes rather than +// copying them, so a receiver holds one table a direction and its memory is +// the bound below and nothing else. +struct TableVocabulary +{ + // THE CONFORMING DEFAULT BOUND (§3.3): 32 KiB a direction, eight times the + // 500-id unit that is already a large one. A connection's table is bounded + // by nothing the wire carries, so the receiver declares the maximum and an + // announcement above it is refused by name before an entry is touched. + static const int64_t kDefaultMaxEntries = 4096; + + TableIdTable table; + uint64_t build_version = 0; + bool announced = false; + int64_t max_entries = kDefaultMaxEntries; +}; + +// AnnounceRead reads an announcement into one direction's table (§3.3). +// +// THE BOUND IS CHECKED BEFORE ANYTHING IS ALLOCATED: the entry count is a +// fixed little-endian u64 at the end, so a receiver reads it, compares it and +// refuses without touching an entry. After that it is §3's ordinary FILE read, +// because the announcement IS a file, with EXACTLY ONE STRICT CHECK over its +// body: the reserved build-version field present, exactly once, under kind 9, +// eight bytes wide. Everything else is an ordinary field under §4's tolerance, +// so an unknown one is skipped and counted and the announcement can GAIN a +// field in a later minor without a lockstep redeploy. +// +// The FIRST announcement sets the table and it is the only one that can. A +// SECOND is refused by name: it does not replace the table, it does not amend +// it and it changes nothing. A refused announcement sets NO TABLE. +inline bool AnnounceRead( TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + if ( vocabulary.announced ) + { + to->refused = true; + to->reason = second_announcement; + return false; + } + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireForm ) + { + to->refused = true; + to->reason = buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + return false; + } + if ( bytes < 9 ) { to->malformed = true; return false; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + if ( ( lo | ( hi << 32 ) ) > (uint64_t) vocabulary.max_entries ) + { + to->refused = true; + to->reason = vocabulary_too_large; + return false; + } + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else { to->refused = true; to->reason = newer_form; } + return false; + } + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) { to->malformed = true; return false; } + // the body, under §4's tolerance and this form's one strict check + TableReader r( buffer + 1, body_bytes, to, &table ); + uint64_t version = 0; + int32_t seen = 0; + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { to->malformed = true; return false; } + if ( ref == 0 ) { break; } + if ( ref > (uint64_t) table.count || !r.has( 1 ) ) { to->malformed = true; return false; } + const uint64_t id = table.at( ref ); + const uint8_t kind = r.get8(); + if ( id != kTableBuildVersionFieldId ) + { + to->unknown++; + if ( !r.skip( kind ) ) { to->malformed = true; return false; } + continue; + } + if ( kind != 9 || !r.has( 8 ) ) { to->refused = true; to->reason = no_vocabulary; return false; } + version = r.get64(); + seen++; + } + if ( seen != 1 ) { to->refused = true; to->reason = no_vocabulary; return false; } + vocabulary.table = table; + vocabulary.build_version = version; + vocabulary.announced = true; + return true; +} + +inline float table_bits_to_float( uint32_t bits ) { float f; memcpy( &f, &bits, 4 ); return f; } +inline uint32_t table_float_to_bits( float f ) { uint32_t b; memcpy( &b, &f, 4 ); return b; } +inline double table_bits_to_double( uint64_t bits ) { double d; memcpy( &d, &bits, 8 ); return d; } +inline uint64_t table_double_to_bits( double d ) { uint64_t b; memcpy( &b, &d, 8 ); return b; } + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +#ifndef LISTDEMO_SCHEMA_TABLE_ARENA +#define LISTDEMO_SCHEMA_TABLE_ARENA + +namespace listdemo { + +// ---- variable-length tables: tuning constants (docs/SPEC-TABLES.md) ---- +// +// The segment size and the count multiply to exactly 2^32: the u32 reference +// is the arena's hard ceiling, and these constants saturate it rather than +// leaving address space unreachable. Slab handout costs one atomic per slab, +// so per-node allocation costs no synchronization at all. + +static const uint32_t kTableSegmentBits = 22; // 4 MiB segments +static const uint32_t kTableSegmentSize = 1u << kTableSegmentBits; +static const uint32_t kTableSegmentMask = kTableSegmentSize - 1u; +static const uint32_t kTableMaxSegments = 1u << ( 32 - kTableSegmentBits ); // 1024 -> 4 GiB +static const uint32_t kTableSlabBytes = 64u * 1024u; // one atomic per slab +static const uint32_t kTableAlign = 8; // every node starts 8-aligned +static const uint32_t kTableAllocFailed = 0xFFFFFFFFu; + +// ---- THE CALLER'S ALLOCATOR (docs/SPEC-TABLES.md §6.5) ---- +// +// Every allocation the variable-length runtime makes goes through one of +// these — the arena's segments, the pack walk's identity map, the numbering's +// entry array, the packed region, and the tool path's node directory. There is +// no other call to the C library on this path, so a counting allocator sees +// every byte and a game's own heap can own all of it. +// +// It is the shape TableBlockAllocator already has (§19.1): two function +// pointers and a context the caller carries. What it adds is a CONTRACT ON +// alloc — the bytes come back ZEROED. Lock copies whole nodes, PADDING +// INCLUDED, so anything left uninitialized reaches a packed region; the default +// pair reaches that through calloc, which costs nothing measurable because a +// fresh segment is untouched pages either way. +struct TableAllocator +{ + void * ( *alloc )( void * context, int64_t bytes ); // ZEROED bytes, NULL on failure + void ( *free )( void * context, void * pointer ); + void * context; +}; + +// The default pair, and it is the one every entry point takes when the caller +// names none. It calls schema_allocate / schema_release, so a program with its +// own C-library replacement can move the floor without writing a struct at all. +inline void * table_default_alloc( void * context, int64_t bytes ) { (void) context; return schema_allocate( bytes ); } +inline void table_default_free( void * context, void * pointer ) { (void) context; schema_release( pointer ); } + +inline TableAllocator TableDefaultAllocator() +{ + TableAllocator allocator; + allocator.alloc = table_default_alloc; + allocator.free = table_default_free; + allocator.context = NULL; + return allocator; +} + +// ---- TableRef: a relocatable reference (never a machine pointer) ---- +// +// Two encodings, one slot, and the FORM says which is in force: +// +// in the arena — the node's arena offset (segment index in the high bits) +// in a region — the SELF-RELATIVE byte delta from this slot's own address, +// so a deref is one add, needs no base pointer, and a whole +// region relocates by memcpy with zero fix-up +// +// 0 is null in both, and a slot can never name the node that contains it, so +// zero names nothing real in either form. +// +// A REGION DELTA HAS NO REQUIRED SIGN (§6.3). A region is packed depth-first, +// so a node's FIRST reference points forward; every LATER reference to that +// same node points BACK at the one body it already has, which is exactly what +// makes one node one node in a region. Sharing and a back-reference are the +// same fact, and nothing validates a reference by its sign. +// +// IT IS EIGHT BYTES, SIGNED, so ONE REGION REACHES EVERYTHING (§6.3, §7): a +// four-byte slot bounded a region at 2 GiB, and the scale a cook exists for is +// *"100mbs or many gigabytes of data in Assets.bin"*. +struct TableRef +{ + int64_t value = 0; + bool null() const { return value == 0; } +}; + +// TableSlot is what Alloc hands back: usable as the node pointer (write +// fields through it) AND as the reference to store in a pointer field. +template struct TableSlot +{ + T * ptr = NULL; + TableRef ref; + T * operator->() const { return ptr; } + T & operator*() const { return *ptr; } + operator T *() const { return ptr; } + operator TableRef() const { return ref; } + bool null() const { return ptr == NULL; } +}; + +inline uint32_t TableAlignUp( uint32_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( kTableAlign - 1 ); } +inline int64_t TableAlignUp64( int64_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( int64_t( kTableAlign ) - 1 ); } + +// ---- a BYTE BUFFER's node (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// A *bytes or *string slot is a TableRef like every pointer slot, and it names +// a BLOB NODE: this eight-byte header and then the bytes, at offset eight so +// the data is eight-aligned. A *string blob carries one more zero byte after +// its data, so a region hands back a C string with no copy. The node's extent +// is the header plus its bytes, rounded to the arena's alignment like every +// node's; on the wire it is a record whose body is the bytes (§3.1). +struct TableBlob +{ + uint32_t length; + uint32_t zero; +}; + +static const int64_t kTableBlobHeader = 8; // length (u32), then four zero bytes +static const int64_t kTableBlobMaxLength = 0xFFFFFFFF; // a record's length is a u32 (§3.1) + +// the node's storage: the header, the bytes, a string's terminator, rounded +// to the arena's alignment like every node +inline int64_t TableBlobStorage( int64_t length, bool terminated ) +{ + return TableAlignUp64( kTableBlobHeader + length + ( terminated ? 1 : 0 ) ); +} + +// What a read answers: a pointer INTO the region and the length, NULL and +// zero for a null slot. Off a locked region, a loaded one or an opened cook +// the pointer is one add from the slot, and nothing is copied. +struct TableBytesView +{ + const uint8_t * data; + int64_t length; +}; + +struct TableStringView +{ + const char * data; // zero-terminated + int64_t length; +}; + +// What AllocBytes and AllocString hand back: the bytes to write through, the +// length asked for, and the reference to store in the slot — the three +// answers TableSlot gives for a table node. +struct TableBytesSlot +{ + uint8_t * data = NULL; + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +struct TableStringSlot +{ + char * data = NULL; // room for length bytes and the terminator, already zero + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +// ---- the arena: segmented, slab-handed, lock-free by ownership ---- +// +// Allocation is thread-local inside a worker's slab — no atomics on the node +// path. A worker takes its next slab with ONE compare-exchange, and a new +// segment is published with one more. Nothing ever moves: a segment, once +// allocated, lives untouched until the arena is torn down, so a T* obtained +// from Alloc stays valid while other workers allocate, and an offset stays +// correct while the arena grows. +// +// The model this DELIBERATELY refuses: one buffer under a lock, grown by +// realloc. A realloc moves the buffer under workers mid-write; offsets fix +// identity but not the raw references already resolved from them, and the +// resulting corruption is invisible until much later. Segments never move, so +// that bug class cannot be written here. +// +// Slack: at most one slab tail per worker plus one slab per segment (a slab +// that will not fit is skipped rather than split), i.e. under 2% of a segment +// plus threads x 64 KiB. That is the price of never synchronizing per node. +struct TableArena +{ + std::atomic segments[ kTableMaxSegments ]; + std::atomic cursor; // (segment << kTableSegmentBits) | bytes handed out + bool locked = false; // MONOTONIC: Lock() is one-way, there is no unlock + // THE ARENA CARRIES ITS OWN, so everything downstream of a builder — + // segments, pack map, numbering, region, node directory — allocates through + // the one pair the caller named, with nothing to thread by hand. + TableAllocator allocator; +}; + +inline void TableArenaInit( TableArena & arena, TableAllocator allocator ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + arena.segments[i].store( NULL, std::memory_order_relaxed ); + } + arena.cursor.store( 0, std::memory_order_relaxed ); + arena.locked = false; + arena.allocator = allocator; +} + +inline void TableArenaShutdown( TableArena & arena ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + uint8_t * segment = arena.segments[i].exchange( NULL, std::memory_order_acq_rel ); + if ( segment != NULL ) { arena.allocator.free( arena.allocator.context, segment ); } + } + arena.cursor.store( 0, std::memory_order_relaxed ); +} + +// one L1 load plus an add: the segment table is 8 KiB and stays hot +inline uint8_t * TableArenaAt( const TableArena & arena, uint32_t offset ) +{ + return arena.segments[ offset >> kTableSegmentBits ].load( std::memory_order_relaxed ) + ( offset & kTableSegmentMask ); +} + +// TableArenaGrabSlab hands one worker its next private slab. Returns +// kTableAllocFailed when the arena's address space or the allocator is +// exhausted — a loud refusal, never a silent smaller slab. +inline uint32_t TableArenaGrabSlab( TableArena & arena ) +{ + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t segment = cursor >> kTableSegmentBits; + uint32_t used = cursor & kTableSegmentMask; + // strictly less: a slab is never split across segments, and the tail + // is the documented slack + if ( used + kTableSlabBytes < kTableSegmentSize ) + { + if ( arena.segments[segment].load( std::memory_order_acquire ) == NULL ) + { + // THE SEGMENT COMES BACK ZEROED, which is the allocator's + // contract and not an extra pass here: Lock copies whole nodes, + // PADDING INCLUDED, so anything uninitialized reaches a packed + // region. Value-initializing a node with placement new zeroes + // its MEMBERS and not its padding, so the zeroing has to happen + // at the segment or not at all. It costs nothing measurable: a + // fresh segment is untouched pages either way, and the default + // pair's calloc has the kernel hand them over zeroed. + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, (int64_t) kTableSegmentSize ); + if ( memory == NULL ) { return kTableAllocFailed; } + uint8_t * expected = NULL; + if ( !arena.segments[segment].compare_exchange_strong( expected, memory, std::memory_order_acq_rel ) ) + { + // another worker published this segment first + arena.allocator.free( arena.allocator.context, memory ); + } + } + if ( arena.cursor.compare_exchange_weak( cursor, cursor + kTableSlabBytes, std::memory_order_acq_rel ) ) + { + return ( segment << kTableSegmentBits ) | used; + } + continue; + } + uint32_t next_segment = segment + 1; + if ( next_segment >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + arena.cursor.compare_exchange_weak( cursor, next_segment << kTableSegmentBits, std::memory_order_acq_rel ); + } +} + +// TableArenaGrabSpan reserves a SPAN of the arena's address space for one node +// larger than a slab — a BYTE BUFFER of any size (docs/SPEC-TABLES.md §2.5) — +// and allocates it as one contiguous block. It takes whole segment indices +// from the cursor, starting at the index after the cursor's so nothing else +// is ever handed out inside the span, and publishes the block under the first +// of them; the indices the span covers past that one stay NULL, which is +// enough, because only a node's START is ever resolved through the segment +// table and a blob's bytes follow its header inside the one allocation. The +// unused tail of the segment the cursor was in is slack, like a slab tail. +// Returns kTableAllocFailed when the address space or the allocator is +// exhausted — a loud refusal, never a smaller blob. +inline uint32_t TableArenaGrabSpan( TableArena & arena, int64_t bytes ) +{ + if ( bytes <= 0 || bytes > ( (int64_t) kTableMaxSegments - 2 ) * (int64_t) kTableSegmentSize ) { return kTableAllocFailed; } + const uint32_t spanned = (uint32_t) ( ( bytes + kTableSegmentSize - 1 ) >> kTableSegmentBits ); + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t start = ( cursor >> kTableSegmentBits ) + 1; + if ( start + spanned >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + uint32_t next = ( start + spanned ) << kTableSegmentBits; + if ( !arena.cursor.compare_exchange_weak( cursor, next, std::memory_order_acq_rel ) ) { continue; } + // the span is this worker's now: nothing else can publish under its + // first index, so a plain store suffices, and the block comes back + // ZEROED like every segment — the blob's bytes and its tail are zeros + // until written + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, bytes ); + if ( memory == NULL ) { return kTableAllocFailed; } + arena.segments[start].store( memory, std::memory_order_release ); + return start << kTableSegmentBits; + } +} + +// ---- TableWorker: one thread's allocation front ---- +// +// The threading contract, stated plainly: +// * Alloc on YOUR OWN worker is safe concurrently with any other worker's. +// No locks, no atomics per node. +// * Writing fields of a node ANOTHER worker allocated is your own +// synchronization problem — this runtime does not arbitrate it. +// * Lock and Save are single-threaded: call them after the workers have +// joined. +struct TableWorker +{ + TableArena * arena = NULL; + uint32_t next = 0; + uint32_t end = 0; + + template TableSlot Alloc() + { + static_assert( alignof( T ) <= kTableAlign, "a table node's alignment must fit the arena's" ); + TableSlot slot; + if ( arena == NULL || arena->locked ) { return slot; } + uint32_t bytes = TableAlignUp( (uint32_t) sizeof( T ) ); + if ( bytes > kTableSlabBytes ) { return slot; } // a node larger than a slab: refused, never split + if ( end == 0 || next + bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return slot; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + uint32_t at = next; + next += bytes; + // A NODE IS BORN IN TWO HALVES: start its lifetime in the raw + // storage, then write the declared defaults ONE MEMBER AT A TIME. + // + // It is "T", not "T{}". Value-initialising the whole aggregate says + // the same thing and costs cl O(BYTES) TO COMPILE — it expands element + // by element in its front end — while both halves here cost + // O(declarations). The slab cap below refuses a large node at RUN + // TIME and bounds nothing at compile time: the cost is paid by + // whatever T a caller instantiates this with. + // Padding is not the difference: value-initialisation zeroes MEMBERS + // and not padding either way, which is why the segment is calloc'd. + // + // TableReset is an OVERLOAD SET, one per closure member, reached from + // this template by argument-dependent lookup on T's own namespace — + // Alloc is a template and cannot spell Reset. + // + // The reset is here because ONE DEFINITION SAYS WHAT THE DECLARED + // DEFAULTS ARE, and it is Reset. Default-initialisation lands on + // the same values today, because a member with a non-zero default + // carries a member initializer that says so — but that is the class + // definition agreeing with Reset, not the arena reading it, and #320's + // fix was itself a pass that MOVED initialisation between the two. + // The arena reads the definition. + slot.ptr = new ( TableArenaAt( *arena, at ) ) T; + TableReset( *slot.ptr ); + slot.ref.value = at; + return slot; + } + + // Alloc a BYTE BUFFER's node of exactly length bytes (docs/SPEC-TABLES.md + // §2.5): the blob header and its bytes, zeroed, in this thread's slab when + // it fits and in a span of the arena's own when it does not. NULL is the + // arena locked, a length below zero or past a record's u32, or the + // allocator refusing. The offset comes back for the reference. + TableBlob * AllocBlob( int64_t length, bool terminated, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( length < 0 || length > kTableBlobMaxLength ) { return NULL; } + const int64_t bytes = TableBlobStorage( length, terminated ); + if ( bytes > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, bytes ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + } + else + { + if ( end == 0 || next + (uint32_t) bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) bytes; + } + TableBlob * blob = (TableBlob *) TableArenaAt( *arena, at ); + blob->length = (uint32_t) length; // the bytes after it are the segment's zeros + blob->zero = 0; + return blob; + } + + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries + // no type id, takes no index and has no Reset, so it goes through the same + // slab and span the blob path uses rather than through Alloc. + uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( bytes <= 0 || align > (int64_t) kTableAlign ) { return NULL; } + const int64_t rounded = TableAlignUp64( bytes ); + if ( rounded > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, rounded ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + return TableArenaAt( *arena, at ); + } + if ( end == 0 || next + (uint32_t) rounded > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) rounded; + return TableArenaAt( *arena, at ); // the segment came back zeroed + } + // a *bytes node: the bytes to write through, and the reference to store + TableBytesSlot AllocBytes( int64_t length ) + { + TableBytesSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, false, at ); + if ( blob == NULL ) { return slot; } + slot.data = (uint8_t *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } + + // a *string node: room for length bytes and the zero byte after them + TableStringSlot AllocString( int64_t length ) + { + TableStringSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, true, at ); + if ( blob == NULL ) { return slot; } + slot.data = (char *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } +}; + +// ---- TablePackMap: the pack walk's identity map (docs/SPEC-TABLES.md §3.1, §6.2) ---- +// +// ONE ENTRY PER REACHABLE NODE, and that map IS identity: a node must know +// where it landed to be named a second time, so Lock packs a shared node ONCE +// and every later reference resolves to the one body it already has. That is +// the same first-visit numbering the wire uses, so the pack order and the node +// order are one order. +// +// COLOURING AN ENTRY WHILE ITS DESCENT IS OPEN COSTS ONE BIT, and it is what +// makes a data cycle free to refuse: a reference to an entry still open is a +// cycle, and Lock returns failure rather than recursing away. The ROOT's entry +// is open for the whole walk. +// +// The map is proportional to NODES, never to bytes, and it lives on the +// AUTHORING side, where §6.5 licenses allocation. Nothing on the reading path +// ever builds one. +struct TablePackEntry +{ + const void * key; // the node's address in the graph being packed + int64_t offset; // where that node landed in the region + uint8_t open; // its descent is still open: a reference here is a cycle +}; + +struct TablePackMap +{ + TablePackEntry * entries = NULL; + int64_t capacity = 0; // a power of two, or zero while empty + int64_t count = 0; + TableAllocator allocator; // the caller's, carried from the walk that built it +}; + +inline void TablePackMapInit( TablePackMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TablePackMapShutdown( TablePackMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TablePackMapInit( map, map.allocator ); +} + +// The two walks behind Lock re-derive the SAME map from the same graph — the +// numbering is never carried between them (§3.1) — so the second starts from +// an empty map and keeps the capacity the first paid for. +inline void TablePackMapReset( TablePackMap & map ) +{ + if ( map.entries != NULL ) { memset( map.entries, 0, (size_t) map.capacity * sizeof( TablePackEntry ) ); } + map.count = 0; +} + +// open addressing, linear probing, a multiply-shift hash over the address: a +// node key is a pointer and its low bits are alignment, so the low bits alone +// would collide on every node of one type +inline int64_t TablePackMapSlot( const TablePackMap & map, const void * key ) +{ + uint64_t hash = (uint64_t) (uintptr_t) key; + hash *= 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != NULL && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TablePackEntry * TablePackMapFind( TablePackMap & map, const void * key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TablePackEntry * entry = &map.entries[ TablePackMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +// QUADRUPLING, not doubling, and the reason is measured: growth rehashes every +// entry, and on a graph of 131,071 nodes the doubling schedule spent 45% of +// Lock in rehashing alone. Quadrupling from 1024 buys 1.35x on that graph and +// keeps the map NODE-proportional (§6.2) — under 128 bytes a node at its +// worst, right after a grow, and about 64 on average. +inline bool TablePackMapGrow( TablePackMap & map ) +{ + TablePackMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 1024; + grown.entries = (TablePackEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TablePackEntry ) ); + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == NULL ) { continue; } + grown.entries[ TablePackMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// REACH a node: one probe answers both questions the walk has. A true "taken" +// says this is a FIRST visit, and the entry is now the node's, coloured open +// at "offset"; otherwise the entry is the one the node already has, and its +// open bit says cycle or sharing. NULL is an allocation failure, and it is a +// refusal like any other: Lock fails rather than packing a graph it cannot +// track. +// +// It is one call and not a find followed by an insert because the walk asks +// this question twice per node — once to measure, once to pack — and every +// probe is a miss into a table larger than L2. +inline TablePackEntry * TablePackMapReach( TablePackMap & map, const void * key, int64_t offset, bool & taken, int64_t & slot ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TablePackMapGrow( map ) ) { return NULL; } + } + slot = TablePackMapSlot( map, key ); + TablePackEntry * entry = &map.entries[slot]; + taken = entry->key != key; // an empty slot is a first visit; the key is never NULL + if ( taken ) + { + entry->key = key; + entry->offset = offset; + entry->open = 1; + map.count++; + } + return entry; +} + +// The descent finished: the node keeps its entry — identity outlives the +// descent — and stops being a cycle. The "hint" is the slot Reach returned, and it +// is checked against the key rather than trusted, so a rehash between the two +// costs a second probe instead of correctness. +inline void TablePackMapClose( TablePackMap & map, const void * key, int64_t hint ) +{ + if ( hint >= 0 && hint < map.capacity && map.entries[hint].key == key ) + { + map.entries[hint].open = 0; + return; + } + TablePackEntry * entry = TablePackMapFind( map, key ); + if ( entry != NULL ) { entry->open = 0; } +} + +// ---- resolution contexts: which encoding a walk is reading ---- + +struct TableArenaCtx { const TableArena * arena; }; +struct TableRegionCtx {}; + +// ---- a BYTE BUFFER's resolution (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// The same two encodings a table pointer has, resolved the same way: a +// self-relative delta in a region — one add, no base — and an arena offset +// while the builder is mutable. The blob is reached through its header, and a +// view is the header plus eight and the header's first word. Nothing here +// allocates and nothing copies: off a locked region, a loaded one or an +// opened cook the view points INTO the region. +inline const TableBlob * TableBlobAt( const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableRegionCtx &, const TableRef & ref ) { return TableBlobAt( ref ); } +inline const TableBlob * TableBlobAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} + +inline TableBytesView TableBytesViewOf( const TableBlob * blob ) +{ + TableBytesView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const uint8_t *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} +inline TableStringView TableStringViewOf( const TableBlob * blob ) +{ + TableStringView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const char *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} + +// the const form's hot path: one add, no base +inline TableBytesView TableBytesAt( const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ref ) ); } +inline TableStringView TableStringAt( const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ref ) ); } +// and the context forms a walk uses: a region context, an arena context, or +// the arena itself while the builder is mutable +template inline TableBytesView TableBytesAt( const Ctx & ctx, const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ctx, ref ) ); } +template inline TableStringView TableStringAt( const Ctx & ctx, const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ctx, ref ) ); } + +// allocate a blob in the arena and point the slot at it; the slot holds the +// arena offset, as every slot does while the builder is mutable +inline uint8_t * TableBytesEmplace( TableWorker & worker, TableRef & slot, int64_t length ) +{ + TableBytesSlot allocated = worker.AllocBytes( length ); + slot = allocated.ref; + return allocated.data; +} +// the text is copied in when one is given; a NULL text leaves the zeros for +// the caller to fill +inline char * TableStringEmplace( TableWorker & worker, TableRef & slot, const char * text, int64_t length ) +{ + TableStringSlot allocated = worker.AllocString( length ); + slot = allocated.ref; + if ( allocated.data != NULL && text != NULL && length > 0 ) { memcpy( allocated.data, text, (size_t) length ); } + return allocated.data; +} + +// ---- the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table, and a +// pointer field rides as an INDEX into it under kind 17. The encoding is +// flat: no pointer edge is a nesting level, so a chain's length is not a depth, +// and two references to one node are one node. +// +// THE FIELD RIDES ONCE: an L with sixty-four bits of capability frames a +// numbering of any size, so the whole numbering is one contiguous payload and a +// save's node bodies have no aggregate ceiling. + +static const uint64_t kTableNodeIndexNull = 0; // absence and null are one value +static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts the table + +// The not-materialized sentinel (§6.3): a record whose type id this build could +// not name. Distinct from every real offset including the root's 0, so an index +// resolving through it yields NULL and can never fabricate the root. +static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; + +// ---- the numbering, on the SAVE side ---- +// +// One entry per reachable node in FIRST-VISIT order, so entry k is node index +// k + 2. The two thunks are what let one loop write a table of mixed types: the +// numbering walk knows each target's type STATICALLY at the site it numbers it, +// so it stores the instantiation there and the loop never asks what a node is. +struct TableNumbering; + +struct TableNodeEntry +{ + const void * node; + uint64_t type_id; + // the type id's MESSAGE-FORM SLOT (docs/SPEC-TABLES.md §3.3), stored where + // the numbering walk stores the id itself and for the same reason: the + // target's type is known STATICALLY at the site that numbers it, so a + // form 2 save reads the slot out of the entry instead of looking an id up. + // Every pointer target's type id is an entry of the announcement, which is + // what makes the slot a compile-time fact of a POINTERED message too. + uint64_t type_slot; + int64_t ( * measure )( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ); + bool ( * save )( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ); +}; + +struct TableNumbering +{ + TablePackMap seen; // node -> index; the ROOT is index 1, open for the whole walk + TableNodeEntry * entries = NULL; + int64_t count = 0; + int64_t capacity = 0; +}; + +// The numbering allocates through the map's pair rather than carrying a second +// copy of it: one numbering is one walk, and a walk has one allocator. +inline void TableNumberingInit( TableNumbering & n, TableAllocator allocator ) +{ + TablePackMapInit( n.seen, allocator ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +inline void TableNumberingShutdown( TableNumbering & n ) +{ + TableAllocator allocator = n.seen.allocator; + TablePackMapShutdown( n.seen ); + allocator.free( allocator.context, n.entries ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +// The index a numbered node was given, for the save that writes it into a +// pointer slot. False means the two walks disagree about the graph, which is a +// refusal and never a guess. +inline bool TableNumberingIndex( const TableNumbering & n, const void * node, uint64_t & index ) +{ + if ( n.seen.capacity == 0 ) { return false; } + const TablePackEntry & entry = n.seen.entries[ TablePackMapSlot( n.seen, node ) ]; + if ( entry.key != node ) { return false; } + index = (uint64_t) entry.offset; + return true; +} + +inline bool TableNumberingAppend( TableNumbering & n, const TableNodeEntry & entry ) +{ + if ( n.count == n.capacity ) + { + // GROW BY COPY, never by realloc: the allocator hook is a PAIR, and a + // game's heap is not required to have a resize primitive at all. The + // schedule quadruples, so the copying is amortized to a constant per + // entry and the growth is the same growth it always was. + int64_t capacity = n.capacity != 0 ? n.capacity * 4 : 256; + TableAllocator allocator = n.seen.allocator; + TableNodeEntry * grown = (TableNodeEntry *) allocator.alloc( allocator.context, capacity * (int64_t) sizeof( TableNodeEntry ) ); + if ( grown == NULL ) { return false; } + if ( n.entries != NULL ) + { + memcpy( grown, n.entries, (size_t) n.count * sizeof( TableNodeEntry ) ); + allocator.free( allocator.context, n.entries ); + } + n.entries = grown; + n.capacity = capacity; + } + n.entries[n.count++] = entry; + return true; +} + +// The thunks the numbering stores. Each resolves to the closure member's own +// MeasureBody / SaveBodyFields through an overload set in the member's DECLARING +// file, reached by argument-dependent lookup at instantiation — the same bridge +// the arena's TableReset uses, and the reason a numbering may span the files of +// one unit without any file naming another's members. +template +inline int64_t TableNodeMeasureThunk( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ) +{ + return TableNodeMeasure( *(const Ctx *) ctx, numbering, ids, *(const T *) node ); +} + +template +inline bool TableNodeSaveThunk( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ) +{ + return TableNodeSave( *(const Ctx *) ctx, numbering, w, ids, *(const T *) node ); +} + +// ---- a BYTE BUFFER's record (docs/SPEC-TABLES.md §2.5, §3.1) ---- +// +// A blob rides as a node record under one of two RESERVED type ids — the fold +// a table's name takes, over the keywords "bytes" and "string", which no table +// can be named — with the bytes as its body and nothing framed inside. These +// two thunks are what the numbering stores for a blob, as it stores a +// member's codec for a table: the length, and the bytes verbatim. +static const uint64_t kTableBytesTypeId = 0x2f2ec0474f1c4fe4ull; // fnv1a64( "bytes" ) +static const uint64_t kTableStringTypeId = 0x704be0d8faaffc58ull; // fnv1a64( "string" ) + +template +inline int64_t TableBlobMeasureThunk( const void *, const TableNumbering &, TableIds &, const void * node ) +{ + return (int64_t) ( (const TableBlob *) node )->length; +} + +template +inline bool TableBlobSaveThunk( const void *, const TableNumbering &, TableWriter & w, TableIds &, const void * node ) +{ + const TableBlob * blob = (const TableBlob *) node; + w.raw( (const void *) ( blob + 1 ), (int64_t) blob->length ); + return true; +} + +// TableNodeTableMeasure and TableNodeTableSave are the framing, and they are +// ONE fill rule written twice — measure derives it from the graph and save +// derives the same one, which is what makes measure == save hold across a +// pointer graph (§3.1). +// +// The field rides ONCE, under the reserved id, kind 12: the payload opens with +// the count and then carries the records back to back, each a type id +// REFERENCE, a length and a body. The reserved id is interned BEFORE the +// records, and a record's type id before its body, which is the first-use order +// the trailer is written in (§3). +template +inline int64_t TableNodeTablePayload( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + int64_t payload = TableLebBytes( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + payload += TableLebBytes( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return -1; } + payload += TableLebBytes( (uint64_t) body ) + body; + } + return payload; +} + +template +inline int64_t TableNodeTableMeasure( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return 0; } // a root that reaches no nodes writes none of them + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return -1; } + return TableLebBytes( ref ) + 1 + TableLebBytes( (uint64_t) payload ) + payload; +} + +template +inline bool TableNodeTableSave( const Ctx & ctx, TableWriter & w, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return true; } + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return false; } + w.putleb( ref ); + w.put8( 12 ); // kind 12 is the opaque byte payload: a reader that cannot name the id skips by L + w.putleb( (uint64_t) payload ); + w.putleb( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + w.putleb( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return false; } + w.putleb( (uint64_t) body ); + if ( !n.entries[k].save( (const void *) &ctx, n, w, ids, n.entries[k].node ) ) { return false; } + } + return true; +} + +// ---- the numbering, on the LOAD side: a region's NODE DIRECTORY (§6.3) ---- +// +// The wire's numbering made resident: one entry per numbered node, in index +// order, position i describing node index i + 1 — so position 0 is the ROOT at +// offset 0. It is ATTRIBUTION, and attribution is separable: nothing that reads +// a structure touches it, a deref is one add on a self-relative offset, and a +// caller may release it once Load returns. +struct TableNodeDirEntry +{ + uint64_t offset; + uint64_t type_id; +}; + +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; + +// TableNodeMap is what a pointer slot resolves through while a body decodes. +struct TableNodeMap +{ + uint8_t * base = NULL; + const TableNodeDirEntry * entries = NULL; + int64_t count = 0; // the ROOT's entry included, so it is records + 1 + bool good = false; // the node table read whole; a numbering that failed resolves nothing + // WHERE THE NODES LIVE, and therefore what a resolved slot holds: a region + // takes the SELF-RELATIVE delta so a deref is one add, and the tool's + // builder path takes the node's ARENA OFFSET (§6.3). + bool arena = false; + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. + TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; +}; + +// TableNodeResolve places one node index in a pointer slot, and every failure +// is one of §4's events with the pointer left null. The declared TARGET type id +// is checked at every index, the root's included: the root carries no record +// and therefore no wire type id, so the READER'S OWN root type is what the +// claim is checked against. +inline void TableNodeResolve( const TableNodeMap & map, TableRef & slot, uint64_t index, uint64_t target, TableReport * report ) +{ + slot.value = 0; + if ( index == kTableNodeIndexNull || !map.good ) { return; } + if ( index - 1 >= (uint64_t) map.count ) + { + report->malformed = true; // an index above node_count + 1 + return; + } + const TableNodeDirEntry & entry = map.entries[index - 1]; + if ( entry.offset == kTableNodeAbsent ) + { + // a node whose type id this build could not name KEEPS ITS INDEX, and + // every pointer naming it reads null. The unknown was counted once, at + // the node, not once per pointer. + return; + } + if ( entry.type_id != target ) + { + report->kind_mismatch++; + return; + } + slot.value = map.arena ? (int64_t) entry.offset + : (int64_t) ( ( map.base + entry.offset ) - (const uint8_t *) &slot ); +} + +// ---- the record SCAN, and it is the whole of load's bound (§3.1) ---- +// +// Reading follows no reference. The scan walks the root body's top-level fields, +// finds the ONE under the reserved id, and reads records out of its payload in +// order — the field rides once, so nothing is copied to make a body contiguous +// and the generated body decoder never learns the transport exists. +struct TableNodeScan +{ + TableReader fields; // over the ROOT body, skipping past everything else + const uint8_t * payload; // the node-table field's payload + int64_t payload_size; + int64_t payload_offset; + bool opened; // the root body has been walked for the field + uint64_t declared; + int64_t records; + bool present; // the root body carries a node table at all + bool malformed; + const TableIdTable * ids; +}; + +inline TableNodeScan TableNodeScanBegin( const uint8_t * body, int64_t size, TableReport * report, const TableIdTable * ids ) +{ + TableNodeScan s = { TableReader( body, size, report, ids ), NULL, 0, 0, false, 0, 0, false, false, ids }; + return s; +} + +// find the node-table field, or answer false when the root body has none. A +// body carrying an id more than once is legal input and THE LAST OCCURRENCE +// WINS (docs/SPEC-TABLES.md §3), so the walk runs to the terminator and keeps +// the last rather than stopping at the first. +inline bool TableNodeScanOpen( TableNodeScan & s ) +{ + if ( s.opened ) { return false; } + s.opened = true; + for ( ;; ) + { + uint64_t ref = 0; + if ( !s.fields.getleb( ref ) ) { break; } + if ( ref == 0 ) { break; } // the terminator + if ( s.ids == NULL || ref > (uint64_t) s.ids->count ) { break; } + const uint64_t id = s.ids->at( ref ); + if ( !s.fields.has( 1 ) ) { break; } + const uint8_t kind = s.fields.get8(); + if ( id == kTableNodeTableFieldId ) + { + s.present = true; + if ( kind != 12 ) { s.malformed = true; return false; } + uint64_t length = 0; + if ( !s.fields.getleb( length ) || !s.fields.room( length ) ) { s.malformed = true; return false; } + s.payload = s.fields.buffer + s.fields.offset; + s.payload_size = (int64_t) length; + s.fields.offset += (int64_t) length; + continue; + } + if ( !s.fields.skip( kind ) ) { break; } + } + if ( s.payload == NULL ) { return false; } + TableReader head( s.payload, s.payload_size, s.fields.report, s.ids ); + if ( !head.getleb( s.declared ) ) { s.malformed = true; return false; } + s.payload_offset = head.offset; + return true; +} + +// the next record, or false at the end of the table — s.malformed says whether +// the end was the end or the framing giving out +inline bool TableNodeScanNext( TableNodeScan & s, uint64_t & type_id, const uint8_t * & body, int64_t & length ) +{ + if ( !s.opened && !TableNodeScanOpen( s ) ) { return false; } + if ( s.payload == NULL || s.payload_offset >= s.payload_size ) { return false; } + TableReader rec( s.payload, s.payload_size, s.fields.report, s.ids ); + rec.offset = s.payload_offset; + uint64_t ref = 0; + if ( !rec.getleb( ref ) || ref == 0 || s.ids == NULL || ref > (uint64_t) s.ids->count ) + { + s.malformed = true; // a type id reference of 0, or one past the table + return false; + } + type_id = s.ids->at( ref ); + uint64_t declared_length = 0; + if ( !rec.getleb( declared_length ) ) + { + s.malformed = true; // a record whose length is damaged + return false; + } + if ( declared_length > (uint64_t) ( s.payload_size - rec.offset ) ) + { + s.malformed = true; // a record whose length runs past its field + return false; + } + body = s.payload + rec.offset; + length = (int64_t) declared_length; + s.payload_offset = rec.offset + length; + s.records++; + return true; +} + +// The record scan is AUTHORITATIVE: node_count is data from the wire, and a +// count that disagrees with the scan is malformed. Nothing is sized from it +// before the scan has confirmed it. +inline bool TableNodeScanWhole( TableNodeScan & s ) +{ + if ( s.malformed ) { return false; } + if ( !s.present ) { return true; } // no node table at all is not a broken one + return s.declared == (uint64_t) s.records; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_ARENA + +#ifndef LISTDEMO_SCHEMA_TABLE_EXTENT +#define LISTDEMO_SCHEMA_TABLE_EXTENT + +namespace listdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_EXTENT + +#ifndef LISTDEMO_SCHEMA_TABLE_MAP +#define LISTDEMO_SCHEMA_TABLE_MAP + +namespace listdemo { + +// ---- a MAP: a sorted entry array, and the lookup over it (§2.8) ---- +// +// On the wire, in a region and in a cook a map is an array of one generated +// ENTRY table held in ascending key order. What this adds is Find — a binary +// search over that array where it lies — and a builder that inserts, replaces +// and erases by key. Nothing here is stored: a region and a cook carry the +// array and the count, and not one byte about a hash or a probe. + +// entries carved from ONE call to the allocator pair; a new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableMapSegmentEntries = 32; + +// TableDeclRef names a type in an unevaluated context and is never defined — +// what 's declval is for, without the include the generated corpus +// refuses to pay for (the iterator_traits note, §13.9). +template T & TableDeclRef(); + +// THE ORDER IS TOTAL, AND IT IS THE SAME IN NINE LANGUAGES (§2.8). Integers +// compare by VALUE, signed for the signed kinds and unsigned for the unsigned. +// Strings compare by BYTES, unsigned, a shorter string that is a prefix of a +// longer one first: memcmp over the common length, then the lengths. Never a +// locale, never a code point, never a case fold. +inline int TableKeyOrder( uint64_t a, uint64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( int64_t a, int64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( const char * a, int32_t a_length, const char * b, int32_t b_length ) +{ + const int32_t common = a_length < b_length ? a_length : b_length; + if ( common > 0 ) + { + const int order = memcmp( (const void *) a, (const void *) b, (size_t) common ); + if ( order != 0 ) { return order < 0 ? -1 : 1; } + } + return a_length < b_length ? -1 : ( a_length > b_length ? 1 : 0 ); +} + +// the length of a NUL-terminated key at a call site, bounded by the storage it +// has to fit: a key one byte longer than the bound is refused, never truncated +inline int32_t TableKeyLength( const char * key, int32_t bound ) +{ + if ( key == NULL ) { return 0; } + for ( int32_t i = 0; i <= bound; i++ ) { if ( key[i] == 0 ) { return i; } } + return bound + 1; // longer than the bound: the caller refuses it +} + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.8, §7.2) ---- +// +// An int64 self-relative reference to the entry array and an int32 count, then +// padding to eight. The reference is a TableRef like a pointer's: in the arena +// it names the builder's HEAD, in a region it is the delta from the slot to +// the first entry, and 0 is the empty map in both. +template struct TableMap +{ + TableRef entries; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Entry * Entries() const + { + return entries.value != 0 ? (const Entry *) ( (const uint8_t *) &entries + entries.value ) : NULL; + } + int32_t size() const { return count; } + + // FIND: floor( log2 n ) + 1 key compares, in place, no allocation. NULL + // when absent, and on a map[K]*T the RESOLVED pointer, which is what a + // pointer field's accessor answers. + template const Entry * FindEntry( Key key ) const + { + const Entry * base = Entries(); + int32_t low = 0, high = count; + while ( low < high ) + { + const int32_t mid = low + ( high - low ) / 2; + const int order = TableEntryOrder( base[mid], key ); + if ( order == 0 ) { return base + mid; } + if ( order < 0 ) { low = mid + 1; } else { high = mid; } + } + return NULL; + } + // the return type is DEDUCED, so it is worked out when a call site + // instantiates Find and not when the holder's record declares the slot — + // which is what lets the entry's own overloads be declared after it + template auto Find( Key key ) const + { + return TableEntryFound( FindEntry( key ) ); + } + + // ---- iteration: ASCENDING key order, the key beside the value ---- + // + // A proxy BY VALUE, the keyed array's shape (§2.4): for ( auto [ key, + // value ] : map ). It carries no iterator_traits, for the reason + // TableKeyed's does not (§13.9). + struct ConstEntry + { + decltype( TableEntryKey( TableDeclRef() ) ) key; + decltype( TableEntryFound( (const Entry *) NULL ) ) value; + }; + + struct ConstIterator + { + const Entry * at; + ConstEntry operator*() const { return ConstEntry{ TableEntryKey( *at ), TableEntryFound( at ) }; } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Entries() }; } + ConstIterator end() const { return ConstIterator{ Entries() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.8, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first entry is inserted. Each +// segment is a fixed number of entries carved from one call to the allocator +// pair. An entry's address is stable for the arena's life, so a value handed +// back by an insert stays valid while other entries arrive. +struct TableMapHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an insert appends into + int32_t live; + int32_t dead; +}; + +template struct TableMapSegment +{ + TableRef next; + int32_t used; // entries carved from this segment + int32_t padding; + uint32_t dead[ ( kTableMapSegmentEntries + 31 ) / 32 ]; // Erase marks one bit, never the entry + Entry entries[ kTableMapSegmentEntries ]; +}; + +inline bool TableMapSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// ---- the ORDERED CURSOR the four writing walks read (§2.8) ---- +// +// Measure, Save, Lock and Cook each write a map's entries in ascending key +// order with no key twice, deriving the order from the builder's entries as +// each walk derives the numbering (§3.1). Nothing passes between them, so +// measure == save over a map is a real check on two sorts agreeing. +// +// A REGION is already sorted, so its cursor is the array in place and +// allocates nothing. The BUILDER's is the sort: an array of entry pointers +// allocated through the pair and released before the walk returns, because +// sorting the segments themselves would move entries whose addresses a caller +// holds. +template struct TableMapCursor +{ + const Entry * const * order = NULL; // the builder's form: sorted pointers + const Entry * entries = NULL; // the region's form: the array in place + int32_t count = 0; + TableAllocator allocator; + bool ok = false; + const Entry * operator[]( int32_t index ) const + { + return order != NULL ? order[index] : entries + index; + } +}; + +// heapsort: O( n log n ) once per map, no recursion, no allocation past the +// pointer array the caller already paid for +template inline void TableMapSort( const Entry ** order, int32_t count ) +{ + for ( int32_t start = count / 2 - 1; start >= 0; start-- ) + { + int32_t root = start; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= count ) { break; } + if ( child + 1 < count && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * swap = order[root]; order[root] = order[child]; order[child] = swap; + root = child; + } + } + for ( int32_t end = count - 1; end > 0; end-- ) + { + const Entry * swap = order[0]; order[0] = order[end]; order[end] = swap; + int32_t root = 0; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= end ) { break; } + if ( child + 1 < end && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * hold = order[root]; order[root] = order[child]; order[child] = hold; + root = child; + } + } +} + +// the REGION form: the array is already sorted, so the cursor is the array +template +inline TableMapCursor TableMapOrder( const TableRegionCtx &, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.entries = map.Entries(); + cursor.count = map.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: gather the LIVE entries out of the segment chain in +// insertion order, then sort. A dead entry costs nothing on any wire (§2.8). +template +inline TableMapCursor TableMapOrder( const TableArena & arena, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.allocator = arena.allocator; + cursor.count = map.count; + if ( map.entries.value == 0 || map.count <= 0 ) { cursor.ok = map.count == 0; cursor.count = 0; return cursor; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + if ( head->live != map.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + const Entry ** order = (const Entry **) arena.allocator.alloc( arena.allocator.context, (int64_t) map.count * (int64_t) sizeof( const Entry * ) ); + if ( order == NULL ) { return cursor; } + int32_t at = 0; + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 && at < map.count ) + { + const TableMapSegment * segment = (const TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used && at < map.count; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + order[at++] = segment->entries + i; + } + segment_ref = segment->next; + } + if ( at != map.count ) + { + arena.allocator.free( arena.allocator.context, order ); + return cursor; + } + TableMapSort( order, map.count ); + cursor.order = order; + cursor.ok = true; + return cursor; +} + +template +inline TableMapCursor TableMapOrder( const TableArenaCtx & ctx, const TableMap & map ) +{ + return TableMapOrder( *ctx.arena, map ); +} + +template inline void TableMapRelease( TableMapCursor & cursor ) +{ + if ( cursor.order != NULL ) { cursor.allocator.free( cursor.allocator.context, (void *) cursor.order ); } + cursor.order = NULL; +} + +// ---- the builder's five (§2.8) ---- +// +// Insert APPENDS after one LINEAR SCAN of the live entries for the key it may +// replace, Find is that same scan, and Erase is the scan and one bit. The +// builder builds NO INDEX, and that is a rule: the sort happens once, at Lock, +// Save or Cook, and every lookup that matters runs over the sorted region. + +// the head, allocated when the first entry is inserted +template +inline TableMapHead * TableMapReach( TableWorker & worker, TableMap & map ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( map.entries.value != 0 ) { return (TableMapHead *) TableArenaAt( *worker.arena, (uint32_t) map.entries.value ); } + uint32_t at = 0; + TableMapHead * head = (TableMapHead *) worker.AllocRaw( (int64_t) sizeof( TableMapHead ), (int64_t) alignof( TableMapHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + map.entries.value = (int64_t) at; + return head; +} + +// one entry's storage, appended: the current segment when it has room, a new +// one carved from one call to the pair when it does not +template +inline Entry * TableMapAppend( TableWorker & worker, TableMapHead * head, TableMap & map ) +{ + TableMapSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableMapSegmentEntries ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableMapSegment *) worker.AllocRaw( (int64_t) sizeof( TableMapSegment ), (int64_t) alignof( TableMapSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableMapSegment * previous = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Entry * entry = segment->entries + segment->used; + segment->used++; + head->live++; + map.count++; + return entry; +} + +// the LINEAR SCAN: the live entries in insertion order, O( n ) key compares +template +inline Entry * TableMapScan( const TableArena & arena, const TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) == 0 ) { return segment->entries + i; } + } + segment_ref = segment->next; + } + return NULL; +} + +// ERASE marks the entry DEAD, one bit in the segment's slot and not in the +// entry table, and decrements the live count. Its storage is reclaimed at +// RESET and never reused mid-build, because reusing a slot would make "an +// entry's address is stable" false for exactly one case. +template +inline bool TableMapErase( TableArena & arena, TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return false; } + TableMapHead * head = (TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) != 0 ) { continue; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + map.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INSERTION order, live entries only (§2.8) ---- +template struct TableMapEach +{ + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableMapSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableMapSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + auto operator*() const { return TableEntryEach( segment->entries + index ); } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableMapEach TableMapEachOf( const TableArena & arena, const TableMap & map ) +{ + TableMapEach each = { &arena, TableRef() }; + if ( map.entries.value != 0 ) + { + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + each.first = head->first; + } + return each; +} + +// ---- the LOAD side: where a decoded entry lands (§2.8) ---- +// +// THE READER TRUSTS NOTHING and spends one compare per entry. Every load path +// applies the same rules and produces one report (§4), so the region load of +// §6.5 and LoadBuilder never disagree about a wire. These two shapes are what +// makes that true with one generated decoder: a REGION carves the entry array +// out of the holder node's own extent, and the TOOL's path appends into the +// builder's arena, and the decoder above them cannot tell which it has. + +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. + +// TableMapFill is one map field being decoded: where the next entry lands, and +// the entry that last LANDED, which is what the ascending check compares +// against. +template struct TableMapFill +{ + TableMap * map = NULL; + Entry * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; +}; + +template +inline TableMapFill TableMapFillBegin( const TableNodeMap & nodes, TableMap & map, uint32_t n ) +{ + TableMapFill fill; + fill.map = ↦ + map.entries.value = 0; + map.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Entry ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Entry ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Entry *) base; + fill.capacity = (int32_t) n; + map.entries.value = (int64_t) ( base - (const uint8_t *) &map.entries ); + fill.ok = true; + return fill; +} + +// the entry that last LANDED — NULL before the first +template inline Entry * TableMapFillLast( TableMapFill & fill ) +{ + if ( fill.map->count <= 0 ) { return NULL; } + if ( fill.array != NULL ) { return fill.array + ( fill.map->count - 1 ); } + return TableMapLive( *fill.worker->arena, *fill.map, fill.map->count - 1 ); +} + +// the next slot, at the entry type's declared defaults +template inline Entry * TableMapFillNext( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + if ( fill.map->count >= fill.capacity ) { return NULL; } + Entry * entry = fill.array + fill.map->count; + TableReset( *entry ); + fill.map->count++; + return entry; + } + TableMapHead * head = TableMapReach( *fill.worker, *fill.map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( *fill.worker, head, *fill.map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// A MAP WITH HALF ITS KEYS IS NOT A MAP (§2.8): at the first entry whose key +// kind disagrees with the reader's declaration the map resets to EMPTY, one +// kind_mismatch is counted for the map, and its remaining bytes are skipped. +template inline void TableMapFillReset( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + fill.map->entries.value = 0; + fill.map->count = 0; + return; + } + if ( fill.map->entries.value != 0 ) + { + TableMapHead * head = (TableMapHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.map->entries.value ); + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + } + fill.map->count = 0; +} + +// an EMPTY map's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableMapFillEnd( TableMapFill & fill ) +{ + if ( fill.array != NULL && fill.map->count == 0 ) { fill.map->entries.value = 0; } +} + +// the k-th LIVE entry of a builder map, in insertion order — what the tool +// path's ascending check compares against +template +inline Entry * TableMapLive( const TableArena & arena, const TableMap & map, int32_t index ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + int32_t at = 0; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( at == index ) { return segment->entries + i; } + at++; + } + segment_ref = segment->next; + } + return NULL; +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.8, §6.5) ---- +// +// LoadMeasure's term for a map is N x sizeof( Entry ) rounded to +// alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this +// reads no field: it walks the map's own header and, where an entry's value +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect, +// and a refusal carries its reason (§6.5). +// A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its +// own L and the body's terminator, and under this form's variable lengths that +// footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a +// map's L can carry, and therefore what a LoadMeasure may be asked for. +static const int64_t kTableMapEntryFloor = 2; + +inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry + at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); + at += (int64_t) n * entry_size; + if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// ---- the TEXT form's placement (docs/SPEC-TABLES.md §2.8, §16) ---- +// +// The text is a plain JSON object keyed by the KEY, and the generic walk fills +// it through the ENTRY'S OWN descriptor — so all it needs from here is one +// entry at one key, handed back at its defaults. It is the builder's Insert +// with the ENTRY returned rather than its value, because the walk writes the +// value through a field row and not through a typed pointer. +template +inline Entry * TableMapPlace( TableWorker & worker, TableMap & map, Key key ) +{ + if ( worker.arena == NULL ) { return NULL; } + Entry * found = TableMapScan( *worker.arena, map, key ); + if ( found != NULL ) + { + TableResetMapValue( *found ); // a repeated key is LAST-WINS, whole + return found; + } + TableMapHead * head = TableMapReach( worker, map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( worker, head, map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// ---- the OPTIONAL RUNTIME INDEX (§2.8) ---- +// +// Open addressing with LINEAR PROBING over the sorted array, built AT LOAD for +// a map large enough that log n compares over a cold array cost more than one +// hash and a probe. IT IS NEVER STORED: the caller measures it, owns its +// storage, builds it in one pass and releases it whenever. +// +// ITS HASH AND ITS LOAD FACTOR ARE NOT A CROSS-PORT CONTRACT, and that is a +// rule. What a port is held to is the CONTRACT of the lookup: the same value +// the sorted array's Find returns for the same key, and no allocation past the +// storage the caller handed in. +struct TableMapIndex +{ + int32_t * slots = NULL; // entry indices, +1; 0 is an empty slot + int32_t capacity = 0; + bool good = false; +}; + +// this runtime's own, and no port reproduces it: fnv1a64 over the key's bytes +inline uint64_t TableMapHash( const void * bytes, int32_t length ) +{ + uint64_t hash = 0xCBF29CE484222325ull; + const uint8_t * at = (const uint8_t *) bytes; + for ( int32_t i = 0; i < length; i++ ) { hash ^= (uint64_t) at[i]; hash *= 0x100000001B3ull; } + return hash; +} +inline uint64_t TableMapHash( uint64_t key ) { return TableMapHash( (const void *) &key, (int32_t) sizeof( key ) ); } + +// this runtime's own load factor, and no port reproduces it either: the next +// power of two at or above twice the count, so a probe run stays short +inline int32_t TableMapIndexSlots( int32_t count ) +{ + int32_t slots = 8; + while ( slots < count * 2 ) { slots *= 2; } + return slots; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_MAP + +#ifndef LISTDEMO_SCHEMA_TABLE_LIST +#define LISTDEMO_SCHEMA_TABLE_LIST + +namespace listdemo { + +// ---- an UNBOUNDED ARRAY: a counted array whose count the data decides (§2.9) ---- +// +// On the wire, in a region and in a cook a list is the kind 14 body a [..N]T +// writes, its elements by-value records inside the holder's node extent. What +// this adds is the slot, a builder that appends into segments that never +// move, and a const surface that indexes and iterates in place. There is no +// sort, no key and no lookup: the order is INSERTION order, and it is +// identity the way position is identity in a fixed array. + +// elements carved from ONE call to the allocator pair. A new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableListSegmentElements = 32; + +// THE ELEMENT STORAGE: T itself, and a TableRef slot for a []*T, whose +// elements are references exactly as a pointer field's slot is (§2.1) +template struct TableListStorage { typedef T Element; }; +template struct TableListStorage { typedef TableRef Element; }; + +// WHAT THE CONST FORM ANSWERS: the element by reference, and on a []*T the +// RESOLVED pointer, one add on the self-relative delta, NULL for a null slot, +// exactly as At answers it (§6.2, §6.3) +template struct TableListConst +{ + typedef const T & Result; + static Result At( const T * element ) { return *element; } +}; +template struct TableListConst +{ + typedef const T * Result; + static Result At( const TableRef * element ) + { + return element->value != 0 ? (const T *) ( (const uint8_t *) element + element->value ) : NULL; + } +}; + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.9, §7.2) ---- +// +// An int64 self-relative reference to the element array and an int32 count, +// then padding to eight. The reference is a TableRef like a pointer's: in the +// arena it names the builder's HEAD, in a region it is the delta from the slot +// to the first element, and 0 is the empty list in both. It is the map's slot +// exactly, because it is the same two facts. +template struct TableList +{ + typedef typename TableListStorage::Element Element; + + TableRef elements; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Element * Elements() const + { + return elements.value != 0 ? (const Element *) ( (const uint8_t *) &elements + elements.value ) : NULL; + } + int32_t size() const { return count; } + + // INDEXING IS BOUNDS-CHECKED IN EVERY BUILD (§2.4, §2.9): the extent is a + // number that CAME FROM A FILE, so an index past it is not a mistake a + // release build gets to make cheaply. There is no undefined-behavior path + // here in any configuration. The assert carries the message where a + // debugger can read it and NDEBUG removes that. The fatal is what stands + // after it. Both go through the hooks: define schema_assert and + // schema_fatal and this refusal lands in your own handler. + void RefuseIndex( int32_t index ) const + { + if ( (uint32_t) index >= (uint32_t) count ) + { + schema_assert( false && "an unbounded array is indexed inside its count, which came from a file" ); + schema_fatal(); + } + } + typename TableListConst::Result operator[]( int32_t index ) const + { + RefuseIndex( index ); + return TableListConst::At( Elements() + index ); + } + + // ---- iteration: INDEX order, the element and no key ---- + // + // It carries no iterator_traits, for the reason TableKeyed's does not + // (§13.9). + struct ConstIterator + { + const Element * at; + typename TableListConst::Result operator*() const { return TableListConst::At( at ); } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Elements() }; } + ConstIterator end() const { return ConstIterator{ Elements() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.9, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first element is added. Each +// segment is a fixed number of elements carved from one call to the allocator +// pair. An element's address is stable for the arena's life, so a T * handed +// back by Add stays valid while other elements arrive. +struct TableListHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an Add appends into + int32_t live; + int32_t dead; +}; + +template struct TableListSegment +{ + TableRef next; + int32_t used; // elements carved from this segment + int32_t padding; + uint32_t dead[ ( kTableListSegmentElements + 31 ) / 32 ]; // Erase marks one bit, never the element + Element elements[ kTableListSegmentElements ]; +}; + +inline bool TableListSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// the head, allocated when the first element is added +template +inline TableListHead * TableListReach( TableWorker & worker, TableList & list ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( list.elements.value != 0 ) { return (TableListHead *) TableArenaAt( *worker.arena, (uint32_t) list.elements.value ); } + uint32_t at = 0; + TableListHead * head = (TableListHead *) worker.AllocRaw( (int64_t) sizeof( TableListHead ), (int64_t) alignof( TableListHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + list.elements.value = (int64_t) at; + return head; +} + +// one element's storage, appended: the current segment when it has room, a +// new one carved from one call to the pair when it does not. NULL means NOT +// ADDED: an arena that cannot carve another segment, or a count at the int32 +// cap (§2.2, §2.9). +template +inline typename TableList::Element * TableListAppend( TableWorker & worker, TableListHead * head, TableList & list ) +{ + typedef typename TableList::Element Element; + if ( list.count >= INT32_MAX ) { return NULL; } // the int32 storage cap + TableListSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableListSegmentElements ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableListSegment *) worker.AllocRaw( (int64_t) sizeof( TableListSegment ), (int64_t) alignof( TableListSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableListSegment * previous = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Element * element = segment->elements + segment->used; + segment->used++; + head->live++; + list.count++; + return element; +} + +// ADD, whole: the head, the append, and the element at its declared defaults +// (§2.9). The text form's placement is this same call, because a list has no +// key to place under (§16). +template +inline typename TableList::Element * TableListPlace( TableWorker & worker, TableList & list ) +{ + typedef typename TableList::Element Element; + TableListHead * head = TableListReach( worker, list ); + if ( head == NULL ) { return NULL; } + Element * element = TableListAppend( worker, head, list ); + if ( element == NULL ) { return NULL; } + new ( element ) Element(); // value-init: the declared defaults, and null for a slot + return element; +} + +// ERASE, ADDRESSED BY THE POINTER (§2.9): the element Add handed back is the +// handle, because a list has no key and the address is the one thing the +// builder promises never moves (§6.4). It marks the element DEAD, one bit in +// the segment's slot and not in the element storage, and decrements the live +// count. False when the pointer is not this list's. Its storage is reclaimed +// at RESET and never reused mid-build, the map's rule for the map's reason. +template +inline bool TableListErase( TableArena & arena, TableList & list, const typename TableList::Element * element ) +{ + typedef typename TableList::Element Element; + if ( list.elements.value == 0 || element == NULL ) { return false; } + TableListHead * head = (TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableListSegment * segment = (TableListSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + if ( element >= segment->elements && element < segment->elements + segment->used ) + { + const int32_t i = (int32_t) ( element - segment->elements ); + if ( TableListSegmentDead( segment->dead, i ) ) { return false; } // already erased + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + list.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INDEX order, live elements only (§2.9) ---- +template struct TableListEach +{ + typedef typename TableList::Element Element; + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableListSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableListSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + Element * operator*() const { return segment->elements + index; } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableListEach TableListEachOf( const TableArena & arena, const TableList & list ) +{ + TableListEach each = { &arena, TableRef() }; + if ( list.elements.value != 0 ) + { + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + each.first = head->first; + } + return each; +} + +// ---- the INDEX-ORDER CURSOR the four writing walks read (§2.9) ---- +// +// Measure, Save, Lock and Cook each visit a list's live elements in the order +// they were added, and they allocate nothing to do it: a region's cursor is +// the array in place, and the builder's walks the segment chain. Indexing the +// builder's form is SEQUENTIAL by construction, every walk steps i, i + 1, +// i + 2, so the cursor remembers where the last access landed and moves one +// live slot per step. An access behind the memo restarts from the first +// segment, which no walk here does. +template struct TableListCursor +{ + const Element * elements = NULL; // the region's form: the array in place + const TableArena * arena = NULL; // the builder's form: the segments + TableRef first; + int32_t count = 0; + bool ok = false; + // the memo: the segment and slot the last access landed on, and the live + // index that slot holds + mutable const TableListSegment * segment = NULL; + mutable int32_t within = -1; + mutable int32_t logical = -1; + + const Element * At( int32_t index ) const + { + if ( elements != NULL ) { return elements + index; } + if ( segment == NULL || index < logical ) + { + segment = first.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL; + within = -1; + logical = -1; + } + while ( logical < index ) + { + for ( ;; ) + { + within++; + while ( segment != NULL && within >= segment->used ) + { + segment = segment->next.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + within = 0; + } + if ( segment == NULL ) { return NULL; } // the slot and the head disagree + if ( !TableListSegmentDead( segment->dead, within ) ) { break; } + } + logical++; + } + return segment->elements + within; + } + const Element & operator[]( int32_t index ) const { return *At( index ); } +}; + +// the REGION form: the array is the cursor +template +inline TableListCursor::Element> TableListElements( const TableRegionCtx &, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.elements = list.Elements(); + cursor.count = list.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: the live elements out of the segment chain, in the +// order they were added. A dead element costs nothing on any wire (§2.9). +template +inline TableListCursor::Element> TableListElements( const TableArena & arena, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.arena = &arena; + cursor.count = list.count; + if ( list.elements.value == 0 || list.count <= 0 ) { cursor.ok = list.count == 0; cursor.count = 0; return cursor; } + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + if ( head->live != list.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + cursor.first = head->first; + cursor.ok = true; + return cursor; +} + +template +inline TableListCursor::Element> TableListElements( const TableArenaCtx & ctx, const TableList & list ) +{ + return TableListElements( *ctx.arena, list ); +} + +// ---- the LOAD side: where a decoded element lands (§2.9) ---- +// +// The same two shapes the map's fill takes, because the decoder above them +// cannot tell which it has: a REGION carves the element array out of the +// holder node's own extent, PRE-ORDER, and the TOOL's path appends into the +// builder's arena. +template struct TableListFill +{ + typedef typename TableList::Element Element; + TableList * list = NULL; + Element * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; + bool refused = false; // a count above the int32 cap on the tool's path: LoadBuilder answers NULL +}; + +template +inline TableListFill TableListFillBegin( const TableNodeMap & nodes, TableList & list, uint64_t n ) +{ + typedef typename TableList::Element Element; + TableListFill fill; + fill.list = &list; + list.elements.value = 0; + list.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( n > (uint64_t) INT32_MAX ) + { + // A COUNT ABOVE THE int32 STORAGE CAP (§2.2, §2.9): into a region it was + // refused by LoadMeasure before this ran, and into a builder it is the + // refusal LoadBuilder answers NULL for, moving no counter + fill.refused = nodes.carve->worker != NULL; + return fill; + } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Element ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Element ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Element *) base; + fill.capacity = (int32_t) n; + list.elements.value = (int64_t) ( base - (const uint8_t *) &list.elements ); + fill.ok = true; + return fill; +} + +// the next slot, at the element's declared defaults. NULL when the arena +// could not carve, which the decoder reports as framing damage +template inline typename TableList::Element * TableListFillNext( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count >= fill.capacity ) { return NULL; } + Element * element = fill.array + fill.list->count; + new ( element ) Element(); + fill.list->count++; + return element; + } + return TableListPlace( *fill.worker, *fill.list ); +} + +// A SLOT WHOSE ELEMENT NEVER LANDED is given back (§2.9, §4): the array keeps +// what it decoded, and an element whose own framing gave out before one byte +// of it decoded was not decoded. The region's form uncounts it, and the builder's +// marks it dead, which is what the storage rule allows mid-build. +template inline void TableListFillDrop( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count > 0 ) { fill.list->count--; } + return; + } + if ( fill.list->elements.value == 0 ) { return; } + TableListHead * head = (TableListHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.list->elements.value ); + if ( head->last.value == 0 ) { return; } + TableListSegment * segment = (TableListSegment *) TableArenaAt( *fill.worker->arena, (uint32_t) head->last.value ); + if ( segment->used <= 0 ) { return; } + const int32_t i = segment->used - 1; + if ( TableListSegmentDead( segment->dead, i ) ) { return; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + fill.list->count--; +} + +// an EMPTY list's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableListFillEnd( TableListFill & fill ) +{ + if ( fill.array != NULL && fill.list->count == 0 ) { fill.list->elements.value = 0; } +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.9, §6.5) ---- +// +// N x sizeof( T ) rounded to alignof( T ), AT EVERY DEPTH. N is framing and +// not a value, so this reads no field: it walks the list's own header and, +// where a table element holds a list or a map of its own, the elements' +// headers under it. Every -1 carries its REASON (§6.5): the int32 cap first, +// because a count past it cannot fit any body, and then the body's own L. +inline bool TableListWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t elem_size, int64_t elem_align, uint8_t elem_kind, int64_t elem_floor, + TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != elem_kind ) { return true; } // another element kind: §4's ordinary kind mismatch, the field reads empty + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; } // an N the list's L cannot carry + at = ( at + elem_align - 1 ) & ~( elem_align - 1 ); + at += (int64_t) n * elem_size; + if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_LIST + +#ifndef LISTDEMO_SCHEMA_BUILD_VERSION +#define LISTDEMO_SCHEMA_BUILD_VERSION + +namespace listdemo { + +// THE BUILD VERSION (docs/SPEC-TABLES.md §20): one digest over every fact the bytes +// this build produces depend on — the type wire's protocol id, every record's +// layout as the compiler's own C ABI model computes it, and the facts that +// decide what a load PUTS in those slots. It is the number a cook's header +// carries and the number Open compares, and the number a block's prologue +// carries and BlockOpen compares: a build version answers "which build?" and +// not "which form?", and what separates the two forms is their MAGIC. +// +// There are TWO ids in the design and they are not interchangeable: the +// PROTOCOL ID is the type wire's and nothing else, and the BUILD VERSION is +// what everything cooked or blocked is keyed by. A table edit moves this and +// never the protocol id; a type edit moves both. +static const uint64_t BuildVersion = 0x8d7c0edaca4571c7ull; + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_BUILD_VERSION + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK +#define LISTDEMO_SCHEMA_TABLE_COOK + +namespace listdemo { + +// ---- the cooked form (docs/SPEC-TABLES.md §7) ---- +// +// A cooked file is a HEADER, a DATA part and an ATTRIBUTION part, in that +// order. Every word of the header is a u64 written in the byte order the cook +// was produced in, and the header is 64 bytes: +// +// 0 magic 0x4b4f4f434d484353, read BYTEWISE before anything else +// 8 build_version the unit's id (docs/SPEC-TABLES.md §20) +// 16 byte_order 1 little, 2 big — the order that WROTE the file +// 24 data_length the region's bytes, rounded up to alignment +// 32 attribution_length the directory's bytes, or 0 +// 40 alignment the region's alignment, never below eight +// 48 reserved zero +// 56 reserved zero +// +// The DATA part is Lock's region written verbatim (§7.2) — the root at its +// base — and it is what a runtime points at. The ATTRIBUTION part is the node +// directory (§6.3), and NOTHING THAT READS THE STRUCTURE TOUCHES IT: it is +// written beside the data for schema cook-check, so a build that ships no +// tooling need not carry it at all. +static const int64_t kTableCookHeaderBytes = 64; + +// THE MAGIC'S VALUE, and a consumer written from the page needs the constant +// rather than a description of one. It is "SCHMCOOK" read as ASCII in the byte +// order a little-endian store produces — the same shape the block form's +// SCHMABLK takes, so a hex dump of a little-endian cook is legible and the two +// accelerators sit in one vocabulary. +// +// IT IS STORED IN THE PRODUCER'S ORDER, which is what makes it the byte-order +// check as well as the form check: a consumer reads back this build's +// constant, or that constant byte-reversed — which identifies a cook of the +// OTHER order — or something that is not a cook. All three answers but the +// first refuse, and a cook and a BLOCK are separated here too, because a +// form's identity belongs in its magic rather than in a second digest. +static const uint64_t TableCookMagic = 0x4b4f4f434d484353ull; + +// THIS BUILD's byte order, as the header's own word carries it. The magic is +// what REFUSES a foreign order; this word is what RECORDS which order wrote +// the file, so a refusal names the order rather than inferring it and a tool +// dumping a cook reads the fact. A file whose magic matched and whose order +// word did not is corrupt, and there is no reading that recovers it. +// +// The BUILD VERSION cannot do either job: §20.1 digests byteorder as a +// GENERATION input, little for every target schema generates for today, so +// two builds of one schema for two orders emit the same id. +#if defined( __BYTE_ORDER__ ) && defined( __ORDER_BIG_ENDIAN__ ) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__ +static const uint64_t TableCookByteOrder = 2; // big +#else +static const uint64_t TableCookByteOrder = 1; // little +#endif + +// The greatest region alignment a cooked file may name. The DATA part begins +// at align_up( 64, alignment ), which is 64 for every unit this language can +// declare — the largest alignment it has is sixteen — so a word past this cap +// describes a file no build of this schema wrote (docs/SPEC-TABLES.md §7.1). +static const uint64_t TableCookMaxAlign = 64; + +// The header read, BYTEWISE. memcpy is the portable spelling of "these eight +// bytes, in this machine's order"; every compiler this repo builds under folds +// it to one load, and it is the only read in the whole of Open that is not a +// comparison. +inline uint64_t table_cook_read64( const uint8_t * p ) +{ + uint64_t v; + memcpy( &v, p, sizeof( v ) ); + return v; +} + +// TableCookOpen: THE WHOLE CHECK, in one place, because §7 states the +// enumeration once and every generated Open is that one enumeration plus +// its own root's two layout facts. +// +// THE CHECK, in order: the magic read bytewise, the byte order it establishes, +// the build version against this build's own, both RESERVED words zero, the +// region alignment the header names, the two part lengths against the length +// the caller passed — a truncated file and a file with trailing bytes are the +// same refusal — the root's own storage inside the data part, and the +// alignment of the base. +// +// AND THAT IS ALL OF IT. On a match the bytes ARE what this build wrote, in +// this build's layout and this build's byte order, so there is nothing to +// validate and nothing to fix up: the caller gets the root. Nothing per node +// happens here, which is what makes open O(1) in the file's size; a walk of +// any shape would forfeit that, and validating an untrusted file is schema +// cook-check's job and a person's decision (§7.4). +// +// EVERY NUMBER BELOW COMES OUT OF THE FILE, so the arithmetic is unsigned and +// each term is BOUNDED BEFORE IT IS ADDED: a forged length near 2^64 must +// refuse, and an addition that wrapped would be the defect the comparison +// after it was supposed to catch. Nothing past length is read on any path, +// including every refusing one. +inline const uint8_t * TableCookOpen( const void * bytes, uint64_t length, uint64_t root_size, uint64_t root_align ) +{ + if ( bytes == NULL ) { return NULL; } + if ( length < (uint64_t) kTableCookHeaderBytes ) { return NULL; } + const uint8_t * raw = (const uint8_t *) bytes; + // the MAGIC, bytewise and first: it is what establishes the byte order + // every other header word is read in, so nothing else may be read before + // it. A byte-reversed constant is a cook of the other order and refuses + // here, which is why the order never reaches a fix-up pass. + if ( table_cook_read64( raw ) != TableCookMagic ) { return NULL; } + if ( table_cook_read64( raw + 16 ) != TableCookByteOrder ) { return NULL; } + if ( table_cook_read64( raw + 8 ) != BuildVersion ) { return NULL; } + // the RESERVED words: a non-zero one means a writer used a form this build + // does not understand, and Open refuses rather than ignoring it. + if ( table_cook_read64( raw + 48 ) != 0 ) { return NULL; } + if ( table_cook_read64( raw + 56 ) != 0 ) { return NULL; } + const uint64_t data_length = table_cook_read64( raw + 24 ); + const uint64_t attribution_length = table_cook_read64( raw + 32 ); + const uint64_t alignment = table_cook_read64( raw + 40 ); + // THE ALIGNMENT WORD IS DATA, and it is the one header field the rest of + // the check does arithmetic WITH rather than only comparison against. A + // region's alignment is a power of two, never below eight (the floor that + // puts the attribution part on an eight-byte boundary without a second + // padding rule) and never past the cap above; a word that is none of those + // rounds nothing and aligns nothing, so it is refused before it is used. + if ( alignment < 8 || alignment > TableCookMaxAlign ) { return NULL; } + if ( ( alignment & ( alignment - 1 ) ) != 0 ) { return NULL; } + // and it must be an alignment THE ROOT CAN SIT AT, since the root is at + // the region's base: both are powers of two, so "at least the root's" + // is one division. + if ( ( alignment % root_align ) != 0 ) { return NULL; } + // The DATA part begins at align_up( 64, alignment ). It is DERIVED and not + // a header field, because a fact a reader computes is a fact two writers + // cannot disagree about. + const uint64_t data_offset = ( (uint64_t) kTableCookHeaderBytes + alignment - 1 ) & ~( alignment - 1 ); + if ( length < data_offset ) { return NULL; } + // the two part lengths against the length the caller passed. The whole + // file is data_offset + data_length + attribution_length, and a length + // that is not EXACTLY that refuses — truncation and trailing bytes are one + // refusal, and both terms are subtracted rather than added so no sum can + // carry. + if ( data_length > length - data_offset ) { return NULL; } + if ( attribution_length != length - data_offset - data_length ) { return NULL; } + // the ROOT sits at the region's base, so the region has to hold it: a + // shorter data part describes a root partly outside the file, which is the + // one way a match-and-point reader could hand back storage it never + // received. + if ( data_length < root_size ) { return NULL; } + const uint8_t * base = raw + data_offset; + // the alignment of the BASE. The header pads the data part to the region's + // alignment, so a base an allocator or mmap gave you is already aligned — + // mmap gives page alignment for free — and a base that is not is a caller's + // buffer this form cannot be read out of. + if ( ( (uintptr_t) base % (uintptr_t) alignment ) != 0 ) { return NULL; } + return base; +} + +// ---- the cooked form, the WRITE side (docs/SPEC-TABLES.md §7.6) ---- +// +// THE BYTE ORDER IS THE TARGET'S, NOT THE HOST'S. A cook is produced in the +// byte order of the build that will read it (§7), so the fixing happens here — +// offline, once, on the writing side — and never at Open. Passing +// TableByteOrder::Big on a little-endian machine produces a big-endian build's +// file, and nothing about the writing host reaches the bytes. +enum class TableByteOrder +{ + Little = 1, // the header's byte_order word, and the order every scalar is written in + Big = 2, +}; + +// One store, width as an argument. Every call site passes a literal width, so +// the loop folds to a store (and a byte swap on the foreign order); a name per +// width would claim four §11 names to save nothing. +inline void table_cook_put( uint8_t * at, uint64_t value, int32_t width, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * i ) ); } + } + else + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * ( width - 1 - i ) ) ); } + } +} + +// A 128-bit store as two lanes: sixteen bytes, the low lane first in the +// little order and the high lane first — each lane big-endian — in the big +// order, exactly as a u64 is one lane of eight (docs/SPEC-TABLES.md §7.2). +inline void table_cook_put128( uint8_t * at, uint64_t lo, uint64_t hi, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) { table_cook_put( at, lo, 8, order ); table_cook_put( at + 8, hi, 8, order ); } + else { table_cook_put( at, hi, 8, order ); table_cook_put( at + 8, lo, 8, order ); } +} + +// A buffer piece: the USED bytes and nothing else. The tail is already zero — +// the whole extent was zeroed before any field was written — so this copies the +// used prefix and leaves the rest, which is what makes a string's unused tail a +// consequence of one memset rather than a rule per buffer. A used length past +// the buffer, or below zero, is a value no reader could have produced and it is +// clamped rather than trusted: this writes inside the caller's buffer on every +// input. +inline void table_cook_bytes( uint8_t * at, const void * source, int64_t used, int64_t capacity ) +{ + if ( used <= 0 ) { return; } + const int64_t n = used < capacity ? used : capacity; + memcpy( at, source, (size_t) n ); +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE +#define LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// ---- the cooked form's WRITE side for a POINTERED root (docs/SPEC-TABLES.md §7.6) ---- +// +// A pointered root's cook is the region of §7.2: every node the numbering +// reached (§3.1), once, at its own type's alignment, in index order, the root +// at offset zero. This is that region while it is being laid out and written — +// the tool's own Layout and Write, in one struct. +// +// The OFFSETS are one per node, the root's zero at position 0 and node index k +// at position k - 1, which is the directory's own order (§6.3); they are the +// one allocation the write makes beyond the numbering, and they go through the +// same pair. A measure needs no offsets and leaves the pointer NULL. +struct TableCookRegion +{ + const TableNumbering * numbering = NULL; // node -> index, from the walk that placed it + int64_t * offsets = NULL; // index - 1 -> the node's region offset; NULL while measuring + int64_t count = 0; // nodes, the root included + int64_t bytes = 0; // the data part's length, rounded to align + int64_t align = 0; // the region's alignment: the nodes' greatest, never below eight + uint8_t * base = NULL; // where the data part is being written; NULL while measuring +}; + +// A reference slot: the SELF-RELATIVE delta from the slot's own address to the +// node's start (§6.3), and zero for null. The node is found by the address the +// numbering keyed it under, which is the same address the walk resolved through +// the same context — so a reference the numbering does not carry is a slot the +// walk never reached (a counted array's slot past its count, an absent +// optional's value) holding a node the region will not hold, and it is refused +// rather than written as a delta to nowhere. +inline bool table_cook_ref( const TableCookRegion & region, uint8_t * at, const void * pointee, TableByteOrder order ) +{ + if ( pointee == NULL ) { table_cook_put( at, 0, 8, order ); return true; } + uint64_t index = 0; + if ( !TableNumberingIndex( *region.numbering, pointee, index ) ) { return false; } + if ( index == 0 || index > (uint64_t) region.count ) { return false; } + const int64_t delta = region.offsets[index - 1] - (int64_t) ( at - region.base ); + table_cook_put( at, (uint64_t) delta, 8, order ); + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// table Unit — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Unit { + int32_t v = 0; +}; + +// table Bounded — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Bounded { + Unit items[8]; // used count beside it; count in [0, 8] + int32_t items_count = 0; + int32_t tag = 0; +}; + +// table Unbounded — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Unbounded { + TableList items; // Unit: the element array, empty until an Add + int32_t tag = 0; +}; + +// ---- prefill: the declared defaults, in place (docs/SPEC-TABLES.md) ---- + +inline void UnitReset( Unit & value ); +inline void BoundedReset( Bounded & value ); +inline void UnboundedReset( Unbounded & value ); + +inline void UnitReset( Unit & value ) +{ + value.v = 0; +} + +inline void BoundedReset( Bounded & value ) +{ + UnitReset( value.items[0] ); + for ( int32_t i = 1; i < 8; i++ ) { value.items[i] = value.items[0]; } + value.items_count = 0; + value.tag = 0; +} + +inline void UnboundedReset( Unbounded & value ) +{ + value.items.elements.value = 0; // Unit: empty + value.items.count = 0; + value.items.padding = 0; + value.tag = 0; +} + +// ---- the arena's reset hook (docs/SPEC-TABLES.md §6) ---- +// +// TableWorker::Alloc is a template and cannot name a member's Reset, so +// the arena reaches it through this overload set by argument-dependent +// lookup. It is how a node born in raw arena storage comes to hold the +// declared defaults without value-initialising the whole aggregate. + +inline void TableReset( Unit & value ) { UnitReset( value ); } +inline void TableReset( Bounded & value ) { BoundedReset( value ); } +inline void TableReset( Unbounded & value ) { UnboundedReset( value ); } + +// ---- pointer targets: allocation and resolution (docs/SPEC-TABLES.md §2) ---- +// +// A reference resolves differently in the two forms, and the CONTEXT says +// which: in the arena it is an offset; in a region it is a self-relative +// delta, so the const deref below is one add and needs no base pointer. + +// ---- codecs: measure/save/load per closure member ---- + +inline int64_t UnitMeasureBody( TableIds & ids, const Unit & value ); +LISTDEMO_TABLE_INLINE bool UnitSaveBody( TableWriter & w, TableIds & ids, const Unit & value ); +LISTDEMO_TABLE_INLINE bool UnitLoadBody( TableReader & r, Unit & value ); +inline int64_t BoundedMeasureBody( TableIds & ids, const Bounded & value ); +LISTDEMO_TABLE_INLINE bool BoundedSaveBody( TableWriter & w, TableIds & ids, const Bounded & value ); +LISTDEMO_TABLE_INLINE bool BoundedLoadBody( TableReader & r, Bounded & value ); +template inline int64_t UnboundedMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Unbounded & value ); +template inline bool UnboundedSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Unbounded & value ); +template inline bool UnboundedSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Unbounded & value ); +inline bool UnboundedLoadBody( TableReader & r, const TableNodeMap & nodes, Unbounded & value ); + +// ---- pointer-graph walkers: number (measure/save), pack (Lock) ---- + +template inline bool UnboundedNumber( const Ctx & ctx, TableNumbering & numbering, const Unbounded & value ); +template inline int64_t UnboundedPackMeasure( const Ctx & ctx, TablePackMap & seen, const Unbounded & value ); +template inline bool UnboundedPack( const Ctx & ctx, TablePackMap & seen, const Unbounded & src, Unbounded & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +// ---- the numbering's bridge to each member's codec (docs/SPEC-TABLES.md §3.1) ---- + +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Unbounded & value ) { return UnboundedMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Unbounded & value ) { return UnboundedSaveBody( ctx, numbering, w, ids, value ); } + +inline int64_t UnitMeasureBody( TableIds & ids, const Unit & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.v != 0 ) { bytes += TableLebBytes( ids.ref( 0xaf63eb4c86020609ull, 23 ) ) + 1 + 4; } // v + return bytes; +} + +inline int64_t UnitMeasure( const Unit & value ) +{ + TableIds ids; + const int64_t body = UnitMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool UnitSaveBody( TableWriter & w, TableIds & ids, const Unit & value ) +{ + if ( value.v != 0 ) + { + w.putleb( ids.ref( 0xaf63eb4c86020609ull, 23 ) ); w.put8( 4 ); // v + w.put32( uint32_t( value.v ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t UnitSave( const Unit & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !UnitSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == UnitMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool UnitLoadBody( TableReader & r, Unit & value ) +{ + UnitReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xaf63eb4c86020609ull: // v + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.v = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict UnitLoadVerdict( Unit & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + UnitReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + UnitReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !UnitLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool UnitLoad( Unit & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return UnitLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t UnitMeasureMessage( const Unit & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = UnitMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t UnitSaveMessage( const Unit & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !UnitSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == UnitMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool UnitLoadMessage( Unit & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + UnitReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return UnitLoadBody( r, value ); +} + +inline int64_t BoundedMeasureBody( TableIds & ids, const Bounded & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.items_count < 0 || value.items_count > 8 ) { return -1; } // storage invariant + if ( value.items_count > 0 ) + { + const uint64_t ref_items = ids.ref( 0x3e7884bf4f412c6full, 6 ); + int64_t body_items = 0; + body_items += 1 + TableLebBytes( (uint64_t) ( value.items_count ) ); // the element kind byte and the count + for ( int32_t elem_i = 0; elem_i < value.items_count; elem_i++ ) + { + const int64_t elem_bytes = UnitMeasureBody( ids, value.items[elem_i] ); + if ( elem_bytes < 0 ) { return -1; } + body_items += TableLebBytes( (uint64_t) ( elem_bytes ) ) + ( elem_bytes ); + } + bytes += TableLebBytes( ref_items ) + 1 + TableLebBytes( (uint64_t) ( body_items ) ) + ( body_items ); // items + } + if ( value.tag != 0 ) { bytes += TableLebBytes( ids.ref( 0x56d7ab194448a4f3ull, 7 ) ) + 1 + 4; } // tag + return bytes; +} + +inline int64_t BoundedMeasure( const Bounded & value ) +{ + TableIds ids; + const int64_t body = BoundedMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool BoundedSaveBody( TableWriter & w, TableIds & ids, const Bounded & value ) +{ + if ( value.items_count < 0 || value.items_count > 8 ) { return false; } // storage invariant + if ( value.items_count > 0 ) + { + const uint64_t ref_items = ids.ref( 0x3e7884bf4f412c6full, 6 ); + int64_t body_items = 0; + body_items += 1 + TableLebBytes( (uint64_t) ( value.items_count ) ); // the element kind byte and the count + for ( int32_t elem_i = 0; elem_i < value.items_count; elem_i++ ) + { + const int64_t elem_bytes = UnitMeasureBody( ids, value.items[elem_i] ); + if ( elem_bytes < 0 ) { return false; } + body_items += TableLebBytes( (uint64_t) ( elem_bytes ) ) + ( elem_bytes ); + } + w.putleb( ref_items ); w.put8( 14 ); w.putleb( (uint64_t) body_items ); // items + w.put8( 13 ); w.putleb( (uint64_t) ( value.items_count ) ); + for ( int32_t elem_i = 0; elem_i < value.items_count; elem_i++ ) + { + { + const int64_t elem_len = UnitMeasureBody( ids, value.items[elem_i] ); + if ( elem_len < 0 ) return false; + w.putleb( (uint64_t) elem_len ); + if ( !UnitSaveBody( w, ids, value.items[elem_i] ) ) return false; + } + } + } + if ( value.tag != 0 ) + { + w.putleb( ids.ref( 0x56d7ab194448a4f3ull, 7 ) ); w.put8( 4 ); // tag + w.put32( uint32_t( value.tag ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t BoundedSave( const Bounded & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !BoundedSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == BoundedMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool BoundedLoadBody( TableReader & r, Bounded & value ) +{ + BoundedReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x3e7884bf4f412c6full: // items + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER — the element kind byte and the + // count, so fewer than two bytes — is INERT (§4): the field keeps the + // value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + // A DAMAGED COUNT stops the elements and nothing else: the field + // RODE, so an optional is still PRESENT (§2.3) — only a foreign + // ELEMENT KIND says the payload is not this array's at all. + if ( !counted_ok ) { r.report->malformed = true; } + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + uint64_t keep = count; + if ( keep > 8 ) { keep = 8; r.report->clamped++; } + // elements are BOUNDED by the field body: a count the length + // cannot cover keeps the decoded prefix, flags malformed, and + // the parent continues at the next field — following fields' + // bytes are never fabricated into elements + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + uint64_t decoded = 0; + for ( uint64_t i = 0; i < keep; i++ ) + { + uint64_t elem_len = 0; + if ( !sub.getleb( elem_len ) || !sub.room( elem_len ) ) { r.report->malformed = true; break; } + { + TableReader elem( sub.buffer + sub.offset, (int64_t) elem_len, r.report, r.ids ); + UnitLoadBody( elem, value.items[(int32_t) i] ); + } + sub.offset += (int64_t) elem_len; + decoded = i + 1; + } + value.items_count = (int32_t) decoded; + } + } + r.offset = body_end; // excess elements and slack skip via the length + break; + } + case 0x56d7ab194448a4f3ull: // tag + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.tag = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict BoundedLoadVerdict( Bounded & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + BoundedReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + BoundedReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !BoundedLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool BoundedLoad( Bounded & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return BoundedLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t BoundedMeasureMessage( const Bounded & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = BoundedMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t BoundedSaveMessage( const Bounded & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !BoundedSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == BoundedMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool BoundedLoadMessage( Bounded & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + BoundedReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return BoundedLoadBody( r, value ); +} + +template +inline int64_t UnboundedMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Unbounded & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // items: a kind 14 array of kind 13 elements, INDEX order (§2.9) + TableListCursor cursor_items = TableListElements( ctx, value.items ); + if ( !cursor_items.ok ) { return -1; } // the slot and the head disagree + if ( cursor_items.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_items = ids.ref( 0x3e7884bf4f412c6full, 6 ); + int64_t body_items = 0; + body_items += 1 + TableLebBytes( (uint64_t) ( cursor_items.count ) ); // the element kind byte and the count + for ( int32_t elem_i_items = 0; elem_i_items < cursor_items.count; elem_i_items++ ) + { + const int64_t elem_bytes_items = UnitMeasureBody( ids, cursor_items[elem_i_items] ); + if ( elem_bytes_items < 0 ) { return -1; } + body_items += TableLebBytes( (uint64_t) ( elem_bytes_items ) ) + ( elem_bytes_items ); + } + bytes += TableLebBytes( ref_items ) + 1 + TableLebBytes( (uint64_t) ( body_items ) ) + ( body_items ); + } + } + if ( value.tag != 0 ) { bytes += TableLebBytes( ids.ref( 0x56d7ab194448a4f3ull, 7 ) ) + 1 + 4; } // tag + return bytes; +} + +template +inline bool UnboundedSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Unbounded & value ) +{ + (void) ctx; (void) numbering; + { + TableListCursor cursor_items = TableListElements( ctx, value.items ); // items + if ( !cursor_items.ok ) { return false; } + if ( cursor_items.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_items = ids.ref( 0x3e7884bf4f412c6full, 6 ); + int64_t body_items = 0; + body_items += 1 + TableLebBytes( (uint64_t) ( cursor_items.count ) ); // the element kind byte and the count + for ( int32_t elem_i_items = 0; elem_i_items < cursor_items.count; elem_i_items++ ) + { + const int64_t elem_bytes_items = UnitMeasureBody( ids, cursor_items[elem_i_items] ); + if ( elem_bytes_items < 0 ) { return false; } + body_items += TableLebBytes( (uint64_t) ( elem_bytes_items ) ) + ( elem_bytes_items ); + } + w.putleb( ref_items ); w.put8( 14 ); w.putleb( (uint64_t) body_items ); // items + w.put8( 13 ); w.putleb( (uint64_t) ( cursor_items.count ) ); + for ( int32_t elem_i_items = 0; elem_i_items < cursor_items.count; elem_i_items++ ) + { + { + const int64_t elem_len_items = UnitMeasureBody( ids, cursor_items[elem_i_items] ); + if ( elem_len_items < 0 ) return false; + w.putleb( (uint64_t) elem_len_items ); + if ( !UnitSaveBody( w, ids, cursor_items[elem_i_items] ) ) return false; + } + } + } + } + if ( value.tag != 0 ) + { + w.putleb( ids.ref( 0x56d7ab194448a4f3ull, 7 ) ); w.put8( 4 ); // tag + w.put32( uint32_t( value.tag ) ); + } + return !w.overflow; +} + +template +inline bool UnboundedSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Unbounded & value ) +{ + if ( !UnboundedSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool UnboundedLoadBody( TableReader & r, const TableNodeMap & nodes, Unbounded & value ) +{ + (void) nodes; + UnboundedReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x3e7884bf4f412c6full: // items + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.items, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Unit * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + uint64_t elem_len_items = 0; + if ( !sub.getleb( elem_len_items ) || !sub.room( elem_len_items ) ) { r.report->malformed = true; break; } + { + TableReader elem_items( sub.buffer + sub.offset, (int64_t) elem_len_items, r.report, r.ids ); + UnitLoadBody( elem_items, ( *slot ) ); + } + sub.offset += (int64_t) elem_len_items; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x56d7ab194448a4f3ull: // tag + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.tag = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// UnboundedWireExtent: the extent Unbounded's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool UnboundedWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x3e7884bf4f412c6full && field_kind == 14 ) // items: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Unit ), (int64_t) alignof( Unit ), 13, 2, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// UnboundedExtentAt: the node extent Unbounded's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as UnboundedExtentPack advances it (§2.8, §2.9). +template +inline bool UnboundedExtentAt( const Ctx & ctx, const Unbounded & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.items ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Unit ) - 1 ) & ~( (int64_t) alignof( Unit ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Unit ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t UnboundedExtent( const Ctx & ctx, const Unbounded & value ) +{ + int64_t at = 0; + if ( !UnboundedExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// UnboundedExtentPack: carve Unbounded's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset UnboundedExtentAt advances (§2.8, §2.9). +template +inline bool UnboundedExtentPack( const Ctx & ctx, const Unbounded & src, Unbounded & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.items ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Unit ) - 1 ) & ~( (int64_t) alignof( Unit ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Unit ); + if ( at + bytes > capacity ) { return false; } + Unit * placed = (Unit *) ( extent + at ); + at += bytes; + dst.items.count = cursor.count; + dst.items.padding = 0; + dst.items.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.items.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Unit ) ); // trivially copyable, by construction + } + } + return true; +} + +// ---- Unbounded.items: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Unit * UnboundedItemsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool UnboundedItemsErase( TableArena & arena, TableList & list, const Unit * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach UnboundedItemsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// UnboundedNumber: number everything Unbounded POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool UnboundedNumber( const Ctx & ctx, TableNumbering & numbering, const Unbounded & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// UnboundedPackMeasure: the packed region bytes of everything Unbounded POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t UnboundedPackMeasure( const Ctx & ctx, TablePackMap & seen, const Unbounded & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// UnboundedPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool UnboundedPackEdges( const Ctx & ctx, TablePackMap & seen, const Unbounded & src, Unbounded & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool UnboundedPack( const Ctx & ctx, TablePackMap & seen, const Unbounded & src, Unbounded & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Unbounded ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Unbounded ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !UnboundedExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return UnboundedPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool UnboundedPackEdges( const Ctx & ctx, TablePackMap & seen, const Unbounded & src, Unbounded & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// ---- Unbounded: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: UnboundedBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Unbounded is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct UnboundedBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + UnboundedBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~UnboundedBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + UnboundedBuilder( const UnboundedBuilder & ) = delete; + UnboundedBuilder & operator=( const UnboundedBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Unbounded * GetRoot() { return arena.locked ? NULL : (Unbounded *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Unbounded * AsConst() const { return (const Unbounded *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool UnboundedBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Unbounded & root = *(const Unbounded *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = UnboundedPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = UnboundedExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + Unbounded * destination = new ( packed ) Unbounded; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !UnboundedPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Unbounded on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// UnboundedNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t UnboundedNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// UnboundedNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void UnboundedNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// UnboundedNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t UnboundedNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// UnboundedNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t UnboundedNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// UnboundedNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void UnboundedNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = UnboundedNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? UnboundedNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool UnboundedNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Unbounded & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return UnboundedNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t UnboundedMeasureWire( const Ctx & ctx, const Unbounded & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( UnboundedNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = UnboundedMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t UnboundedSaveWire( const Ctx & ctx, const Unbounded & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !UnboundedNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = UnboundedSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == UnboundedMeasure( root ) +} + +inline int64_t UnboundedMeasure( const Unbounded * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return UnboundedMeasureWire( ctx, *root, allocator ); +} + +inline int64_t UnboundedSave( const Unbounded * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return UnboundedSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t UnboundedMeasure( const UnboundedBuilder & builder ) +{ + if ( builder.region != NULL ) { return UnboundedMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return UnboundedMeasureWire( ctx, *(const Unbounded *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t UnboundedSave( const UnboundedBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return UnboundedSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return UnboundedSaveWire( ctx, *(const Unbounded *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t UnboundedMeasureMessage( const Unbounded * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return UnboundedMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t UnboundedSaveMessage( const Unbounded * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return UnboundedSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t UnboundedMeasureMessage( const UnboundedBuilder & builder ) +{ + if ( builder.region != NULL ) { return UnboundedMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return UnboundedMeasureWire( ctx, *(const Unbounded *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t UnboundedSaveMessage( const UnboundedBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return UnboundedSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return UnboundedSaveWire( ctx, *(const Unbounded *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// UnboundedLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t UnboundedLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !UnboundedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = UnboundedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// UnboundedLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Unbounded * UnboundedLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Unbounded ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !UnboundedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = UnboundedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x2ec2f34386026d8bull; + Unbounded * root = new ( region ) Unbounded; // lifetime only: LoadBody's first act is UnboundedReset + UnboundedReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = UnboundedNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + UnboundedNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + UnboundedNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Unbounded ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + UnboundedLoadBody( r, nodes, *root ); + return root; +} + +// UnboundedLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t UnboundedLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !UnboundedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = UnboundedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// UnboundedLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Unbounded * UnboundedLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Unbounded ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !UnboundedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = UnboundedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x2ec2f34386026d8bull; + Unbounded * root = new ( region ) Unbounded; // lifetime only: LoadBody's first act is UnboundedReset + UnboundedReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = UnboundedNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + UnboundedNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + UnboundedNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Unbounded ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + UnboundedLoadBody( r, nodes, *root ); + return root; +} + +// UnboundedLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool UnboundedLoadBuilder( UnboundedBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Unbounded * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0x2ec2f34386026d8bull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = UnboundedNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + UnboundedNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = UnboundedLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- the cooked form: point at a cook (docs/SPEC-TABLES.md §7) ---- + +// UnitOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Unit IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Unit * UnitOpen( const void * bytes, uint64_t length ) +{ + return (const Unit *) TableCookOpen( bytes, length, (uint64_t) sizeof( Unit ), (uint64_t) alignof( Unit ) ); +} + +// BoundedOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Bounded IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Bounded * BoundedOpen( const void * bytes, uint64_t length ) +{ + return (const Bounded *) TableCookOpen( bytes, length, (uint64_t) sizeof( Bounded ), (uint64_t) alignof( Bounded ) ); +} + +// UnboundedOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH UnboundedAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Unbounded * UnboundedOpen( const void * bytes, uint64_t length ) +{ + return (const Unbounded *) TableCookOpen( bytes, length, (uint64_t) sizeof( Unbounded ), (uint64_t) alignof( Unbounded ) ); +} + +// ---- the cooked form: WRITE a cook (docs/SPEC-TABLES.md §7.6) ---- +// +// The bytes are `schema cook`'s, and the tool stays the reference: the two +// writers are held to one file, byte for byte, in both byte orders. A cook is +// content-addressed by (asset hash, build version), so two writers of one +// instance produce ONE artifact or the pair means nothing. + +inline void UnitCookBody( uint8_t * at, const Unit & value, TableByteOrder order ); +inline void BoundedCookBody( uint8_t * at, const Bounded & value, TableByteOrder order ); +template inline bool UnboundedCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Unbounded & value, TableByteOrder order ); + +inline void UnitCookBody( uint8_t * at, const Unit & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.v, 4, order ); +} + +inline void BoundedCookBody( uint8_t * at, const Bounded & value, TableByteOrder order ) +{ + // all 8 slots: the storage is allocate-max, and a slot past the count rides as it lies (§7.2) + for ( int32_t i = 0; i < 8; i++ ) + { + UnitCookBody( at + 0 + i * 4, value.items[ i ], order ); + } + table_cook_put( at + 32, (uint64_t) (uint32_t) value.items_count, 4, order ); + table_cook_put( at + 36, (uint64_t) value.tag, 4, order ); +} + +template inline bool UnboundedCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Unbounded & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // items: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.tag, 4, order ); + return true; +} + +template inline bool UnitCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Unit & value, TableByteOrder order ); +template inline bool BoundedCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Bounded & value, TableByteOrder order ); +template inline bool UnboundedCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Unbounded & value, TableByteOrder order ); + +// UnitCookExtent: Unit's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool UnitCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Unit & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// BoundedCookExtent: Bounded's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool BoundedCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Bounded & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// UnboundedCookExtent: Unbounded's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool UnboundedCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Unbounded & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // items: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.items ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( Unit ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 4; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + UnitCookBody( array + i * 4, cursor[i], order ); + } + } + return true; +} + +// UnitCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool UnitCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Unit & value, TableByteOrder order ) +{ + UnitCookBody( at, value, order ); + int64_t extent_at = 0; + return UnitCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// BoundedCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool BoundedCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Bounded & value, TableByteOrder order ) +{ + BoundedCookBody( at, value, order ); + int64_t extent_at = 0; + return BoundedCookExtent( ctx, region, at + 40, extent_at, at, value, order ); +} + +// UnboundedCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool UnboundedCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Unbounded & value, TableByteOrder order ) +{ + if ( !UnboundedCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return UnboundedCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// UnitCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Unit IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t UnitCookMeasure( const Unit & value ) +{ + (void) value; + return 88; // 64 header + 8 data + 16 attribution +} + +// UnitCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract UnitMeasure/UnitSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool UnitCook( const Unit & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) UnitCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 8, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + UnitCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 72, 0, 8, order ); + table_cook_put( raw + 80, 0x8f26e0f086cc2787ull, 8, order ); + return true; +} + +// BoundedCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Bounded IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t BoundedCookMeasure( const Bounded & value ) +{ + (void) value; + return 120; // 64 header + 40 data + 16 attribution +} + +// BoundedCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract BoundedMeasure/BoundedSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool BoundedCook( const Bounded & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) BoundedCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 40, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + BoundedCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 104, 0, 8, order ); + table_cook_put( raw + 112, 0x0acac10912f5892aull, 8, order ); + return true; +} + +// UnboundedCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool UnboundedCookLayout( const Ctx & ctx, const Unbounded & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = UnboundedExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// UnboundedCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t UnboundedCookMeasureFrom( const Ctx & ctx, const Unbounded & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( UnboundedNumberFrom( ctx, numbering, root ) && UnboundedCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// UnboundedCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool UnboundedCookFrom( const Ctx & ctx, const Unbounded & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = UnboundedNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && UnboundedCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = UnboundedCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0x2ec2f34386026d8bull, 8, order ); // the root: fnv1a64( "Unbounded" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// UnboundedCookMeasure / UnboundedCook over a REGION root — a locked builder's AsConst, a +// region UnboundedLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t UnboundedCookMeasure( const Unbounded * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return UnboundedCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool UnboundedCook( const Unbounded * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return UnboundedCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t UnboundedCookMeasure( const UnboundedBuilder & builder ) +{ + if ( builder.region != NULL ) { return UnboundedCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return UnboundedCookMeasureFrom( ctx, *(const Unbounded *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool UnboundedCook( const UnboundedBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return UnboundedCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return UnboundedCookFrom( ctx, *(const Unbounded *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ---- relocatability, enforced: the wire is a pure length-prefixed +// stream AND the decoded storage is pointer-free — every closure type +// must stay trivially copyable and standard-layout, so instances can be +// memcpy'd, mmap'd, shared across processes, and walked through +// descriptor offsets. A failure here means a pointer, virtual or +// non-trivial member crept into generated storage. +// +// They ask the COMPILER ITSELF, which is what every C++ standard library +// answers the same two questions with — and it costs this header no +// include at all. +// A pointer FIELD is a TableRef — eight bytes and no address — so the +// property holds in BOTH forms: a fixed-size table is one relocatable +// struct, and a packed region is one relocatable block whose references +// are self-relative and therefore survive a plain memcpy. +static_assert( __is_trivially_copyable( Unit ), "Unit must stay relocatable" ); +static_assert( __is_standard_layout( Unit ), "Unit must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Bounded ), "Bounded must stay relocatable" ); +static_assert( __is_standard_layout( Bounded ), "Bounded must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Unbounded ), "Unbounded must stay relocatable" ); +static_assert( __is_standard_layout( Unbounded ), "Unbounded must stay standard-layout for offsetof" ); + +// ---- the cook's layout contract (docs/SPEC-TABLES.md §20.3) ---- +// +// The compiler derived every number below from the declaration and folded it +// into the BUILD VERSION; these asserts are this compiler saying whether it +// agrees. The model is not self-evidently right — on 32-bit System V +// alignof(uint64_t) is 4, not 8 — which is precisely why it is asserted +// rather than assumed. +static_assert( sizeof( Unit ) == 4, "Unit's sizeof moved: the build version was taken over 4, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Unit ) == 4, "Unit's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Unit, v ) == 0, "Unit's field v moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Bounded ) == 40, "Bounded's sizeof moved: the build version was taken over 40, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Bounded ) == 4, "Bounded's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Bounded, items ) == 0, "Bounded's field items moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Bounded, tag ) == 36, "Bounded's field tag moved: the build version was taken over offset 36 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Unbounded ) == 24, "Unbounded's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Unbounded ) == 8, "Unbounded's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Unbounded, items ) == 0, "Unbounded's field items moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Unbounded, tag ) == 16, "Unbounded's field tag moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); + +static_assert( alignof( Unit ) <= kTableAlign, "Unbounded.items: an unbounded array's element alignment must fit the arena's" ); + +// ---- reflection descriptors (tables only, docs/SPEC-TABLES.md) ---- + +inline const TableTypeInfo * UnitTableType(); +inline const TableTypeInfo * BoundedTableType(); +inline const TableTypeInfo * UnboundedTableType(); +// The descriptors are CONSTANT-INITIALISED data, and a field's target is +// the ADDRESS of another descriptor. These declarations are what let a +// self- or mutually-referential graph — Node naming itself through *Node — +// be expressed as constant data instead of a lazy link, which could not +// have been written race-free OR recursion-safe. The whole reflection +// surface is therefore immutable: read it from any thread, any time. +extern const TableTypeInfo UnitTableInfo; +extern const TableTypeInfo BoundedTableInfo; +extern const TableTypeInfo UnboundedTableInfo; + +inline const TableFieldInfo UnitTableFields[] = { + { "v", "v", "int32", 0xaf63eb4c86020609ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Unit, v ), (uint32_t) sizeof( Unit::v ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo UnitTableInfo = { "Unit", (uint32_t) sizeof( Unit ), 1, UnitTableFields, +[]( void * p ) { UnitReset( *(Unit *) p ); }, false }; +inline const TableTypeInfo * UnitTableType() { return &UnitTableInfo; } + +inline const TableFieldInfo BoundedTableFields[] = { + { "items", "items", "Unit", 0x3e7884bf4f412c6full, 13, true, false, NULL, NULL, true, false, 8, (uint32_t) offsetof( Bounded, items ), (uint32_t) sizeof( Bounded::items[0] ), (uint32_t) offsetof( Bounded, items_count ), 0xffffffffu, &UnitTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "tag", "tag", "int32", 0x56d7ab194448a4f3ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Bounded, tag ), (uint32_t) sizeof( Bounded::tag ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo BoundedTableInfo = { "Bounded", (uint32_t) sizeof( Bounded ), 2, BoundedTableFields, +[]( void * p ) { BoundedReset( *(Bounded *) p ); }, false }; +inline const TableTypeInfo * BoundedTableType() { return &BoundedTableInfo; } + +inline const TableFieldInfo UnboundedTableFields[] = { + { "items", "items", "Unit", 0x3e7884bf4f412c6full, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Unbounded, items ), (uint32_t) sizeof( Unit ), (uint32_t) offsetof( Unbounded, items.count ), 0xffffffffu, &UnitTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "tag", "tag", "int32", 0x56d7ab194448a4f3ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Unbounded, tag ), (uint32_t) sizeof( Unbounded::tag ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo UnboundedTableInfo = { "Unbounded", (uint32_t) sizeof( Unbounded ), 2, UnboundedTableFields, +[]( void * p ) { UnboundedReset( *(Unbounded *) p ); }, true }; +inline const TableTypeInfo * UnboundedTableType() { return &UnboundedTableInfo; } + +// ---- the text form (docs/SPEC-TABLES.md §16) ---- + +// Unit in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// MigrateTable.cpp; link it to use them. +bool UnitFromJson( Unit & value, const char * text, int64_t bytes, TableReport * report ); +int64_t UnitToJsonMeasure( const Unit & value ); +int64_t UnitToJson( const Unit & value, char * buffer, int64_t capacity ); + +// Bounded in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// MigrateTable.cpp; link it to use them. +bool BoundedFromJson( Bounded & value, const char * text, int64_t bytes, TableReport * report ); +int64_t BoundedToJsonMeasure( const Bounded & value ); +int64_t BoundedToJson( const Bounded & value, char * buffer, int64_t capacity ); + +// Unbounded in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in MigrateTable.cpp; link it to use them. +bool UnboundedFromJson( UnboundedBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t UnboundedToJsonMeasure( const Unbounded * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t UnboundedToJson( const Unbounded * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/ReportTable.cpp b/testdata/golden/tables/lists/ReportTable.cpp new file mode 100644 index 000000000..2e5d5ad3f --- /dev/null +++ b/testdata/golden/tables/lists/ReportTable.cpp @@ -0,0 +1,3153 @@ +// Code generated by the schema compiler from Report.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — the TABLE wire's text form (docs/SPEC-TABLES.md §16). +// Compile this file to use FromJson / ToJson; a project that +// never reads or writes a text does not compile it and pays nothing. + +#include "ReportTable.h" + +#include // the text form: number formatting +#include // the text form: exact number conversion +#include // the text form: the runtime's decimal point + +// The guard is not vestigial. Several listdemo Table.cpp files may be +// concatenated into ONE translation unit — a unity build — and without it +// each would redefine the walk. It is also why the walk's functions may be +// weak (vague linkage) across separate objects: ODR requires their +// definitions to be token-identical, and the generic-walk gate is what +// proves that, byte for byte, across every generated .cpp. +#ifndef LISTDEMO_SCHEMA_TABLE_JSON +#define LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +// ---- the pointer adapters (docs/SPEC-TABLES.md §16.7) ---- +// +// The walk below is ONE walk, byte-identical in every generated .cpp, and a +// pointer is the one kind it cannot walk alone: reading one needs the +// builder's arena and writing one needs a region's deref, and neither exists +// in a unit that declares no pointer. So the walk calls these three and does +// not define them. A unit with no pointer defines them as stubs no field ever +// reaches; a pointered unit defines them in the graph half that follows the +// walk. + +struct TableJsonIn; +struct TableJsonOut; + +// a pointer field's object, or the `&node` reference standing in for it, into +// the slot; the cursor is on the opening brace +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// the node a pointer slot names, in place — or as `&node` when it is shared +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// the FIRST key of an object the walk is skipping begins with `&`: the cursor is +// on its value. A dropped definition still takes its label (§16.7); a fixed reader +// skips the value whole, as it skips everything else it does not place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); + +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- +// +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map +inline bool TableJsonIsMap( const TableFieldInfo * f ); +// the map as a plain JSON object keyed by the KEY, in ASCENDING key order +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that object back into the slot, in whatever order the text gives it +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); + +// ---- json walk: begin ---- +// +// The TEXT form (docs/SPEC-TABLES.md §16): one table, one text, one walk over the +// reflection descriptors (§8). Reading fills ONE caller-owned instance and +// allocates nothing beyond it; writing targets a caller buffer with the +// wire's measure/write symmetry. Everything AROUND this — which file goes +// with which instance, what key an instance is filed under, how instances +// link into a root table's collections — is a packer's opinion and stays +// with the tool that holds it. +// +// The dialect: trailing commas are accepted on read (the authoring files +// this exists for carry them) and never written; comments are not JSON and +// are refused; unknown keys are skipped and counted; a duplicate key is +// last-wins and counted; a key present with the wrong JSON type is skipped +// and counted, never coerced. + +static const int32_t kTableJsonMaxDepth = 128; + +// A key longer than this cannot name a field, so it is skipped as unknown. +static const int32_t kTableJsonMaxKey = 256; + +// The longest numeric token the walk will convert. Anything longer is a +// value no field can hold and counts as a kind mismatch. +static const int32_t kTableJsonMaxNumber = 512; + +// The decimal point the C runtime is CURRENTLY using. Number conversion is +// the one locale-sensitive corner of the grammar — JSON's point is always +// '.', the runtime's is whatever the program set — so every number crosses +// this one character on the way out and on the way back in. Nothing else in +// the walk consults the locale. +inline char TableJsonDecimalPoint() +{ + const struct lconv * conv = localeconv(); + if ( conv != NULL && conv->decimal_point != NULL && conv->decimal_point[0] != 0 ) + { + return conv->decimal_point[0]; + } + return '.'; +} + +// ---- storage access: the descriptors give an offset and a width, and the +// ---- storage is the HOST's, so every load and store goes through a width +// ---- switch rather than a memcpy into the low bytes of a wider word + +// finite: not a NaN, not an infinity. Written without — the walk's +// runtime surface stays the handful of functions it already names. +// A vocabulary entry the descriptor could not spell. The generated name +// functions answer "???" for a value outside the declared set, and that is +// not a name — writing it would put a spelling in the text that the reader +// then counts as unknown, turning a refusal into a silent loss. +inline bool TableJsonNamed( const char * name ) +{ + return name != NULL && strcmp( name, "???" ) != 0; +} + +inline bool TableJsonFinite( double v ) +{ + return v == v && v <= 1.7976931348623157e308 && v >= -1.7976931348623157e308; +} + +inline uint64_t TableJsonGetRaw( const void * storage, uint32_t width ) +{ + switch ( width ) + { + case 1: { uint8_t v = 0; memcpy( &v, storage, 1 ); return v; } + case 2: { uint16_t v = 0; memcpy( &v, storage, 2 ); return v; } + case 4: { uint32_t v = 0; memcpy( &v, storage, 4 ); return v; } + case 8: { uint64_t v = 0; memcpy( &v, storage, 8 ); return v; } + } + return 0; +} + +inline void TableJsonSetRaw( void * storage, uint32_t width, uint64_t value ) +{ + switch ( width ) + { + case 1: { uint8_t v = (uint8_t) value; memcpy( storage, &v, 1 ); break; } + case 2: { uint16_t v = (uint16_t) value; memcpy( storage, &v, 2 ); break; } + case 4: { uint32_t v = (uint32_t) value; memcpy( storage, &v, 4 ); break; } + case 8: { uint64_t v = value; memcpy( storage, &v, 8 ); break; } + } +} + +inline int64_t TableJsonGetSigned( const void * storage, uint32_t width ) +{ + uint64_t raw = TableJsonGetRaw( storage, width ); + if ( width < 8 ) + { + uint64_t sign = uint64_t( 1 ) << ( width * 8 - 1 ); + if ( ( raw & sign ) != 0 ) + { + raw |= ~( ( sign << 1 ) - 1 ); + } + } + return (int64_t) raw; +} + +// ---- the WIDE kinds (docs/SPEC-TABLES.md §3, §16.2) ---- +// +// The 128-bit integers and the fixed-point family convert EXACTLY, over two +// 64-bit lanes: a 128-bit integer is a decimal integer, a fixed value a +// decimal in WHOLE UNITS (1.0, -0.25, 3.0000152587890625) and nothing +// on either path passes through a double. Nothing here needs a 128-bit type +// either, which is what keeps this walk one text for every unit. +struct TableJsonWide +{ + uint64_t lo; + uint64_t hi; +}; + +inline bool TableJsonKindWide( uint8_t kind ) { return kind >= 18 && kind <= 29; } +inline bool TableJsonKindWideSigned( uint8_t kind ) { return kind == 18 || ( kind >= 20 && kind <= 24 ); } +inline bool TableJsonKindFixed( uint8_t kind ) { return kind >= 20 && kind <= 29; } + +inline bool TableJsonWideZero( TableJsonWide v ) { return v.lo == 0 && v.hi == 0; } +inline bool TableJsonWideNegative( TableJsonWide v ) { return ( v.hi >> 63 ) != 0; } + +inline int TableJsonWideCompare( TableJsonWide a, TableJsonWide b, bool is_signed ) +{ + if ( is_signed && TableJsonWideNegative( a ) != TableJsonWideNegative( b ) ) { return TableJsonWideNegative( a ) ? -1 : 1; } + if ( a.hi != b.hi ) { return a.hi < b.hi ? -1 : 1; } + if ( a.lo != b.lo ) { return a.lo < b.lo ? -1 : 1; } + return 0; +} + +inline TableJsonWide TableJsonWideShl( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.hi = v.lo << ( n - 64 ); return r; } + r.hi = ( v.hi << n ) | ( v.lo >> ( 64 - n ) ); + r.lo = v.lo << n; + return r; +} + +inline TableJsonWide TableJsonWideShr( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.lo = v.hi >> ( n - 64 ); return r; } + r.lo = ( v.lo >> n ) | ( v.hi << ( 64 - n ) ); + r.hi = v.hi >> n; + return r; +} + +inline TableJsonWide TableJsonWideNeg( TableJsonWide v ) +{ + TableJsonWide r; + r.lo = ~v.lo + 1; + r.hi = ~v.hi + ( r.lo == 0 ? 1 : 0 ); + return r; +} + +// v = v * m + a; the return is the carry out of 128 bits +inline uint32_t TableJsonWideMulAdd( TableJsonWide * v, uint32_t m, uint32_t a ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t carry = a; + for ( int i = 0; i < 4; i++ ) + { + uint64_t p = limb[i] * m + carry; + limb[i] = p & 0xffffffffull; + carry = p >> 32; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) carry; +} + +// v = v / d; the return is the remainder +inline uint32_t TableJsonWideDiv( TableJsonWide * v, uint32_t d ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t rem = 0; + for ( int i = 3; i >= 0; i-- ) + { + uint64_t cur = ( rem << 32 ) | limb[i]; + limb[i] = cur / d; + rem = cur % d; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) rem; +} + +// The storage of a wide kind, as lanes. A sixteen-byte storage is serialize's +// pair — native __int128 in the host's byte order, or the emulated struct with +// its low lane first — so the lanes are read in the host's order; a narrower +// storage is one lane, sign-extended for a signed kind. +inline TableJsonWide TableJsonWideLoad( const void * storage, uint32_t width, bool is_signed ) +{ + TableJsonWide v = { 0, 0 }; + if ( width == 16 ) + { + uint64_t half[2]; + memcpy( half, storage, 16 ); + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + v.lo = little ? half[0] : half[1]; + v.hi = little ? half[1] : half[0]; + return v; + } + v.lo = is_signed ? (uint64_t) TableJsonGetSigned( storage, width ) : TableJsonGetRaw( storage, width ); + v.hi = ( is_signed && ( v.lo >> 63 ) != 0 ) ? ~uint64_t( 0 ) : 0; + return v; +} + +inline void TableJsonWideStore( void * storage, uint32_t width, TableJsonWide v ) +{ + if ( width == 16 ) + { + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + uint64_t half[2]; + half[0] = little ? v.lo : v.hi; + half[1] = little ? v.hi : v.lo; + memcpy( storage, half, 16 ); + return; + } + TableJsonSetRaw( storage, width, v.lo ); +} + +// a counted field's companion: a string's length, a bytes' length, a counted +// array's count. Bounded by the declared extent on the way out, so a storage +// invariant a caller broke cannot walk off the end of the array. +inline int32_t TableJsonCount( const void * base, const TableFieldInfo * f ) +{ + if ( !f->counted ) + { + return f->array_bound; + } + int32_t count = 0; + memcpy( &count, (const uint8_t *) base + f->count_offset, sizeof( count ) ); + if ( count < 0 ) { count = 0; } + if ( count > f->array_bound ) { count = f->array_bound; } + return count; +} + +inline void TableJsonSetCount( void * base, const TableFieldInfo * f, int32_t count ) +{ + if ( f->counted ) + { + memcpy( (uint8_t *) base + f->count_offset, &count, sizeof( count ) ); + } +} + +// ---- what a field's kind expects to see in the text ---- +// +// One classifier, consulted by both directions, so a reader and a writer can +// never disagree about a kind's JSON form. 'o' object, 'a' array, 's' +// string, 'n' number, 'b' boolean. +// +// A vocabulary field is spelled by NAME: an enum is one name, a flags mask +// is the array of the names of its set bits. The two are told apart by the +// id column — an enum variant rides under a wire id, a flags BIT never does +// (docs/SPEC-TABLES.md §4), so a name function with no id function is flags. +// +// bytes(N) is the one kind whose element kind does not decide its form: it +// shares u8 with a plain array of u8, and rides as base64. The schema type +// name settles it, and "bytes" is a keyword no declaration can claim. +inline bool TableJsonIsBytes( const TableFieldInfo * f ) +{ + return f->is_array && f->kind == 6 && strcmp( f->type_name, "bytes" ) == 0; +} + +// An ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): its JSON form is an OBJECT +// keyed by variant name, not a positional array, because that is what the +// storage is — one slot per variant, addressed by the variant. +inline bool TableJsonIsKeyed( const TableFieldInfo * f ) +{ + return f->key_name != NULL; +} + +// THE KEY A STORAGE SLOT HOLDS (§2.4, §8): the storage shifts left, so slot i +// holds the key i + 1 and nothing is stored for None. This is the ONE place +// the walker spells the shift. +inline uint64_t TableJsonKeyedSlotKey( int64_t slot ) +{ + return (uint64_t) ( slot + 1 ); +} + +// A slot whose key names a variant of the keying enum. Every slot in +// [0, array_bound) does, unless the enum carries max-headroom variants outside +// a table closure, where a reserved value names nothing and its key id is 0 — +// the reserved id no declared name can fold to (§5). +inline bool TableJsonKeyedSlotValid( const TableFieldInfo * f, int64_t slot ) +{ + return f->key_id( TableJsonKeyedSlotKey( slot ) ) != 0; +} + +inline bool TableJsonIsFlags( const TableFieldInfo * f ) +{ + return f->enum_name != NULL && f->variant_id == NULL; +} + +inline bool TableJsonIsEnum( const TableFieldInfo * f ) +{ + return f->variant_id != NULL && f->arms == NULL; +} + +inline char TableJsonShape( const TableFieldInfo * f ) +{ + if ( TableJsonIsMap( f ) ) return 'o'; // a MAP: an object keyed by the KEY (§2.8) + if ( f->kind == 12 ) return 's'; // string + if ( TableJsonIsBytes( f ) ) return 's'; // bytes: base64 + if ( TableJsonIsKeyed( f ) ) return 'o'; // an object keyed by variant NAME + if ( f->is_array ) return 'a'; + if ( f->arms != NULL ) return 'o'; // union: an object with ONE key + if ( f->kind == 13 ) return 'o'; // nested table or type + if ( f->kind == 17 ) return f->table != NULL ? 'o' : 's'; // a pointer: the pointee's object in place, or null (§16.7); a byte buffer's string (§2.5) + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// the ELEMENT shape of an array field — the same classifier one level down +inline char TableJsonElementShape( const TableFieldInfo * f ) +{ + if ( f->arms != NULL ) return 'o'; // an element of an array of unions: one key, the arm (§2.6) + if ( f->kind == 13 ) return 'o'; + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// A guarded group rides only when its guard reads true — the wire's own +// elision (§4), carried into the text so a text and a wire written from one +// instance say the same thing. The guard is spelled as its branch condition +// over bool fields of the SAME type ("at_rest", "!at_rest", +// "active && has_target"), so evaluating it is a walk of the same +// descriptor. Nothing is inferred in the other direction: reading places +// every key it can name, and the guard is a plain bool key (§16.2). +inline bool TableJsonGuardHolds( const void * base, const TableTypeInfo * info, const char * guard ) +{ + const char * p = guard; + for ( ;; ) + { + while ( *p == ' ' || *p == '&' ) { p++; } + if ( *p == 0 ) { return true; } + bool want = true; + if ( *p == '!' ) { want = false; p++; } + const char * start = p; + while ( *p != 0 && *p != ' ' && *p != '&' ) { p++; } + size_t length = (size_t) ( p - start ); + bool value = false; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( strlen( f->name ) == length && strncmp( f->name, start, length ) == 0 ) + { + value = TableJsonGetRaw( (const uint8_t *) base + f->offset, f->elem_size ) != 0; + break; + } + } + if ( value != want ) { return false; } + } +} + +// ---- writing ---- + +// The writer sink MEASURES when the buffer is NULL and WRITES when it is +// not, over one code path — so measure and write agree byte for byte, the +// wire's invariant (§9) carried across. +struct TableJsonOut +{ + char * buffer; + int64_t capacity; + int64_t offset; + bool overflow; + void * graph; // the pointered write's identity map (§16.7); NULL for a fixed table + + void raw( const char * data, int64_t count ) + { + if ( buffer != NULL ) + { + if ( offset + count > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) count ); + } + offset += count; + } + void put( char c ) { raw( &c, 1 ); } + void text( const char * s ) { raw( s, (int64_t) strlen( s ) ); } + void line( int32_t depth ) + { + put( '\n' ); + for ( int32_t i = 0; i < depth; i++ ) { raw( " ", 2 ); } + } +}; + +inline const char * TableJsonBase64Alphabet() +{ + return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; +} + +inline void TableJsonWriteBase64( TableJsonOut & out, const uint8_t * data, int32_t length ) +{ + const char * alphabet = TableJsonBase64Alphabet(); + out.put( '"' ); + int32_t i = 0; + for ( ; i + 3 <= length; i += 3 ) + { + uint32_t triple = ( uint32_t( data[i] ) << 16 ) | ( uint32_t( data[i+1] ) << 8 ) | uint32_t( data[i+2] ); + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], + alphabet[ ( triple >> 6 ) & 0x3f ], alphabet[ triple & 0x3f ] }; + out.raw( quad, 4 ); + } + if ( i < length ) + { + int32_t left = length - i; + uint32_t triple = uint32_t( data[i] ) << 16; + if ( left == 2 ) { triple |= uint32_t( data[i+1] ) << 8; } + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], '=', '=' }; + if ( left == 2 ) { quad[2] = alphabet[ ( triple >> 6 ) & 0x3f ]; } + out.raw( quad, 4 ); + } + out.put( '"' ); +} + +// One UTF-8 sequence at s, or -1 when the bytes there are not one. Rejects +// the lot: a stray continuation, an overlong form, a surrogate half, and +// anything past U+10FFFF. +inline int32_t TableJsonUtf8( const char * s, int32_t remaining, int32_t * width ) +{ + unsigned char lead = (unsigned char) s[0]; + int32_t want = 0; + int32_t code = 0; + if ( lead < 0x80 ) { *width = 1; return lead; } + else if ( lead >= 0xc2 && lead <= 0xdf ) { want = 2; code = lead & 0x1f; } + else if ( lead >= 0xe0 && lead <= 0xef ) { want = 3; code = lead & 0x0f; } + else if ( lead >= 0xf0 && lead <= 0xf4 ) { want = 4; code = lead & 0x07; } + else { return -1; } + if ( remaining < want ) { return -1; } + for ( int32_t i = 1; i < want; i++ ) + { + unsigned char next = (unsigned char) s[i]; + if ( ( next & 0xc0 ) != 0x80 ) { return -1; } + code = ( code << 6 ) | ( next & 0x3f ); + } + if ( want == 3 && code < 0x800 ) { return -1; } // overlong + if ( want == 4 && code < 0x10000 ) { return -1; } // overlong + if ( code >= 0xd800 && code <= 0xdfff ) { return -1; } // a surrogate half + if ( code > 0x10ffff ) { return -1; } + *width = want; + return code; +} + +// A JSON text MUST be valid UTF-8 (RFC 8259 §8.1). The read path is +// byte-transparent — the wire imposes no encoding (§3) and a string may hold +// anything — so the WRITER is where that obligation is met: a byte that is +// not part of a well-formed sequence is written as U+FFFD, one per bad byte, +// and never raw. A text this walk writes is therefore readable by any +// conforming parser, which a raw byte would not be. The cost is stated +// plainly: for a string holding invalid UTF-8, the round trip is NOT +// byte-identical, because the alternative is emitting a text that is not +// JSON. +inline void TableJsonWriteString( TableJsonOut & out, const char * s, int32_t length ) +{ + static const char hex[] = "0123456789abcdef"; + out.put( '"' ); + for ( int32_t i = 0; i < length; i++ ) + { + unsigned char c = (unsigned char) s[i]; + switch ( c ) + { + case '"': out.raw( "\\\"", 2 ); break; + case '\\': out.raw( "\\\\", 2 ); break; + case '\b': out.raw( "\\b", 2 ); break; + case '\f': out.raw( "\\f", 2 ); break; + case '\n': out.raw( "\\n", 2 ); break; + case '\r': out.raw( "\\r", 2 ); break; + case '\t': out.raw( "\\t", 2 ); break; + default: + if ( c < 0x20 ) + { + char escape[6] = { '\\', 'u', '0', '0', hex[ c >> 4 ], hex[ c & 0xf ] }; + out.raw( escape, 6 ); + } + else if ( c < 0x80 ) + { + out.put( (char) c ); + } + else + { + int32_t width = 0; + if ( TableJsonUtf8( s + i, length - i, &width ) < 0 ) + { + out.raw( "\xef\xbf\xbd", 3 ); // U+FFFD, one per bad byte + } + else + { + out.raw( s + i, width ); + i += width - 1; + } + } + break; + } + } + out.put( '"' ); +} + +inline void TableJsonWriteUnsigned( TableJsonOut & out, uint64_t value ) +{ + char digits[24]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) ( value % 10 ) ); + value /= 10; + } while ( value != 0 ); + char text[24]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); +} + +inline void TableJsonWriteSigned( TableJsonOut & out, int64_t value ) +{ + if ( value < 0 ) + { + out.put( '-' ); + TableJsonWriteUnsigned( out, uint64_t( 0 ) - (uint64_t) value ); + return; + } + TableJsonWriteUnsigned( out, (uint64_t) value ); +} + +// A wide kind writes its raw storage as §16.2's text: a 128-bit integer as a +// decimal integer; a fixed value in WHOLE UNITS as the shortest exact decimal +// with at least one fractional digit (1.0, -0.25), the spelling the schema text +// gives a fixed default. The fraction terminates because a dyadic fraction has +// a finite decimal expansion — at most F digits. +inline void TableJsonWriteWide( TableJsonOut & out, const void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + TableJsonWide v = TableJsonWideLoad( storage, f->elem_size, is_signed ); + if ( is_signed && TableJsonWideNegative( v ) ) + { + out.put( '-' ); + v = TableJsonWideNeg( v ); + } + int frac = f->frac_bits; + TableJsonWide whole = TableJsonWideShr( v, frac ); + char digits[40]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) TableJsonWideDiv( &whole, 10 ) ); + } while ( !TableJsonWideZero( whole ) ); + char text[40]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); + if ( !TableJsonKindFixed( f->kind ) ) { return; } + out.put( '.' ); + // the fraction bits alone: v with everything at and above bit F cleared + TableJsonWide fraction = v; + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + if ( frac == 0 ) { fraction.lo = 0; } + if ( TableJsonWideZero( fraction ) ) + { + out.put( '0' ); + return; + } + while ( !TableJsonWideZero( fraction ) ) + { + // ×10: the digit is what lands at and above bit F, including the + // carry out of 128 bits when F leaves no room for it below + uint32_t carry = TableJsonWideMulAdd( &fraction, 10, 0 ); + uint64_t digit = TableJsonWideShr( fraction, frac ).lo; + if ( frac > 64 ) { digit |= uint64_t( carry ) << ( 128 - frac ); } + out.put( (char) ( '0' + (int) digit ) ); + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + } +} + +// A float writes at the SHORTEST precision that reads back as the same value +// at the field's own width, so a round trip is exact and a text stays +// readable. Non-finite values have no JSON spelling at all, and the writer +// REFUSES rather than losing one silently — the same rule measure and save +// already apply to an enum value no variant names (§5). +inline bool TableJsonWriteFloat( TableJsonOut & out, double value, bool single ) +{ + if ( !TableJsonFinite( value ) ) { return false; } + char text[64]; + int low = single ? 6 : 15; + int high = single ? 9 : 17; + int length = 0; + for ( int digits = low; ; digits++ ) + { + length = snprintf( text, sizeof( text ), "%.*g", digits, value ); + if ( length <= 0 || length >= (int) sizeof( text ) ) { return false; } + if ( digits >= high ) { break; } + // the round-trip check runs BEFORE the decimal point is normalised: + // the token still carries whatever point snprintf just produced + if ( single ) + { + if ( (double) strtof( text, NULL ) == value ) { break; } + } + else + { + if ( strtod( text, NULL ) == value ) { break; } + } + } + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int i = 0; i < length; i++ ) + { + if ( text[i] == point ) { text[i] = '.'; } + } + } + out.raw( text, length ); + return true; +} + +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration writes through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ); + +// one scalar, at one storage address: a nested object, a union, a +// vocabulary, or a number +inline bool TableJsonWriteScalar( TableJsonOut & out, const void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; None is {} + const TableUnionInfo * arms = f->arms(); + uint64_t tag = TableJsonGetRaw( (const uint8_t *) storage + arms->tag_offset, arms->tag_size ); + if ( tag == 0 ) + { + out.raw( "{}", 2 ); + return true; + } + if ( (int64_t) tag > f->enum_max ) + { + return false; // a tag no arm names, exactly as measure refuses it + } + const char * arm = f->enum_name( tag ); + // and refuse on the NAME, not merely on the bound: §16.2 says a value + // no variant NAMES is refused, so the check is the name. Writing + // whatever came back would emit "???", a spelling the reader counts + // as unknown — a silent round-trip loss in place of a refusal. + if ( !TableJsonNamed( arm ) ) { return false; } + out.put( '{' ); + out.line( depth + 1 ); + TableJsonWriteString( out, arm, (int32_t) strlen( arm ) ); + out.raw( ": ", 2 ); + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2): an arm that names + // no declaration carries the FIELD descriptor a field of its type + // would carry, offsets taken inside the union storage (§2.6), so the + // value walks through the field writer one key down. + if ( arms->arms[tag].field != NULL ) + { + if ( !TableJsonWriteField( out, storage, arms->arms[tag].field, depth + 1 ) ) + { + return false; + } + } + else if ( arms->arms[tag].table == NULL ) + { + out.raw( "null", 4 ); // a payload-free arm: the name selects it (§2.6) + } + else if ( !TableJsonWriteValue( out, (const uint8_t *) storage + arms->arms[tag].offset, arms->arms[tag].table, depth + 1 ) ) + { + return false; + } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->kind == 13 ) + { + return TableJsonWriteValue( out, storage, f->table, depth ); + } + if ( TableJsonIsEnum( f ) ) + { + uint64_t value = TableJsonGetRaw( storage, f->elem_size ); + // a value no variant names has no text spelling, exactly as it has no + // wire identity: the writer REFUSES rather than writing None over it, + // the rule measure and save already apply (docs/SPEC-TABLES.md §5) + if ( (int64_t) value > f->enum_max ) { return false; } + if ( value != 0 && f->variant_id( value ) == 0 ) { return false; } + const char * name = f->enum_name( value ); + if ( !TableJsonNamed( name ) ) { return false; } + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + return true; + } + if ( TableJsonIsFlags( f ) ) + { + uint64_t bits = TableJsonGetRaw( storage, f->elem_size ); + if ( bits == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + bool first = true; + for ( int64_t bit = 0; bit < 64; bit++ ) + { + if ( ( bits & ( uint64_t( 1 ) << bit ) ) == 0 ) { continue; } + if ( bit > f->enum_max ) + { + return false; // a bit no variant names has no text spelling + } + const char * name = f->enum_name( (uint64_t) bit ); + if ( !TableJsonNamed( name ) ) { return false; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + } + out.line( depth ); + out.put( ']' ); + return true; + } + switch ( f->kind ) + { + case 1: + out.text( TableJsonGetRaw( storage, f->elem_size ) != 0 ? "true" : "false" ); + return true; + case 10: + { + float v = 0.0f; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, (double) v, true ); + } + case 11: + { + double v = 0.0; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, v, false ); + } + case 2: case 3: case 4: case 5: + TableJsonWriteSigned( out, TableJsonGetSigned( storage, f->elem_size ) ); + return true; + default: + if ( TableJsonKindWide( f->kind ) ) + { + TableJsonWriteWide( out, storage, f ); + return true; + } + TableJsonWriteUnsigned( out, TableJsonGetRaw( storage, f->elem_size ) ); + return true; + } +} + +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ) +{ + const uint8_t * storage = (const uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonWriteMap( out, (const void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } + if ( f->kind == 17 && !f->is_array ) + { + return TableJsonWritePointer( out, storage, f, depth ); + } + if ( f->kind == 17 ) + { + // an ARRAY OF POINTERS (§2.1): the pointer row per element — the + // pointee's object in place, null, or `&node` for a shared one (§16.7) + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWritePointer( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; + } + if ( f->kind == 12 ) + { + TableJsonWriteString( out, (const char *) storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + TableJsonWriteBase64( out, storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + // one entry per SLOT, keyed by the variant that owns it, so inserting + // a variant next season moves nothing in the text either. Slot i holds + // the key i + 1: nothing is stored for None, so nothing is written for it. + out.put( '{' ); + bool first = true; + for ( int64_t slot = 0; slot < f->array_bound; slot++ ) + { + if ( !TableJsonKeyedSlotValid( f, slot ) ) { continue; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + const char * key = f->key_name( TableJsonKeyedSlotKey( slot ) ); + TableJsonWriteString( out, key, (int32_t) strlen( key ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteScalar( out, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + if ( first ) { out.raw( "}", 1 ); return true; } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->is_array ) + { + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWriteScalar( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + out.line( depth ); + out.put( ']' ); + return true; + } + return TableJsonWriteScalar( out, storage, f, depth ); +} + +// One instance's fields, in DECLARATION ORDER, defaults included — a text is +// for people and tools, and a text that elides is a text a reader has to know +// the schema to complete. `any` says whether the object is already open on +// entry — a shared node's `&node` opens it before the fields (§16.7) — and +// whether it is open on return. +inline bool TableJsonWriteFields( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth, bool & any ) +{ + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( f->guard[0] != 0 && !TableJsonGuardHolds( base, info, f->guard ) ) { continue; } + // an ABSENT optional writes no key: presence of the key IS the + // presence (§16.2), so an absent field is an absent key and nothing + // else would read back as absent + if ( f->optional && + TableJsonGetRaw( (const uint8_t *) base + f->present_offset, 1 ) == 0 ) + { + continue; + } + if ( !any ) { out.put( '{' ); } + else { out.put( ',' ); } + any = true; + out.line( depth + 1 ); + TableJsonWriteString( out, f->json, (int32_t) strlen( f->json ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, base, f, depth + 1 ) ) { return false; } + } + return true; +} + +// One instance as one object. The writer carries the reader's depth cap +// (§16.2): a pointer chain nests as deep as it is long (§16.7), and a text the +// writer produced past the cap would be a text the reader refuses. +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { return false; } + bool any = false; + if ( !TableJsonWriteFields( out, base, info, depth, any ) ) { return false; } + if ( !any ) + { + out.raw( "{}", 2 ); + return true; + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- reading ---- + +struct TableJsonIn +{ + const char * text; + int64_t size; + int64_t pos; + TableReport * report; + bool bad; // the text is not JSON: the walk stops and keeps what it placed + void * graph; // the pointered read's builder and label map (§16.7); NULL for a fixed table +}; + +inline void TableJsonSpace( TableJsonIn & in ) +{ + while ( in.pos < in.size ) + { + char c = in.text[in.pos]; + if ( c == ' ' || c == '\t' || c == '\n' || c == '\r' ) { in.pos++; continue; } + // comments are not JSON, and a walk that guessed at one would be + // reading a dialect nobody wrote down + if ( c == '/' ) { in.bad = true; } + return; + } +} + +inline char TableJsonPeek( TableJsonIn & in ) +{ + TableJsonSpace( in ); + return in.pos < in.size ? in.text[in.pos] : 0; +} + +// the shape of the value sitting at the cursor, without consuming it +inline char TableJsonValueShape( TableJsonIn & in ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return 'o'; + case '[': return 'a'; + case '"': return 's'; + case 't': case 'f': return 'b'; + case 'n': return 'z'; + case 0: return 0; + default: return 'n'; + } +} + +inline bool TableJsonLiteral( TableJsonIn & in, const char * word ) +{ + int64_t length = (int64_t) strlen( word ); + if ( in.pos + length > in.size || memcmp( in.text + in.pos, word, (size_t) length ) != 0 ) + { + in.bad = true; + return false; + } + in.pos += length; + return true; +} + +// one \uXXXX escape body; -1 when the four hex digits are not there +inline int TableJsonHex4( TableJsonIn & in ) +{ + if ( in.pos + 4 > in.size ) { return -1; } + int value = 0; + for ( int i = 0; i < 4; i++ ) + { + char c = in.text[in.pos + i]; + int digit; + if ( c >= '0' && c <= '9' ) { digit = c - '0'; } + else if ( c >= 'a' && c <= 'f' ) { digit = c - 'a' + 10; } + else if ( c >= 'A' && c <= 'F' ) { digit = c - 'A' + 10; } + else { return -1; } + value = ( value << 4 ) | digit; + } + in.pos += 4; + return value; +} + +inline int32_t TableJsonEncodeUtf8( uint32_t code, char * unit ) +{ + if ( code < 0x80 ) { unit[0] = (char) code; return 1; } + if ( code < 0x800 ) + { + unit[0] = (char) ( 0xc0 | ( code >> 6 ) ); + unit[1] = (char) ( 0x80 | ( code & 0x3f ) ); + return 2; + } + if ( code < 0x10000 ) + { + unit[0] = (char) ( 0xe0 | ( code >> 12 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( code & 0x3f ) ); + return 3; + } + unit[0] = (char) ( 0xf0 | ( code >> 18 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 12 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[3] = (char) ( 0x80 | ( code & 0x3f ) ); + return 4; +} + +// Scan one JSON string into a caller buffer. Bytes are appended ONE CODE +// POINT AT A TIME — an escape's encoding, or a UTF-8 sequence read whole — +// so a string longer than the field is clamped AT A CODE POINT BOUNDARY and +// never cut through a multi-byte character. Clamping is counted, never +// fatal, exactly as it is on the wire (§4). A NULL destination scans past a +// string without keeping it. +inline bool TableJsonScanString( TableJsonIn & in, char * out, int32_t capacity, int32_t * length ) +{ + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + int32_t placed = 0; + bool clamped = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos]; + if ( c == '"' ) { in.pos++; break; } + char unit[4]; + int32_t unit_length = 0; + if ( c == '\\' ) + { + in.pos++; + if ( in.pos >= in.size ) { in.bad = true; return false; } + char escape = in.text[in.pos++]; + switch ( escape ) + { + case '"': unit[0] = '"'; unit_length = 1; break; + case '\\': unit[0] = '\\'; unit_length = 1; break; + case '/': unit[0] = '/'; unit_length = 1; break; + case 'b': unit[0] = '\b'; unit_length = 1; break; + case 'f': unit[0] = '\f'; unit_length = 1; break; + case 'n': unit[0] = '\n'; unit_length = 1; break; + case 'r': unit[0] = '\r'; unit_length = 1; break; + case 't': unit[0] = '\t'; unit_length = 1; break; + case 'u': + { + int high = TableJsonHex4( in ); + if ( high < 0 ) { in.bad = true; return false; } + uint32_t code = (uint32_t) high; + if ( high >= 0xd800 && high <= 0xdbff && in.pos + 2 <= in.size && + in.text[in.pos] == '\\' && in.text[in.pos + 1] == 'u' ) + { + int64_t mark = in.pos; + in.pos += 2; + int low = TableJsonHex4( in ); + if ( low >= 0xdc00 && low <= 0xdfff ) + { + code = 0x10000 + ( ( (uint32_t) high - 0xd800 ) << 10 ) + ( (uint32_t) low - 0xdc00 ); + } + else + { + in.pos = mark; // a lone lead surrogate rides as itself + } + } + // a surrogate half that never found its partner has no + // UTF-8 encoding: encoding it anyway would manufacture + // CESU-8 — invalid UTF-8 — out of input that was valid + // JSON, so it reads as the replacement character + if ( code >= 0xd800 && code <= 0xdfff ) { code = 0xfffd; } + unit_length = TableJsonEncodeUtf8( code, unit ); + break; + } + default: in.bad = true; return false; + } + } + else if ( (unsigned char) c < 0x20 ) + { + in.bad = true; // a raw control character is not a JSON string body + return false; + } + else + { + // a UTF-8 sequence read WHOLE, so the clamp below can only land + // between code points. Only bytes that ACTUALLY look like + // continuations are taken: the wire imposes no encoding (§3), so + // a string may legitimately hold a stray lead byte, and one at + // the end of a text must not swallow the closing quote. + unsigned char lead = (unsigned char) c; + int32_t want = 1; + if ( ( lead & 0xe0 ) == 0xc0 ) { want = 2; } + else if ( ( lead & 0xf0 ) == 0xe0 ) { want = 3; } + else if ( ( lead & 0xf8 ) == 0xf0 ) { want = 4; } + unit[0] = c; + in.pos++; + unit_length = 1; + while ( unit_length < want && in.pos < in.size && + ( (unsigned char) in.text[in.pos] & 0xc0 ) == 0x80 ) + { + unit[unit_length++] = in.text[in.pos++]; + } + } + if ( out == NULL ) + { + placed += unit_length; // measured and not kept: a byte buffer's read sizes its node this way (§2.5) + } + else if ( placed + unit_length <= capacity ) + { + memcpy( out + placed, unit, (size_t) unit_length ); + placed += unit_length; + } + else + { + clamped = true; + } + } + if ( clamped ) { in.report->clamped++; } + if ( length != NULL ) { *length = placed; } + return true; +} + +// the numeric token at the cursor, copied out whole; false = not a number +// Scan one number, to JSON's OWN grammar (RFC 8259 §6) and not to a run of +// number-ish characters: +// +// number = [ "-" ] int [ frac ] [ exp ] +// int = "0" / ( digit1-9 *digit ) +// frac = "." 1*digit +// exp = ( "e" / "E" ) [ "-" / "+" ] 1*digit +// +// Scanning the production is what makes a typo in an authoring file a +// DIAGNOSTIC rather than a value: "1-2" scans as 1 and leaves "-2" where the +// object expects a comma, so the text is malformed — which is what §16.2 +// already promises. A permissive scan would hand "1-2" to a digit loop and +// report a clamp, and a config pipeline would never hear about it. Leading +// "+", leading zeros, ".5" and "3." are not JSON either. +inline bool TableJsonWalkNumber( TableJsonIn & in, bool * integral ) +{ + TableJsonSpace( in ); + bool whole = true; + if ( in.pos < in.size && in.text[in.pos] == '-' ) { in.pos++; } + // int: a lone zero, or a non-zero digit and any digits after it + if ( in.pos >= in.size ) { return false; } + if ( in.text[in.pos] == '0' ) + { + in.pos++; + } + else if ( in.text[in.pos] >= '1' && in.text[in.pos] <= '9' ) + { + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + } + else + { + return false; + } + // frac + if ( in.pos < in.size && in.text[in.pos] == '.' ) + { + in.pos++; + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + // exp + if ( in.pos < in.size && ( in.text[in.pos] == 'e' || in.text[in.pos] == 'E' ) ) + { + in.pos++; + if ( in.pos < in.size && ( in.text[in.pos] == '-' || in.text[in.pos] == '+' ) ) { in.pos++; } + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + *integral = whole; + return true; +} + +// the same production, with the token kept for conversion +inline bool TableJsonScanNumber( TableJsonIn & in, char * token, int32_t capacity, int32_t * length, bool * integral ) +{ + TableJsonSpace( in ); + int64_t start = in.pos; + if ( !TableJsonWalkNumber( in, integral ) ) { return false; } + int64_t count = in.pos - start; + if ( count <= 0 || count >= capacity ) { return false; } + memcpy( token, in.text + start, (size_t) count ); + token[count] = 0; + *length = (int32_t) count; + return true; +} + +// the token's exact double, through the runtime's own converter — which +// speaks the LOCALE's decimal point, so the token crosses back over that +// character on its way in +inline double TableJsonTokenDouble( const char * token, int32_t length, bool single ) +{ + char work[kTableJsonMaxNumber]; + memcpy( work, token, (size_t) length ); + work[length] = 0; + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int32_t i = 0; i < length; i++ ) + { + if ( work[i] == '.' ) { work[i] = point; } + } + } + if ( single ) { return (double) strtof( work, NULL ); } + return strtod( work, NULL ); +} + +// the token's exact integer, parsed digit by digit so no width and no +// locale can move it. Saturation is reported as a clamp, the wire's rule for +// a value outside what the reader can hold (§4). +inline int64_t TableJsonTokenInteger( const char * token, int32_t length, bool is_signed, bool * saturated ) +{ + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) + { + negative = token[i] == '-'; + i++; + } + uint64_t magnitude = 0; + bool over = false; + for ( ; i < length; i++ ) + { + uint64_t digit = (uint64_t) ( token[i] - '0' ); + if ( magnitude > ( UINT64_MAX - digit ) / 10 ) { over = true; break; } + magnitude = magnitude * 10 + digit; + } + if ( !is_signed ) + { + // -0 IS zero, and clamping it would report an event that did not + // happen; only a real negative magnitude is out of range here + if ( negative ) { *saturated = magnitude != 0; return 0; } + if ( over ) { *saturated = true; return (int64_t) UINT64_MAX; } + *saturated = false; + return (int64_t) magnitude; + } + if ( negative ) + { + if ( over || magnitude > ( uint64_t( 1 ) << 63 ) ) { *saturated = true; return INT64_MIN; } + *saturated = false; + if ( magnitude == ( uint64_t( 1 ) << 63 ) ) { return INT64_MIN; } + return -(int64_t) magnitude; + } + if ( over || magnitude > (uint64_t) INT64_MAX ) { *saturated = true; return INT64_MAX; } + *saturated = false; + return (int64_t) magnitude; +} + +// A number token into a wide kind's raw storage (docs/SPEC-TABLES.md §16.2). A +// 128-bit integer takes any token whose VALUE is integral; a fixed field any +// token whose value is EXACTLY representable in its Q I.F — a finer fraction +// is the wrong shape for the field, counted as a kind mismatch and never +// rounded, the rule SPEC.md §4.6 gives a fixed default. A magnitude past 128 +// bits saturates and counts as a clamp, as an int64 field saturates at +// INT64_MAX; the declared range clamps after it, on the RAW scale, as it does +// for every bounded scalar. +// +// The token is normalized to its digits with the decimal point after "point" +// of them. An integer part past 40 digits is above 2^128 whatever the digits +// are, and a value below 10^-40 is finer than 2^-127, the finest fraction any +// F can spell — so outside that band the answer is known without the +// arithmetic, and a token spelling 1e999999999 costs nothing to refuse. +inline bool TableJsonReadWide( TableJsonIn & in, const char * token, int32_t length, void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + int frac = f->frac_bits; + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { negative = token[i] == '-'; i++; } + const char * int_digits = token + i; + int32_t int_len = 0; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { int_len++; i++; } + const char * frac_digits = token + i; + int32_t frac_len = 0; + if ( i < length && token[i] == '.' ) + { + i++; + frac_digits = token + i; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { frac_len++; i++; } + } + int64_t exp = 0; + if ( i < length && ( token[i] == 'e' || token[i] == 'E' ) ) + { + i++; + bool exp_negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { exp_negative = token[i] == '-'; i++; } + while ( i < length && token[i] >= '0' && token[i] <= '9' ) + { + if ( exp < 100000 ) { exp = exp * 10 + ( token[i] - '0' ); } + i++; + } + if ( exp_negative ) { exp = -exp; } + } + // the digits, with the point after "point" of them; leading and trailing + // zeros stripped. digit( k ) reads the k-th of the int and frac runs. + int32_t start = 0, end = int_len + frac_len; + int64_t point = int_len + exp; + while ( start < end && ( start < int_len ? int_digits[start] : frac_digits[start - int_len] ) == '0' ) { start++; point--; } + while ( end > start && ( end - 1 < int_len ? int_digits[end - 1] : frac_digits[end - 1 - int_len] ) == '0' ) { end--; } + + TableJsonWide raw = { 0, 0 }; + bool saturated = false; + TableJsonWide signed_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) >> 1 }; + TableJsonWide signed_min = { 0, uint64_t( 1 ) << 63 }; + TableJsonWide unsigned_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) }; + if ( start == end ) + { + // zero, and -0 IS zero + } + else if ( point > 40 ) + { + saturated = true; + if ( !negative ) { raw = is_signed ? signed_max : unsigned_max; } + else if ( is_signed ) { raw = signed_min; } + } + else if ( point < -40 ) + { + in.report->kind_mismatch++; // finer than any F can spell + return true; + } + else + { + // the fraction FIRST, so an inexact value is the wrong shape whatever + // its magnitude: its digits, with the zeros a negative point puts in + // front, doubled F times; each doubling's carry is the next bit, and + // the value is exact iff nothing is left after the last one + char fd[kTableJsonMaxNumber + 48]; + int32_t fn = 0; + for ( int64_t z = point; z < 0; z++ ) { fd[fn++] = 0; } + for ( int32_t k = (int32_t) ( point > 0 ? point : 0 ) + start; k < end; k++ ) + { + fd[fn++] = (char) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ); + } + TableJsonWide fraction = { 0, 0 }; + for ( int b = 0; b < frac; b++ ) + { + int carry = 0; + for ( int32_t k = fn - 1; k >= 0; k-- ) + { + int d = fd[k] * 2 + carry; + fd[k] = (char) ( d % 10 ); + carry = d / 10; + } + fraction = TableJsonWideShl( fraction, 1 ); + fraction.lo |= (uint64_t) carry; + } + for ( int32_t k = 0; k < fn; k++ ) + { + if ( fd[k] != 0 ) + { + in.report->kind_mismatch++; + return true; + } + } + // then the whole part, saturating past 128 bits + TableJsonWide whole = { 0, 0 }; + for ( int64_t k = start; k < start + point && !saturated; k++ ) + { + uint32_t digit = k < end ? (uint32_t) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ) : 0; + if ( TableJsonWideMulAdd( &whole, 10, digit ) != 0 ) { saturated = true; } + } + if ( !saturated && frac > 0 && !TableJsonWideZero( TableJsonWideShr( whole, 128 - frac ) ) ) { saturated = true; } + if ( !saturated ) + { + raw = TableJsonWideShl( whole, frac ); + raw.lo |= fraction.lo; + raw.hi |= fraction.hi; + } + if ( is_signed ) + { + if ( !saturated && !negative && TableJsonWideNegative( raw ) ) { saturated = true; } + if ( !saturated && negative && TableJsonWideCompare( raw, signed_min, false ) > 0 ) { saturated = true; } + if ( saturated ) { raw = negative ? signed_min : signed_max; } + else if ( negative ) { raw = TableJsonWideNeg( raw ); } + } + else + { + if ( saturated ) { raw = unsigned_max; } + if ( negative && !TableJsonWideZero( raw ) ) { raw.lo = 0; raw.hi = 0; saturated = true; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->wide != NULL ) + { + TableJsonWide lo = { f->wide->lo[0], f->wide->lo[1] }; + TableJsonWide hi = { f->wide->hi[0], f->wide->hi[1] }; + if ( TableJsonWideCompare( raw, lo, is_signed ) < 0 ) { raw = lo; in.report->clamped++; } + else if ( TableJsonWideCompare( raw, hi, is_signed ) > 0 ) { raw = hi; in.report->clamped++; } + } + TableJsonWideStore( storage, f->elem_size, raw ); + return true; +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ); + +inline bool TableJsonSkipContainer( TableJsonIn & in, char close, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; // the opening bracket + bool first = true; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == close ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + if ( close == '}' ) + { + // the key is kept, because a skipped OBJECT may still be a + // pointer's: an `&node` opening it names a node the storage could + // not hold, and the numbering has to survive the drop (§16.7). + // Anywhere but first, the prefix is the reserved key out of place + // — in a pointered unit; a fixed unit skips the value whole. + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( key[0] == '&' && in.graph != NULL ) + { + if ( !first ) { in.report->malformed = true; in.bad = true; return false; } + if ( !TableJsonSkippedAmpersand( in, key, depth ) ) { return false; } + first = false; + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } + } + first = false; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return TableJsonSkipContainer( in, '}', depth ); + case '[': return TableJsonSkipContainer( in, ']', depth ); + case '"': return TableJsonScanString( in, NULL, 0, NULL ); + case 't': return TableJsonLiteral( in, "true" ); + case 'f': return TableJsonLiteral( in, "false" ); + case 'n': return TableJsonLiteral( in, "null" ); + case 0: in.bad = true; return false; + default: + { + // consumed, never converted: skipping needs no buffer, and this + // is the one walk a hostile text drives to the depth cap. It is + // the SAME production the value path scans, so an unknown key + // cannot smuggle past a number a named key would refuse. + bool integral = false; + if ( !TableJsonWalkNumber( in, &integral ) ) { in.bad = true; return false; } + return true; + } + } +} + +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration reads through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ); + +// place one scalar at one storage address +inline bool TableJsonReadScalar( TableJsonIn & in, void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; {} is None, and + // two keys is a text this walk will not guess at + const TableUnionInfo * arms = f->arms(); + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, 0 ); + if ( TableJsonPeek( in ) == '}' ) { in.pos++; return true; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t tag = 0; + for ( int64_t t = 1; t <= f->enum_max; t++ ) + { + if ( strcmp( f->enum_name( (uint64_t) t ), key ) == 0 ) { tag = t; break; } + } + if ( tag == 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + void * payload = (uint8_t *) storage + arms->arms[tag].offset; + const TableFieldInfo * arm = arms->arms[tag].field; + bool placed = true; + if ( arm != NULL ) + { + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2). A value of + // the wrong shape for that row is a KIND MISMATCH: the union + // reads None, the event is counted, and the enclosing object + // continues — the rule a FIELD's value lives under, one key + // down. A pointer arm's null is a null pointer, not a shape + // error, exactly as a pointer field's is (§16.7). + char got = TableJsonValueShape( in ); + if ( arm->kind == 17 && !arm->is_array && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + memset( payload, 0, (size_t) arms->arms[tag].size ); + } + else if ( got != TableJsonShape( arm ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( arm->kind == 17 && !arm->is_array ) + { + // A POINTER ARM'S VALUE IS THE POINTEE IN PLACE, or a + // node reference to one (§16.7) — the read a pointer + // FIELD takes, which is not the scalar walk + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadPointer( in, payload, arm, depth + 1 ) ) { return false; } + } + else + { + // SELECTION ZERO-ESTABLISHES THE ARM (SPEC §5): an arm + // takes no specified default, so zero is the establish + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadField( in, storage, arm, depth + 1 ) ) { return false; } + } + } + else if ( arms->arms[tag].table != NULL ) + { + if ( TableJsonValueShape( in ) != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else + { + arms->arms[tag].table->reset( payload ); + if ( !TableJsonReadTable( in, payload, arms->arms[tag].table, depth + 1 ) ) { return false; } + } + } + else + { + // A PAYLOAD-FREE ARM'S VALUE IS null (§2.6): the arm name + // selects it and there is nothing to place + if ( TableJsonValueShape( in ) != 'z' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( !TableJsonLiteral( in, "null" ) ) + { + return false; + } + } + if ( placed ) + { + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, (uint64_t) tag ); + } + } + char c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; // a second key: a one-of with two arms is not a value + return false; + } + if ( f->kind == 13 ) + { + f->table->reset( storage ); + return TableJsonReadTable( in, storage, f->table, depth + 1 ); + } + if ( TableJsonIsEnum( f ) ) + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + for ( int64_t v = 0; v <= f->enum_max; v++ ) + { + if ( strcmp( f->enum_name( (uint64_t) v ), name ) == 0 ) + { + TableJsonSetRaw( storage, f->elem_size, (uint64_t) v ); + return true; + } + } + // a name this build cannot name reads as None and counts as unknown, + // exactly as an unknown variant id does on the wire (§4) + TableJsonSetRaw( storage, f->elem_size, 0 ); + in.report->unknown++; + return true; + } + if ( TableJsonIsFlags( f ) ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + uint64_t bits = 0; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( c != '"' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + bool found = false; + for ( int64_t bit = 0; bit <= f->enum_max; bit++ ) + { + if ( strcmp( f->enum_name( (uint64_t) bit ), name ) == 0 ) + { + bits |= uint64_t( 1 ) << bit; + found = true; + break; + } + } + if ( !found ) { in.report->unknown++; } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + TableJsonSetRaw( storage, f->elem_size, bits ); + return true; + } + if ( f->kind == 1 ) + { + char c = TableJsonPeek( in ); + if ( c == 't' ) { if ( !TableJsonLiteral( in, "true" ) ) { return false; } TableJsonSetRaw( storage, f->elem_size, 1 ); return true; } + if ( !TableJsonLiteral( in, "false" ) ) { return false; } + TableJsonSetRaw( storage, f->elem_size, 0 ); + return true; + } + char token[kTableJsonMaxNumber]; + int32_t length = 0; + bool integral = false; + if ( !TableJsonScanNumber( in, token, kTableJsonMaxNumber, &length, &integral ) ) + { + in.bad = true; + return false; + } + if ( TableJsonKindWide( f->kind ) ) + { + return TableJsonReadWide( in, token, length, storage, f ); + } + if ( f->kind == 10 || f->kind == 11 ) + { + bool single = f->kind == 10; + double value = TableJsonTokenDouble( token, length, single ); + // A magnitude the field's format cannot hold is the WRONG SHAPE for + // the kind, and it never reaches storage: 1e400 is not a float64 and + // 1e300 is not a float32. Storing the infinity the conversion + // produced would leave an instance this walk called CLEAN that + // ToJsonMeasure then refuses forever (a non-finite float has no JSON + // spelling), and §16.1's one invariant is that a text which reads + // clean writes back. + if ( !TableJsonFinite( value ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( f->has_range ) + { + if ( value < f->range_min ) { value = f->range_min; in.report->clamped++; } + else if ( value > f->range_max ) { value = f->range_max; in.report->clamped++; } + } + if ( single ) + { + float narrow = (float) value; + if ( !TableJsonFinite( (double) narrow ) ) + { + in.report->kind_mismatch++; + return true; + } + memcpy( storage, &narrow, sizeof( narrow ) ); + } + else + { + memcpy( storage, &value, sizeof( value ) ); + } + return true; + } + // JSON HAS ONE NUMBER TYPE. 2.0 IS the integer 2 and 1e3 IS 1000, and a + // library that round-trips numbers through a double emits them that way — + // this walker's own float writer emits 1e+21. So an integer field takes + // any number whose VALUE is integral, however it was spelled; only a + // genuinely fractional value is the wrong shape for it. + bool is_signed = f->kind >= 2 && f->kind <= 5; + bool saturated = false; + int64_t value = 0; + if ( integral ) + { + value = TableJsonTokenInteger( token, length, is_signed, &saturated ); + } + else + { + double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( is_signed ) + { + if ( d >= 9223372036854775808.0 ) { value = INT64_MAX; saturated = true; } + else if ( d < -9223372036854775808.0 ) { value = INT64_MIN; saturated = true; } + else if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) d; } + } + else + { + if ( d < 0.0 ) + { + // a negative for an unsigned field clamps to zero, as the + // exact digit path already does + if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + value = 0; + saturated = true; + } + else if ( d >= 18446744073709551616.0 ) { value = (int64_t) UINT64_MAX; saturated = true; } + else if ( d != (double) (uint64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) (uint64_t) d; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->has_range ) + { + if ( (double) value < f->range_min ) { value = (int64_t) f->range_min; in.report->clamped++; } + else if ( (double) value > f->range_max ) { value = (int64_t) f->range_max; in.report->clamped++; } + } + // the field's own storage width is the last bound: a value past it + // clamps rather than wrapping, which is what the wire does too + if ( f->elem_size < 8 ) + { + if ( is_signed ) + { + int64_t high = ( int64_t( 1 ) << ( f->elem_size * 8 - 1 ) ) - 1; + int64_t low = -high - 1; + if ( value > high ) { value = high; in.report->clamped++; } + else if ( value < low ) { value = low; in.report->clamped++; } + } + else + { + uint64_t high = ( uint64_t( 1 ) << ( f->elem_size * 8 ) ) - 1; + if ( value < 0 ) { value = 0; in.report->clamped++; } + else if ( (uint64_t) value > high ) { value = (int64_t) high; in.report->clamped++; } + } + } + // at eight bytes the storage IS the parser's width, and an unsigned value + // past INT64_MAX rides here as a negative int64 by design — the token + // parser already turned a NEGATIVE token for an unsigned field into a + // clamped zero, so there is nothing left to bound. + TableJsonSetRaw( storage, f->elem_size, (uint64_t) value ); + return true; +} + +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ) +{ + uint8_t * storage = (uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonReadMap( in, (void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + + if ( f->kind == 12 ) + { + int32_t length = 0; + if ( !TableJsonScanString( in, (char *) storage, f->array_bound, &length ) ) { return false; } + storage[length] = 0; + TableJsonSetCount( base, f, length ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + // base64 decodes STRAIGHT INTO the field's storage, six bits at a + // time — no window, no temporary, so a bytes(N) of any declared + // extent reads the same way. A base64 body carries no escapes, so a + // backslash in one is simply not an alphabet character. + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + memset( storage, 0, (size_t) f->array_bound ); + TableJsonSetCount( base, f, 0 ); + const char * alphabet = TableJsonBase64Alphabet(); + int32_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + bool clamped = false; + bool malformed = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + const char * at = c != 0 ? strchr( alphabet, c ) : NULL; + if ( at == NULL ) { malformed = true; continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( at - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < f->array_bound ) + { + storage[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); + } + else + { + clamped = true; + } + } + } + if ( malformed ) + { + // a body that is not base64 is the wrong shape for the kind: the + // field keeps its default and the event is counted + in.report->kind_mismatch++; + return true; + } + if ( clamped ) { in.report->clamped++; } + TableJsonSetCount( base, f, placed ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + // every slot back to its declared defaults first, so a key the text + // omits keeps them and a repeated field key cannot leave an earlier + // occurrence's slots standing + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + void * slot = storage + (int64_t) i * f->elem_size; + if ( f->kind == 13 ) { f->table->reset( slot ); } + else { memset( slot, 0, (size_t) f->elem_size ); } + } + char shape = TableJsonElementShape( f ); + // A KEYED OBJECT'S KEYS ARE KEYS: a variant named twice is a duplicate + // key like any other, last-wins and counted (§16.2). Tracked the way + // a table's own field keys are — a bounded, allocation-free bitmask; + // a vocabulary wider than this still reads, its repeats simply stop + // being counted. + uint64_t seen[8] = {}; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t slot = -1; + for ( int64_t v = 0; v < f->array_bound; v++ ) + { + // nothing is stored for None, so "None" finds no slot and is + // an unknown key like any other name this reader cannot place + if ( !TableJsonKeyedSlotValid( f, v ) ) { continue; } + if ( strcmp( f->key_name( TableJsonKeyedSlotKey( v ) ), key ) == 0 ) { slot = v; break; } + } + if ( slot >= 0 && slot < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( slot & 63 ); + if ( ( seen[slot >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[slot >> 6] |= bit; + } + if ( slot < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; + } + if ( f->is_array ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + // LAST WINS has to be true of a repeated ARRAY key too, and it is + // wire-visible: a fixed array writes every slot, so a second, shorter + // occurrence overlaying a prefix would leave the first occurrence's + // tail standing. The field goes back to its declared defaults before + // this occurrence's elements are placed — the re-establishment a nested + // table and a union arm already get. A table element's defaults are + // its own (the reset hook); every other element kind's storage + // default is zero, which is what the generated array declares. + if ( f->kind == 13 ) + { + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + f->table->reset( storage + (int64_t) i * f->elem_size ); + } + } + else + { + memset( storage, 0, (size_t) f->array_bound * (size_t) f->elem_size ); + } + TableJsonSetCount( base, f, 0 ); + int32_t placed = 0; + char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( placed >= f->array_bound ) + { + // more elements than the reader's bound: the bounded prefix + // is kept and the excess counts, the wire's rule (§4) + in.report->clamped++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( f->kind == 17 ) + { + // an element of an ARRAY OF POINTERS (§2.1): null is a null slot, an + // object is the pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( storage + (int64_t) placed * f->elem_size, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed++; + } + else + { + if ( !TableJsonReadScalar( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + // a fixed array's tail keeps the defaults the prefill left there, + // exactly as a short wire count does + TableJsonSetCount( base, f, placed ); + return true; + } + return TableJsonReadScalar( in, storage, f, depth ); +} + +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ); + +// ONE table object: keys are field keys, unknown ones are skipped and +// counted, a repeated key is last-wins and counted. The instance is already +// at its declared defaults when this is entered, so a key the text never +// mentions keeps the default an absent field takes on the wire (§4). +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + return TableJsonReadTableKeys( in, base, info, depth, NULL ); +} + +// The keys of an object whose brace is already consumed. A pointer's object +// opens the same way a table's does, but its FIRST key may be `&node` (§16.7) +// and the adapter that reads it has to scan the key to know — so it hands the +// key it scanned in as `first_key`, with the colon consumed, and this places +// it before scanning the rest. +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ) +{ + // duplicate tracking, bounded and allocation-free: a table with more + // fields than this still reads, its repeats simply stop being counted + uint64_t seen[8] = {}; + for ( ;; ) + { + char key[kTableJsonMaxKey]; + char c = 0; + if ( first_key != NULL ) + { + memcpy( key, first_key, strlen( first_key ) + 1 ); // scanned into a buffer this size by the caller + first_key = NULL; + } + else + { + c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + } + int32_t index = -1; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + if ( strcmp( info->fields[i].json, key ) == 0 ) { index = i; break; } + } + if ( key[0] == '&' ) + { + // THE AMPERSAND PREFIX IS RESERVED TO THE FORM (docs/SPEC-TABLES.md + // §16.7). No declaration may take a key beginning with it, so this + // is never a field this build lacks — it is the sharing construct + // somewhere it cannot stand: `&node` is the FIRST key of a pointer's + // object and nothing else, and the adapter that reads a pointer + // has consumed it before these keys are read. MALFORMED, refused + // and counted; never counted as unknown, never skipped. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( index < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + const TableFieldInfo * f = &info->fields[index]; + if ( index < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( index & 63 ); + if ( ( seen[index >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[index >> 6] |= bit; + } + // PRESENCE OF THE KEY IS THE PRESENCE (§16.2): reaching this line + // is the key being present, so an optional is set present + // whatever its value — with one exception the page names: a JSON + // null, which reads as ABSENT rather than as a value. + char got = TableJsonValueShape( in ); + if ( f->kind == 17 && !f->is_array ) + { + // a pointer: null is a null pointer, an object is the pointee + // in place or an `&node` reference to one (§16.7), a string is + // a BYTE BUFFER's bytes (§2.5), and anything else is the wrong + // shape for the kind + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) base + f->offset, f->elem_size, 0 ); + } + else if ( got != TableJsonShape( f ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) base + f->offset, f, depth ) ) + { + return false; + } + } + else if ( f->optional && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + // absent, and back at its defaults: a repeated key whose last + // occurrence is null must not leave an earlier value standing + if ( f->table != NULL ) { f->table->reset( (uint8_t *) base + f->offset ); } + else { memset( (uint8_t *) base + f->offset, 0, (size_t) f->elem_size ); } + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 0 ); + } + else + { + if ( got != TableJsonShape( f ) ) + { + // the wrong JSON type for the kind: skipped, never coerced + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, base, f, depth ) ) + { + return false; + } + if ( f->optional ) + { + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 1 ); + } + } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +// ---- the two entry points the per-table wrappers name ---- + +inline bool TableJsonRead( void * value, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = NULL; + info->reset( value ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, value, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +inline int64_t TableJsonWrite( const void * value, const TableTypeInfo * info, char * buffer, int64_t capacity ) +{ + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = NULL; + if ( !TableJsonWriteValue( out, value, info, 0 ) ) { return -1; } + // THE CANONICAL TEXT ENDS WITH EXACTLY ONE NEWLINE (docs/SPEC-TABLES.md + // §16.1). Every writer emits it — this walk, the C# walk and + // "schema unpack" — and every reader accepts a text with or without one, + // because the trailing whitespace a read already skips is what makes the + // two the same text. It is a byte of the FORM rather than a file + // convention: a text that is written to a file, pasted into a diff and + // handed back through a pipe has to be one text in all three places, and a + // buffer whose last byte is a closing brace is the one shape that is not. + out.put( '\n' ); + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json walk: end ---- + +// ---- json graph walk: begin ---- +// +// THE VARIABLE CLASS's half of the text form (docs/SPEC-TABLES.md §16.7). The +// walk above places every kind but one; this defines the three adapters it +// calls for that one, and the two entry points a pointered table's wrappers +// name. The text is the fixed class's — a pointee is an object in place — and a +// node named more than once carries `&node`: defined once, with its fields, +// and referenced after by `{ "&node": N }` alone. + +// ---- the identity map ---- +// +// ONE map shape serves both directions. Writing keys it by a node's ADDRESS and +// counts the slots that name the node, so the second pass knows at a node's +// first occurrence whether it will be named again; reading keys it by the +// text's own label and answers the node it defined. Open addressing, a +// multiply-shift hash and quadrupling growth — TablePackMap's shape (§6.2), on +// the same terms: proportional to nodes, never to bytes, on the authoring +// side, and released before the call returns. + +struct TableJsonGraphEntry +{ + uint64_t key; // a node's address (write) or a label (read); 0 is an empty slot + int64_t count; // write: how many slots name this node + int64_t label; // write: the `&node` label assigned at its first write, 0 until then + uint8_t open; // the descent is still open: a reference here is a cycle (write), a self-reference (read) + uint32_t node; // read: the node's arena offset; 0 for a definition the reader dropped + const TableTypeInfo * type; // read: the node's table; NULL for a dropped one +}; + +struct TableJsonGraphMap +{ + TableJsonGraphEntry * entries; + int64_t capacity; // a power of two, or zero while empty + int64_t count; + TableAllocator allocator; // the caller's pair (§6.5): the builder's on read, the one handed to ToJson on write +}; + +inline void TableJsonGraphMapInit( TableJsonGraphMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TableJsonGraphMapShutdown( TableJsonGraphMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TableJsonGraphMapInit( map, map.allocator ); +} + +inline int64_t TableJsonGraphMapSlot( const TableJsonGraphMap & map, uint64_t key ) +{ + uint64_t hash = key * 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != 0 && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TableJsonGraphEntry * TableJsonGraphMapFind( TableJsonGraphMap & map, uint64_t key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +inline bool TableJsonGraphMapGrow( TableJsonGraphMap & map ) +{ + TableJsonGraphMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 64; + grown.count = 0; + grown.entries = (TableJsonGraphEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TableJsonGraphEntry ) ); // zeroed, by the pair's contract + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == 0 ) { continue; } + grown.entries[ TableJsonGraphMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// the entry for a key, made if it was not there; `taken` says which. NULL is the +// allocator refusing, and the walk refuses with it. +inline TableJsonGraphEntry * TableJsonGraphMapReach( TableJsonGraphMap & map, uint64_t key, bool & taken ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TableJsonGraphMapGrow( map ) ) { return NULL; } + } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + taken = entry->key != key; + if ( taken ) + { + entry->key = key; + map.count++; + } + return entry; +} + +// ---- reading: into a builder ---- + +struct TableJsonGraphIn +{ + TableWorker * worker; // where every node comes from + TableJsonGraphMap labels; // a label -> the node it defined +}; + +// `&node`'s value, the LABEL: a positive integer spelled as one — digits, no sign, no +// fraction, no exponent, no leading zero (§16.7). Anything else is malformed. +inline bool TableJsonScanLabel( TableJsonIn & in, uint64_t & label ) +{ + TableJsonSpace( in ); + if ( in.pos >= in.size || in.text[in.pos] < '1' || in.text[in.pos] > '9' ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + uint64_t value = 0; + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) + { + uint64_t digit = (uint64_t) ( in.text[in.pos] - '0' ); + if ( value > ( UINT64_MAX - digit ) / 10 ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + value = value * 10 + digit; + in.pos++; + } + label = value; + return true; +} + +// A BYTE BUFFER's text (docs/SPEC-TABLES.md §2.5, §16.2): a string. For a +// *string the string's bytes become the blob; for a *bytes the string is base64 +// and its decoded bytes do. The blob is allocated at EXACTLY the decoded +// length — the string is scanned once without keeping it to learn the length, +// and once into the node — so a blob of any size reads with no window and no +// bound to clamp against. A *bytes body that is not base64 is the wrong shape +// for the kind: the reference stays null and the event is counted. +inline bool TableJsonReadBlob( TableJsonIn & in, void * slot, const TableFieldInfo * f ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + TableRef * ref = (TableRef *) slot; + ref->value = 0; + if ( strcmp( f->type_name, "string" ) == 0 ) + { + const int64_t mark = in.pos; + int32_t length = 0; + if ( !TableJsonScanString( in, NULL, 0, &length ) ) { return false; } + in.pos = mark; + char * data = TableStringEmplace( *graph->worker, *ref, NULL, (int64_t) length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int32_t placed = 0; + return TableJsonScanString( in, data, length, &placed ); + } + // base64: the alphabet characters decide the length, six bits apiece + const char * alphabet = TableJsonBase64Alphabet(); + const int64_t mark = in.pos + 1; + int64_t symbols = 0; + bool malformed = false; + in.pos++; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + if ( c == 0 || strchr( alphabet, c ) == NULL ) { malformed = true; continue; } + symbols++; + } + if ( malformed ) + { + in.report->kind_mismatch++; + return true; + } + const int64_t length = ( symbols * 6 ) / 8; + uint8_t * data = TableBytesEmplace( *graph->worker, *ref, length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int64_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + for ( int64_t at = mark; ; at++ ) + { + char c = in.text[at]; + if ( c == '"' ) { break; } + const char * symbol = c != '=' ? strchr( alphabet, c ) : NULL; + if ( symbol == NULL ) { continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( symbol - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < length ) { data[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); } + } + } + return true; +} + +// A pointer's object. Its FIRST key decides what it is: `&node` naming a label not +// yet defined, with fields after it, is a DEFINITION; `&node` naming one already +// defined, alone, is a REFERENCE; any other key is a node named once, its +// object in place. The node comes from the +// builder's arena, and the slot holds its arena offset (§6.3). A pointer whose +// target is a BYTE BUFFER — no table — takes a string instead (§2.5). +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( f->table == NULL ) { return TableJsonReadBlob( in, slot, f ); } + // the pointee nests one level down, exactly as a by-value table does, and + // takes the same cap: a chain nests as deep as it is long (§16.7) + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + char c = TableJsonPeek( in ); + if ( c == '}' ) + { + // an empty object: a node at its defaults, named once + in.pos++; + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + return true; + } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( strcmp( key, "&node" ) != 0 ) + { + // a node named once: the pointee's object in place, and this key is + // its first field — unless it is the reserved prefix under a spelling + // this form does not have, which ReadTableKeys refuses + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return TableJsonReadTableKeys( in, node, f->table, depth + 1, key ); + } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->labels, label, taken ); + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + // ONE SPELLING, and what follows the label says which half it is: fields + // after a label the text has not defined DEFINE it, and a label alone that + // the text has defined REFERS to it. The other two are malformed — a label + // alone that the text never defined, which would otherwise read as a default + // node under a silent report, and a field after a label already defined, + // which would be a second definition. That is what keeps a typo loud. + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + bool bare = c == '}'; + if ( bare == taken ) { in.report->malformed = true; in.bad = true; return false; } + if ( bare ) + { + // A REFERENCE. A label is defined when its object CLOSES, so a + // reference met inside its own definition — at any depth of by-value + // nesting — names a node whose descent is still open: the cycle the + // wire refuses (§3.1), refused here where it is written. A definition + // the reader dropped names no node, so the slot stays null with + // nothing more counted — the drop was counted where it happened. A + // node of another table than the slot declares is a kind mismatch, as + // on the wire. + in.pos++; + if ( entry->open != 0 ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + if ( entry->type == NULL ) + { + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + if ( entry->type != f->table ) + { + memcpy( slot, &ref, sizeof( ref ) ); + in.report->kind_mismatch++; + return true; + } + ref.value = (int64_t) entry->node; + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + // A DEFINITION: the node is allocated, the label is its, and the keys after + // `&node` are its fields. The entry is OPEN until the object closes, so a + // reference to the label from inside the node's own fields is refused as + // the cycle it is; the node and its table are filled in at the close. + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + entry->open = 1; + if ( !TableJsonReadTableKeys( in, node, f->table, depth + 1, NULL ) ) { return false; } + entry = TableJsonGraphMapFind( graph->labels, label ); // the map may have grown under the descent + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + memcpy( &ref, slot, sizeof( ref ) ); + entry->node = (uint32_t) ref.value; + entry->type = f->table; + entry->open = 0; + return true; +} + +// An `&`-prefixed key opening an object the walk is SKIPPING — a value past an +// array's bound, an unknown key's value, a value of the wrong shape. A +// definition in there still takes its label, so the numbering survives whatever +// the storage could not hold (§16.7): the label is registered with no node, and a +// reference to it reads null. Any other prefixed key is the reserved prefix +// out of place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL || strcmp( key, "&node" ) != 0 ) { in.report->malformed = true; in.bad = true; return false; } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + if ( TableJsonGraphMapReach( graph->labels, label, taken ) == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return true; // a fresh entry is node 0, type NULL: a definition with no node +} + +// ---- writing: from a region's const root ---- + +struct TableJsonGraphOut +{ + TableJsonGraphMap nodes; // a node's address -> how many slots name it, and its `&node` once assigned + bool counting; // PASS ONE: count the references, refuse a cycle, emit nothing + int64_t next_label; +}; + +// The node a slot names: null as `null`, a node named once as its object in +// place, and a node named more than once under the construct. Which of the +// last two it is was learned in pass one; pass two spells it. +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphOut * graph = (TableJsonGraphOut *) out.graph; + if ( graph == NULL ) { return false; } + const void * node = f->resolve( slot ); + if ( node == NULL ) + { + out.raw( "null", 4 ); + return true; + } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->nodes, (uint64_t) (uintptr_t) node, taken ); + if ( entry == NULL ) { return false; } + if ( f->table == NULL ) + { + // A BYTE BUFFER (§2.5, §16.7): its text is a string, which has no + // first key to carry `&node`, so a blob named from more than one + // slot has no spelling this form can carry and the graph is refused — + // as a shared node with nothing to write is. A blob named once is its + // bytes in place: base64 for a *bytes, the string itself for a *string. + if ( graph->counting ) { entry->count++; return true; } + if ( entry->count > 1 ) { return false; } + const TableBlob * blob = (const TableBlob *) node; + if ( blob->length > (uint32_t) 0x7fffffff ) { return false; } + if ( strcmp( f->type_name, "string" ) == 0 ) { TableJsonWriteString( out, (const char *) ( blob + 1 ), (int32_t) blob->length ); } + else { TableJsonWriteBase64( out, (const uint8_t *) ( blob + 1 ), (int32_t) blob->length ); } + return true; + } + if ( graph->counting ) + { + // PASS ONE: one visit per node, every slot that names it counted, and + // a reference to a node whose descent is still open is a cycle — + // refused here as the wire refuses it (§3.1) + entry->count++; + if ( !taken ) { return entry->open == 0; } + entry->open = 1; + if ( !TableJsonWriteValue( out, node, f->table, depth ) ) { return false; } + entry = TableJsonGraphMapFind( graph->nodes, (uint64_t) (uintptr_t) node ); // the map may have grown under the descent + if ( entry == NULL ) { return false; } + entry->open = 0; + return true; + } + // PASS TWO: a node named once is its object in place; a node named more + // than once is DEFINED at its first occurrence — `&node` first, then its + // fields — and REFERENCED by `&node` alone after that, spelled the same way at + // every site. Labels run from 1 in first-write order and are the text's own, + // so a stray number in a hand-edited text is most often one never defined. + if ( entry->count <= 1 ) + { + return TableJsonWriteValue( out, node, f->table, depth ); + } + if ( depth > kTableJsonMaxDepth ) { return false; } + if ( entry->label != 0 ) + { + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + out.line( depth ); + out.put( '}' ); + return true; + } + entry->label = ++graph->next_label; + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + bool any = true; + int64_t before = out.offset; + if ( !TableJsonWriteFields( out, node, f->table, depth, any ) ) { return false; } + // a definition carries at least one field, because a label alone is a + // reference: a shared node with nothing to write has no definition this + // form can spell, and the writer refuses it as it refuses any value it + // cannot spell (§16.3) + if ( out.offset == before ) { return false; } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- the two entry points a pointered table's wrappers name ---- + +// The text into the builder's root. Every node the text names is allocated in +// the builder's arena through the field's own Emplace; the label map is the +// walk's, released before this returns. The root itself takes no label — nothing +// may name it (§16.7) — so an `&node` at the root is refused like any other key +// of the prefix. +inline bool TableJsonReadGraph( TableWorker & worker, void * root, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + if ( worker.arena == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } + TableJsonGraphIn graph; + graph.worker = &worker; + TableJsonGraphMapInit( graph.labels, worker.arena->allocator ); + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = &graph; + info->reset( root ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, root, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + TableJsonGraphMapShutdown( graph.labels ); + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +// The text of a region's const root: measured when the buffer is NULL, written +// when it is not, over one code path. Two passes over one walk — the first +// counts how many slots name each node and refuses a cycle, the second writes +// — so a node's first occurrence knows whether it will be named again. The +// ROOT's entry is open for the whole first pass, so a reference back at it is +// the cycle it is (§3.1), and it takes no label. +inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * info, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + if ( root == NULL ) { return -1; } + TableJsonGraphOut graph; + TableJsonGraphMapInit( graph.nodes, allocator ); + graph.counting = true; + graph.next_label = 0; + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph.nodes, (uint64_t) (uintptr_t) root, taken ); + if ( entry == NULL ) { TableJsonGraphMapShutdown( graph.nodes ); return -1; } + entry->open = 1; + TableJsonOut count; + count.buffer = NULL; + count.capacity = 0; + count.offset = 0; + count.overflow = false; + count.graph = &graph; + bool ok = TableJsonWriteValue( count, root, info, 0 ); + graph.counting = false; + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = &graph; + if ( ok ) { ok = TableJsonWriteValue( out, root, info, 0 ); } + TableJsonGraphMapShutdown( graph.nodes ); + if ( !ok ) { return -1; } + out.put( '\n' ); // the canonical text ends with exactly one newline (§16.1) + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json graph walk: end ---- + +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + +// ---- json map walk: begin ---- + +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} + +// the entry's two rows: fields[0] IS the key and fields[1] IS the value, which +// is what makes a user's own table of pairs the same bytes (§2.8) +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } + +inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } +inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } + +// AN INTEGER KEY IS THE INTEGER'S DECIMAL SPELLING, QUOTED, because a JSON +// object's keys are strings. Written digit by digit so no locale can move it. +inline void TableJsonWriteMapIntegerKey( TableJsonOut & out, const void * storage, const TableFieldInfo * key ) +{ + uint64_t magnitude = 0; + bool negative = false; + if ( TableJsonMapKeySigned( key ) ) + { + int64_t value = 0; + switch ( key->kind ) + { + case 2: value = (int64_t) *(const int8_t *) storage; break; + case 3: value = (int64_t) *(const int16_t *) storage; break; + case 4: value = (int64_t) *(const int32_t *) storage; break; + default: value = *(const int64_t *) storage; break; + } + negative = value < 0; + magnitude = negative ? ( ~(uint64_t) value ) + 1 : (uint64_t) value; + } + else + { + switch ( key->kind ) + { + case 6: magnitude = (uint64_t) *(const uint8_t *) storage; break; + case 7: magnitude = (uint64_t) *(const uint16_t *) storage; break; + case 8: magnitude = (uint64_t) *(const uint32_t *) storage; break; + default: magnitude = *(const uint64_t *) storage; break; + } + } + char digits[24]; + int32_t at = (int32_t) sizeof( digits ); + do { digits[--at] = (char) ( '0' + ( magnitude % 10 ) ); magnitude /= 10; } while ( magnitude != 0 ); + if ( negative ) { digits[--at] = '-'; } + TableJsonWriteString( out, digits + at, (int32_t) sizeof( digits ) - at ); +} + +inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const TableFieldInfo * key ) +{ + const uint8_t * storage = (const uint8_t *) entry + key->offset; + if ( TableJsonMapKeyIsString( key ) ) + { + // A STRING KEY IS THE STRING (§2.8): every JSON key of a map object is + // a KEY OF THE MAP and none is a field key, so the `&` prefix §16.7 + // reserves for field keys is ordinary data here. + TableJsonWriteString( out, (const char *) storage, *(const int32_t *) ( (const uint8_t *) entry + key->count_offset ) ); + return; + } + TableJsonWriteMapIntegerKey( out, (const void *) storage, key ); +} + +// ToJson WRITES ENTRIES IN ASCENDING KEY ORDER, so unpack then pack is +// byte-stable and a diff of two texts is a diff of two maps (§2.8, §17.2). +// A region holds them in that order already, so this is the array in place. +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "{}", 2 ); return true; } + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); + out.put( '{' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); + TableJsonWriteMapKey( out, entry, key ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// AN INTEGER KEY IS READ BY §16.2's INTEGER RULE AND BY NOTHING ELSE, so +// "2.0" and "1e3" are the integers 2 and 1000 and "-0" is zero. The token is +// walked as a JSON number over its own bytes; a token that rule calls +// malformed makes the KEY malformed, and a genuinely fractional value, or one +// outside the key kind's range, is kind_mismatch for that entry. +inline bool TableJsonMapKeyValue( const char * token, int32_t length, const TableFieldInfo * key, + int64_t & value, bool & fits ) +{ + fits = false; + TableReport scratch; + TableJsonIn probe = { token, (int64_t) length, 0, &scratch, false, NULL }; + bool integral = false; + if ( !TableJsonWalkNumber( probe, &integral ) ) { return false; } + if ( probe.pos != (int64_t) length ) { return false; } // trailing bytes: not a number + if ( !integral ) + { + const double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) { return true; } // a value no key kind holds + const double whole = d < 0 ? -d : d; + if ( whole != (double) (int64_t) whole ) { return true; } // genuinely fractional + } + bool saturated = false; + const bool is_signed = TableJsonMapKeySigned( key ); + value = integral ? TableJsonTokenInteger( token, length, is_signed, &saturated ) + : (int64_t) TableJsonTokenDouble( token, length, false ); + if ( saturated ) { return true; } // outside every width: kind_mismatch, never clamped + switch ( key->kind ) + { + case 2: fits = value >= -128 && value <= 127; break; + case 3: fits = value >= -32768 && value <= 32767; break; + case 4: fits = value >= -2147483647 - 1 && value <= 2147483647; break; + case 5: fits = true; break; + case 6: fits = value >= 0 && value <= 255; break; + case 7: fits = value >= 0 && value <= 65535; break; + case 8: fits = value >= 0 && (uint64_t) value <= 4294967295ull; break; + default: fits = integral; break; // uint64: the token's own magnitude + } + return true; +} + +// FromJson READS KEYS IN WHATEVER ORDER THE TEXT GIVES THEM. A repeated key is +// last-wins and counted duplicate, the object rule (§16.2) applied inside the +// map. An empty object is an empty map, and null is kind_mismatch. +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const char shape = TableJsonShape( value ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char token[kTableJsonMaxKey]; + int32_t token_length = 0; + if ( !TableJsonScanString( in, token, kTableJsonMaxKey - 1, &token_length ) ) { return false; } + token[token_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t key_value = 0; + bool place = true; + if ( !TableJsonMapKeyIsString( key ) ) + { + bool fits = false; + if ( !TableJsonMapKeyValue( token, token_length, key, key_value, fits ) ) + { + // A MALFORMED KEY STOPS THE READ where §16.1's rule stops it, + // with the instance holding what was placed before the stop. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( !fits ) { in.report->kind_mismatch++; place = false; } + } + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; + if ( place && entry == NULL ) + { + // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the + // wire's rule, because a clamped key is a merged entry (§2.8). + in.report->clamped++; + } + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) + { + in.report->duplicate++; // last-wins, the object rule inside the map + } + const char got = TableJsonValueShape( in ); + if ( entry == NULL ) + { + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( value->kind == 17 && !value->is_array ) + { + // A POINTER VALUE IS SHARED EXACTLY AS A POINTER FIELD IS (§2.8): + // null is a null slot, an object is the pointee in place or an + // &node reference to one (§16.7), anything else is the wrong shape — + // the same three the field-key loop gives a pointer field, because + // an entry's value IS a field line. + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) entry + value->offset, value->elem_size, 0 ); + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) entry + value->offset, value, depth + 1 ) ) + { + return false; + } + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, entry, value, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json map walk: end ---- + +// ---- json list walk: begin ---- + +// an unbounded array is the out-of-line array that is not a map (§8.1) +inline bool TableJsonIsList( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && !TableJsonIsMap( f ); +} + +// ToJson WRITES THE ELEMENTS IN INDEX ORDER, which is the only order there is, +// so unpack then pack is byte-stable without a rule of its own (§2.9, §17.2). +// A region holds the array in place, so this steps it at the descriptor's pitch. +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + const uint8_t * elements = TableJsonExtentElements( slot ); + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const uint8_t * element = elements + (int64_t) i * f->elem_size; + if ( f->kind == 17 ) + { + // a []*T's elements take the pointer row (§16.7): the pointee's + // object in place, null, or `&node` for a shared one + if ( !TableJsonWritePointer( out, element, f, depth + 1 ) ) { return false; } + } + else if ( !TableJsonWriteScalar( out, element, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; +} + +// FromJson READS EVERY ELEMENT THE TEXT CARRIES, appending each through the +// descriptor's place resolver: `[]` is an empty list, and null is +// kind_mismatch, the array row's own rule (§16.2). LAST WINS holds for a +// repeated key: the list goes back to EMPTY before this occurrence's elements +// land, the builder's storage being reclaimed at reset (§2.9). +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) slot, 8, 0 ); + TableJsonSetRaw( (uint8_t *) slot + 8, 4, 0 ); + const char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + void * element = f->place( *graph->worker, slot, NULL, 0, 0 ); + if ( element == NULL ) + { + // NOT ADDED: the arena could not carve another segment, or the + // count met the int32 cap. The text cannot be placed whole, and + // the read stops where §16.1's rule stops it. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( f->kind == 17 ) + { + // an element of a []*T (§2.9): null is a null slot, an object is the + // pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) element, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, element, f, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + // the wrong shape for the element kind: the slot keeps its + // defaults and the event counts, the array row's rule (§16.2) + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, element, f, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json list walk: end ---- + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +bool BytesFromJson( BytesBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Bytes * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, BytesTableType(), text, bytes, report ); +} + +int64_t BytesToJsonMeasure( const Bytes * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, BytesTableType(), NULL, 0, allocator ); +} + +int64_t BytesToJson( const Bytes * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, BytesTableType(), buffer, capacity, allocator ); +} + +bool IntsFromJson( IntsBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Ints * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, IntsTableType(), text, bytes, report ); +} + +int64_t IntsToJsonMeasure( const Ints * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, IntsTableType(), NULL, 0, allocator ); +} + +int64_t IntsToJson( const Ints * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, IntsTableType(), buffer, capacity, allocator ); +} + +bool FloatsFromJson( FloatsBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Floats * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, FloatsTableType(), text, bytes, report ); +} + +int64_t FloatsToJsonMeasure( const Floats * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, FloatsTableType(), NULL, 0, allocator ); +} + +int64_t FloatsToJson( const Floats * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, FloatsTableType(), buffer, capacity, allocator ); +} + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/ReportTable.h b/testdata/golden/tables/lists/ReportTable.h new file mode 100644 index 000000000..b8b99f464 --- /dev/null +++ b/testdata/golden/tables/lists/ReportTable.h @@ -0,0 +1,7562 @@ +// Code generated by the schema compiler from Report.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — protocol id 0xa5fbe602c119cdd9 (packets only: tables version by field id, not by protocol id) +// The TABLE wire (evolution-tolerant, docs/SPEC-TABLES.md): no serialize +// dependency — includable from any TU. + +#pragma once + +#include +#include // the prefill's scalar-array fills +#include // offsetof, for the reflection descriptors + +// ---- the hooks (docs/USAGE.md, "the C++ table runtime's hooks") ---- +// +// schema_assert — the runtime's own assert, and the refusal a debugger reads. +// NDEBUG removes it, exactly as it removes assert. A caller who already routes +// serialize's asserts writes `#define schema_assert serialize_assert` before +// including this header and both halves land in one handler. +#ifndef schema_assert +#include +#define schema_assert assert +#endif // #ifndef schema_assert + +// schema_fatal — what stands after the assert on a path that cannot continue. +// NDEBUG does not remove it. Supply it and is never included. +#ifndef schema_fatal +#include // abort +#define schema_fatal abort +#endif // #ifndef schema_fatal + +// schema_allocate / schema_release — what "no allocator handed in" means for +// this program. schema_allocate hands back ZEROED bytes and NULL on failure: +// an arena segment is copied whole, padding included, so anything left +// uninitialized here would reach a packed region. Supply both and +// is never included; hand a TableAllocator to a builder to route one +// structure's allocations somewhere else again. +#ifndef schema_allocate +#include // calloc, free +#define schema_allocate( bytes ) calloc( (size_t) 1, (size_t) ( bytes ) ) +#define schema_release( pointer ) free( pointer ) +#endif // #ifndef schema_allocate +#include // a node's lifetime starts in arena storage (placement new) +#include // one atomic per slab: the arena is lock-free by ownership + +#include "Report.h" + +#ifndef LISTDEMO_SCHEMA_TABLE_PRIMITIVES +#define LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +// THE CODEC DOES NOT DEPEND ON THE COMPILER'S INLINING BUDGET. A table of a +// realistic field count emits one large body per type, and the cursor a body +// writes through lives in the caller's `TableWriter`: across a call boundary +// that cursor round-trips through memory, and a `uint8_t *` store may alias the +// writer itself, so every put reloads it. When a budget runs out mid-body the +// codec silently degrades to that shape. Forcing the primitives and the +// fixed-class bodies inline is what keeps the cursor in registers and lets +// adjacent constant framing bytes merge into one store. +#if defined( _MSC_VER ) +#define LISTDEMO_TABLE_INLINE __forceinline +#elif defined( __GNUC__ ) || defined( __clang__ ) +#define LISTDEMO_TABLE_INLINE inline __attribute__(( always_inline )) +#else +#define LISTDEMO_TABLE_INLINE inline +#endif + +namespace listdemo { + +// WHY A READ WAS REFUSED, by name (docs/SPEC-TABLES.md §3.3, §11). A REFUSAL +// is not one of §4's events: nothing is decoded, no counter moves and no +// damage is reported, so five zero counters and a false flag are what a clean +// read prints too and only the verdict tells them apart. The reason says which +// refusal it was. +// +// This is the MESSAGE PATH's vocabulary and not the cooked form's (§7.4): a +// caller meeting one of these has been refused a MESSAGE on a connection, +// which is a different recovery with a different owner than a file a header +// match turned down. +enum TableMessageReason +{ + newer_form, // a FORM BYTE this reader does not carry (§3) + no_vocabulary, // no table for this connection: the message arrived before the announcement, or after a refused one + second_announcement, // a second announcement on a connection: it sets nothing, amends nothing, and the connection closes + vocabulary_too_large, // an announcement above the receiver's declared bound, refused before an entry is touched + message_form_as_file // a form 2 wire where a FILE was expected: its table is somewhere else +}; + +// The table-wire read report — the permissive contract's ledger. Silence +// (all zero) means the data matched this reader's schema exactly. +struct TableReport +{ + int32_t unknown = 0; // unknown field ids skipped (newer data) + int32_t kind_mismatch = 0; // known id, changed type — skipped, never misdecoded + int32_t clamped = 0; // out-of-range values clamped to declared bounds + // a key the TEXT form saw twice: last wins, and the repeat is counted + // (docs/SPEC-TABLES.md §16.2). The wire never raises it — a body carrying an + // id twice is legal input whose last occurrence wins, silently (§3). + int32_t duplicate = 0; + bool malformed = false; // framing damage; decode stopped, partial result kept + // THE REFUSAL VERDICT, which is not one of §4's events and moves no counter + // (docs/SPEC-TABLES.md §3): a FORM BYTE this reader does not carry. Five + // zero counters and a false flag are what a clean read prints too, so the + // verdict is what tells the two apart. + bool refused = false; + // WHICH refusal, and it is read only when refused is set: a read that + // was not refused has no reason, and this member is the one the caller + // must not look at then (docs/SPEC-TABLES.md §3.3). + TableMessageReason reason = newer_form; +}; + + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; +// ---- reflection (tables only, docs/SPEC-TABLES.md) ---- +// +// Static field descriptors for every type in the table closure: name, wire +// id/kind, storage offset, bounds, ranges, enum names and branch guards — +// enough to walk, print, diff, edit or bind any table value at runtime with +// no RTTI and no schema files. TableType() returns X's descriptor. + +struct TableTypeInfo; + +// One arm of a union field: where its payload sits inside the union's storage +// and what its payload looks like. The arm's NAME and its table-wire id come +// from the field's enum_name/variant_id functions at the same tag, so nothing +// is spelled twice (docs/SPEC-TABLES.md §8). +struct TableFieldInfo; + +struct TableUnionArmInfo +{ + uint32_t offset; // offsetof the arm's payload within the union storage + const TableTypeInfo * table; // the arm payload's descriptor, or NULL + // AN ARM IS A FIELD LINE (docs/SPEC-TABLES.md §2.6): an arm that names no + // declared type or table carries the FIELD descriptor a field of that + // type would carry instead — offsets taken within the union storage — so + // a generic walk meets an arm's kind, width, bounds and companions where + // it meets a field's. Exactly one of the two is non-NULL on a set arm. + const TableFieldInfo * field; + uint32_t size; // the arm's whole storage, which selection zero-establishes +}; + +// A union field's shape: the tag, and the arms indexed by it. Arms run +// [0, enum_max]; index 0 is the EMPTY arm and carries no payload. +struct TableUnionInfo +{ + uint32_t tag_offset; // offsetof the tag within the union storage + uint32_t tag_size; // sizeof the tag + const TableUnionArmInfo * arms; +}; + +// The exact raw range of a wide-kind field (docs/SPEC-TABLES.md §8.2): two 128-bit +// values as 64-bit lanes, low lane first, two's complement for the signed kinds. +struct TableWideRange +{ + uint64_t lo[2]; + uint64_t hi[2]; +}; + +// the arena's allocation front, defined with the variable-length runtime +// below; a descriptor names it only through a pointer parameter. +struct TableWorker; + +struct TableFieldInfo +{ + const char * name; // schema field name, e.g. "health" + const char * json; // the TEXT form's key: the json = "key" attribute, else name (§16.3) + const char * type_name; // schema type name, e.g. "float32", "Grade" + uint64_t id; // table-wire field id: fnv1a64 of the name, of the was alias after a rename (§5) + uint8_t kind; // table-wire kind; for arrays/strings/bytes, the ELEMENT kind + bool is_array; // fixed or counted array (bytes included) + bool is_pointer; // a *T pointer field: storage is an 8-byte TableRef; the target is a table + // THE TWO THE TEXT FORM NEEDS (docs/SPEC-TABLES.md §16.7), and they + // are here for the same reason is_pointer is: the walk is ONE walk + // over descriptors and cannot spell a target's own At or + // Emplace. `resolve` reads a slot in a REGION and answers the + // node it names, or NULL; `emplace` allocates one in a BUILDER's + // arena and points the slot at it. NULL on every field that is not + // a pointer, and emitted only in a unit that declares one. + const void * (*resolve)( const void * slot ); + void * (*emplace)( TableWorker & worker, void * slot ); + bool counted; // a _count/_length int32 companion exists (counted arrays, strings, bytes) + bool optional; // a ?T field: a _present bool companion decides whether it rides + int32_t array_bound; // array capacity / string max length; 0 for plain scalars + uint32_t offset; // offsetof the storage member + uint32_t elem_size; // sizeof the member (element size for arrays) + uint32_t count_offset; // offsetof the _count/_length companion, or 0xffffffff + uint32_t present_offset; // offsetof the _present companion, or 0xffffffff + const TableTypeInfo * table; // nested table's descriptor, or NULL + bool has_range; // a declared [min, max] (int or float) + double range_min; // NOTE: int64 ranges beyond 2^53 lose precision here + double range_max; + // the WIDE kinds (18-29, docs/SPEC-TABLES.md §3, §8.2): frac_bits is a fixed + // field's F — its storage holds units × 2^F — and wide is the declared + // range on that RAW scale, exact, as two 128-bit two's-complement values + // in 64-bit lanes (low lane first). NULL where the declaration bounds + // nothing (a bare uint128) and for every other kind; frac_bits is 0 for + // every kind that is not fixed-point. range_min/range_max still carry + // the declared bounds as doubles — whole units for a fixed field — for + // a walker that only shows them. + uint8_t frac_bits; + const TableWideRange * wide; + int64_t enum_max; // enums: highest valid value (None = 0 always valid); + // unions: the arm count (tag range [0, enum_max]); + // flags: the highest declared BIT INDEX; else -1 + // the vocabulary's names, indexed the same way enum_max bounds: an enum's + // value -> name, a union's tag -> arm name, a FLAGS field's bit index -> + // variant name. NULL for every other kind. + const char * (*enum_name)( uint64_t value ); + // the TABLE-WIRE id of one variant (docs/SPEC-TABLES.md §5): for an enum, the + // hash of the variant's name; for a union, the hash of the arm's name. + // 0 is the reserved id — an enum's None, a union's empty. NULL for every + // other kind — a FLAGS field's variants have no per-variant wire id (§4), + // so a NULL here beside a non-NULL enum_name is what says "flags". + // Walk [0, enum_max] to enumerate a vocabulary and its ids. + uint64_t (*variant_id)( uint64_t value ); + // an ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): the array has one slot per + // variant of key_type_name, indexed by the variant's value, and its slots + // ride under variant ids rather than positions. key_name and key_id are + // the key's vocabulary — walk [0, array_bound) to print slots by name. + // NULL on every other field. + const char * key_type_name; + const char * (*key_name)( uint64_t value ); + uint64_t (*key_id)( uint64_t value ); + // union fields: the tag and its arms, behind a function so the whole + // descriptor stays CONSTANT-INITIALISED (a captureless lambda converts to + // a function pointer at compile time; the arms themselves are a static + // inside it). NULL for every other kind. + const TableUnionInfo * (*arms)(); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded +}; + +struct TableTypeInfo +{ + const char * name; // schema type name + uint32_t size; // sizeof the storage struct + int32_t num_fields; + const TableFieldInfo * fields; + // put one instance back at its declared defaults, in place. A generic + // walker that fills a value has to be able to establish the defaults an + // absent field takes, and it holds no type to spell — this is the one + // thing the descriptors could not express without it. Placement-new + // value-init, exactly what the wire's read path does, and no temporary. + void (*reset)( void * storage ); + // the DERIVED mode (docs/SPEC-TABLES.md): false = fixed-size, a plain + // relocatable struct; true = variable-length, built through a Builder + // and read through a region root. Nobody declares it; the compiler + // works it out. + bool variable; +}; + +struct TableWriter +{ + uint8_t * buffer; + int64_t capacity; + int64_t offset = 0; + bool overflow = false; + + // the parameters do not repeat the member names: a parameter that hides a + // member is a warning the estate's compilers disagree about (gcc's + // -Wshadow and cl's C4458 refuse it, clang's -Wshadow does not), and this + // is a header a consumer compiles under its OWN flags + TableWriter( uint8_t * to_buffer, int64_t to_capacity ) : buffer( to_buffer ), capacity( to_capacity ) {} + + LISTDEMO_TABLE_INLINE void raw( const void * data, int64_t bytes ) + { + if ( offset + bytes > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) bytes ); + offset += bytes; + } + LISTDEMO_TABLE_INLINE void put8( uint8_t v ) { raw( &v, 1 ); } + LISTDEMO_TABLE_INLINE void put16( uint16_t v ) { uint8_t b[2] = { uint8_t( v ), uint8_t( v >> 8 ) }; raw( b, 2 ); } + LISTDEMO_TABLE_INLINE void put32( uint32_t v ) { uint8_t b[4] = { uint8_t( v ), uint8_t( v >> 8 ), uint8_t( v >> 16 ), uint8_t( v >> 24 ) }; raw( b, 4 ); } + LISTDEMO_TABLE_INLINE void put64( uint64_t v ) { put32( uint32_t( v ) ); put32( uint32_t( v >> 32 ) ); } + // a 128-bit value as two lanes, the low half first (docs/SPEC-TABLES.md §3) + LISTDEMO_TABLE_INLINE void put128( uint64_t lo, uint64_t hi ) { put64( lo ); put64( hi ); } + // EVERY LENGTH, COUNT, INDEX AND ID REFERENCE IS ONE CANONICAL UNSIGNED + // LEB128 (docs/SPEC-TABLES.md §3): seven value bits a byte, the lowest + // group first, the high bit set on every byte but the last. One value has + // one spelling, so two conforming writers agree byte for byte. + LISTDEMO_TABLE_INLINE void putleb( uint64_t v ) + { + while ( v >= 0x80 ) { put8( uint8_t( v ) | 0x80 ); v >>= 7; } + put8( uint8_t( v ) ); + } +}; + +// TableLebBytes is one value's spelling length, which a MEASURE needs before +// the bytes exist — the length of a body has to be known before it is written, +// because a length whose own width moves cannot be patched in place. +inline int64_t TableLebBytes( uint64_t v ) +{ + int64_t n = 1; + while ( v >= 0x80 ) { v >>= 7; n++; } + return n; +} + +// THE ID TABLE, WRITER SIDE (docs/SPEC-TABLES.md §3). It holds every id the +// body used, once each, in FIRST-USE order over the whole wire, and the body +// names them by position: reference k is the kth entry, counted from 1, and +// reference 0 names NO ID. +// +// Its capacity is a COMPILE-TIME fact of the unit — the distinct names its +// table closure can spell — so a save allocates nothing: the table is a local +// of Measure and of Save. The bucket chain makes ref constant time and makes +// truncate constant time too, which is what an ELIDED field needs: a field +// that turns out not to ride costs nothing in the id table either, so the walk +// interns its id, builds the payload that decides, and undoes the entry when +// nothing rides. +struct TableIds +{ + static const int32_t kCapacity = 60; + static const int32_t kBuckets = 128; + + uint64_t ids[ kCapacity ]; + int32_t chain[ kCapacity ]; + int32_t head[ kBuckets ]; + int32_t count; + bool overflow; + // THE MESSAGE FORM'S SLOTS (docs/SPEC-TABLES.md §3.3). A form 2 wire + // names ids through the CONNECTION's table, which is the unit's whole + // vocabulary in a compiler-settled order — so every reference is known at + // compile time and rides at the header as a literal beside the id. This + // flag is what selects it: false interns the id in first-use order and + // writes a trailer, true answers the slot and writes none, and the walk + // that decides is one walk. + bool vocabulary; + + TableIds() : count( 0 ), overflow( false ), vocabulary( false ) + { + for ( int32_t i = 0; i < kBuckets; i++ ) { head[i] = -1; } + } + + static LISTDEMO_TABLE_INLINE uint32_t bucket_of( uint64_t id ) + { + return uint32_t( ( id * 0x9E3779B97F4A7C15ull ) >> 57 ) & uint32_t( kBuckets - 1 ); + } + + // the reference an id takes: its message-form SLOT under the connection's + // table, or the file's own first-use entry + LISTDEMO_TABLE_INLINE uint64_t ref( uint64_t id, uint64_t slot ) + { + if ( vocabulary ) { return slot; } + return intern( id ); + } + + // the FILE form's half, appending the id on first use + uint64_t intern( uint64_t id ) + { + const uint32_t b = bucket_of( id ); + for ( int32_t i = head[b]; i >= 0; i = chain[i] ) + { + if ( ids[i] == id ) { return uint64_t( i ) + 1; } + } + if ( count >= kCapacity ) { overflow = true; return 1; } + ids[count] = id; chain[count] = head[b]; head[b] = count; count++; + return uint64_t( count ); + } + + // undo every entry appended since mark. An entry removed is the most + // recent one in its bucket, so it sits at that bucket's head. + void truncate( int32_t mark ) + { + // a SLOT costs no entry, so an elided field has nothing to undo + if ( vocabulary ) { return; } + while ( count > mark ) + { + count--; + head[ bucket_of( ids[count] ) ] = chain[count]; + } + } +}; + +// TableIdsBytes is the trailer's own size: the entries, each a fixed +// little-endian u64, and the ENTRY COUNT, the one fixed-width number on the +// wire (docs/SPEC-TABLES.md §3). +inline int64_t TableIdsBytes( const TableIds & ids ) { return int64_t( ids.count ) * 8 + 8; } + +// TableIdsWrite puts the trailer where the walk ended: a writer never patches, +// because first-use order is known only when the walk ends. +inline void TableIdsWrite( TableWriter & w, const TableIds & ids ) +{ + for ( int32_t i = 0; i < ids.count; i++ ) { w.put64( ids.ids[i] ); } + w.put64( uint64_t( ids.count ) ); +} + +// THE ID TABLE, READER SIDE (docs/SPEC-TABLES.md §3). A reader locates it from +// the END of the wire and resolves it ONCE, at open: the entries are eight +// bytes each and a body names them by position, so every field dispatches +// through an index rather than through a search over hashes. +struct TableIdTable +{ + const uint8_t * entries = NULL; + int64_t count = 0; + + // the id a reference names. ref is 1-based and bounds-checked by the + // caller: a reference ABOVE the entry count is framing damage on the body + // that carries it, and 0 names no id at all. + uint64_t at( uint64_t ref ) const + { + const uint8_t * e = entries + ( ref - 1 ) * 8; + uint64_t lo = uint64_t( e[0] ) | uint64_t( e[1] ) << 8 | uint64_t( e[2] ) << 16 | uint64_t( e[3] ) << 24; + uint64_t hi = uint64_t( e[4] ) | uint64_t( e[5] ) << 8 | uint64_t( e[6] ) << 16 | uint64_t( e[7] ) << 24; + return lo | ( hi << 32 ); + } +}; + +struct TableReader +{ + const uint8_t * buffer; + int64_t size; + int64_t offset = 0; + TableReport * report; + const TableIdTable * ids = NULL; + // ONLY THE ROOT BODY CARRIES THE NODE TABLE (docs/SPEC-TABLES.md §3.1), so + // a body has to know which it is: the reserved id inside a NESTED body is + // malformed, because a second numbering cannot exist. Every reader made + // for a payload is nested; the two the wire surfaces make for a root say so. + bool nested = true; + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report ) + : buffer( from_buffer ), size( from_size ), report( to_report ) {} + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report, const TableIdTable * to_ids ) + : buffer( from_buffer ), size( from_size ), report( to_report ), ids( to_ids ) {} + + LISTDEMO_TABLE_INLINE bool has( int64_t bytes ) const { return offset + bytes <= size; } + // A LENGTH IS A 64-BIT NUMBER AND A BUFFER IS NOT (docs/SPEC-TABLES.md + // §3): every length, count and index on this wire has sixty-four bits of + // capability, so one past what remains must be compared UNSIGNED. Casting + // it to int64 first turns 0xFFFFFFFFFFFFFFFF into -1, and a negative + // length looks like room. + LISTDEMO_TABLE_INLINE bool room( uint64_t bytes ) const { return bytes <= (uint64_t) ( size - offset ); } + LISTDEMO_TABLE_INLINE uint8_t get8() { return buffer[offset++]; } + LISTDEMO_TABLE_INLINE uint16_t get16() { uint16_t v = uint16_t( buffer[offset] ) | uint16_t( buffer[offset+1] ) << 8; offset += 2; return v; } + LISTDEMO_TABLE_INLINE uint32_t get32() { uint32_t v = uint32_t( buffer[offset] ) | uint32_t( buffer[offset+1] ) << 8 | uint32_t( buffer[offset+2] ) << 16 | uint32_t( buffer[offset+3] ) << 24; offset += 4; return v; } + LISTDEMO_TABLE_INLINE uint64_t get64() { uint64_t lo = get32(); uint64_t hi = get32(); return lo | ( hi << 32 ); } + LISTDEMO_TABLE_INLINE void get128( uint64_t & lo, uint64_t & hi ) { lo = get64(); hi = get64(); } + + // ONE CANONICAL UNSIGNED LEB128 (docs/SPEC-TABLES.md §3), and a + // non-minimal spelling is MALFORMED: 0x80 0x00 and 0x00 both spell zero, + // and only the second is legal input. An encoding past ten bytes, or a + // tenth byte with a bit above the 64th value bit, is malformed on the same + // rule. false = framing damage on the body carrying it. + bool getleb( uint64_t & value ) + { + // A NUMBER THIS READER REFUSES LEAVES THE CURSOR WHERE IT WAS. The + // caller's next question is often "did this body end exactly at its + // L", and a rejected number that had moved the cursor would answer + // that question with the damage already stepped over. + const int64_t at = offset; + value = 0; + uint32_t shift = 0; + for ( int32_t i = 0; i < 10; i++ ) + { + if ( !has( 1 ) ) { offset = at; return false; } + const uint8_t b = get8(); + if ( i == 9 && b > 1 ) { offset = at; return false; } + value |= uint64_t( b & 0x7F ) << shift; + if ( ( b & 0x80 ) == 0 ) + { + if ( i > 0 && b == 0 ) { offset = at; return false; } // a redundant continuation + return true; + } + shift += 7; + } + offset = at; + return false; + } + + // resolve one id reference against the file's table. false = a reference + // ABOVE the entry count, or a 0 where an id is required, both of which + // are framing damage on the body that carries it. + bool getid( uint64_t & id ) + { + uint64_t ref = 0; + if ( !getleb( ref ) ) { return false; } + if ( ref == 0 || ids == NULL || ref > (uint64_t) ids->count ) { return false; } + id = ids->at( ref ); + return true; + } + + // skip one payload by kind; false = framing damage. FOUR RULES COVER THE + // SET (docs/SPEC-TABLES.md §3), and a kind outside it is not skippable — + // which is why the set is closed and why kind 31 exists. + bool skip( uint8_t kind ) + { + switch ( kind ) + { + // the fixed-width kinds, each by its width: 18-29 are the 128-bit integers and + // the fixed-point family at every storage width (docs/SPEC-TABLES.md §3) + case 1: case 2: case 6: case 20: case 25: return has( 1 ) ? ( offset += 1, true ) : false; + case 3: case 7: case 21: case 26: return has( 2 ) ? ( offset += 2, true ) : false; + case 4: case 8: case 10: case 22: case 27: return has( 4 ) ? ( offset += 4, true ) : false; + case 5: case 9: case 11: case 23: case 28: return has( 8 ) ? ( offset += 8, true ) : false; + case 18: case 19: case 24: case 29: return has( 16 ) ? ( offset += 16, true ) : false; + case 17: case 30: // a NODE INDEX (§3.1) and an ENUM's variant reference: one LEB128 and stop + { + uint64_t ignored = 0; + return getleb( ignored ); + } + case 12: case 13: case 14: case 16: case 31: case 32: // 31 is the ESCAPE, 32 the payload-free kind + { + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + case 15: // union: the arm id reference, then its kind, its L and its payload (reference 0 = empty) + { + uint64_t arm = 0; + if ( !getleb( arm ) ) return false; + if ( arm == 0 ) return true; + if ( !has( 1 ) ) return false; + offset += 1; // the arm's kind byte + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + } + return false; + } +}; + +// The RESERVED node-table id, the one id the language holds back +// (docs/SPEC-TABLES.md §3.1, §5). It rides in every unit, pointered or not, +// because every body has to know that a NESTED body claiming one is damaged. +static const uint64_t kTableNodeTableFieldId = 0xFFFFFFFFFFFFFFFFull; + +// TableWireForm is the FORM BYTE, and it is the whole header +// (docs/SPEC-TABLES.md §3). A reader that meets a byte it does not know +// refuses the wire by name and never reports damage. +const uint8_t kTableWireForm = 1; + +// TableOpen reads the form byte and the trailer, in that order, and hands back +// the ROOT BODY. It answers one of three verdicts, because five zero counters +// and a false flag are what a clean read prints too: +// +// TableOpenOk the form is known and the table read whole +// TableOpenRefused a FORM BYTE this reader does not carry: nothing is +// decoded, nothing is counted, and no damage is reported +// TableOpenDamaged a table that cannot be read whole — fewer than eight +// bytes, a count whose entries run past the front of the +// file, a count that leaves no room for the form byte, or +// ONE ID IN TWO ENTRIES. The whole wire is malformed, +// nothing is decoded, and one event is counted. +// TableOpenBodyStopped the form and the table were good and the ROOT BODY +// could not be walked to its own terminator. What it +// decoded before that is kept, as everywhere on this wire. +enum TableOpenVerdict { TableOpenOk, TableOpenRefused, TableOpenDamaged, TableOpenBodyStopped }; + +inline TableOpenVerdict TableOpen( const uint8_t * buffer, int64_t bytes, TableIdTable & table, int64_t & body_bytes ) +{ + if ( bytes < 1 ) { return TableOpenDamaged; } + if ( buffer[0] != kTableWireForm ) { return TableOpenRefused; } + if ( bytes < 9 ) { return TableOpenDamaged; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + uint64_t count = lo | ( hi << 32 ); + if ( count > (uint64_t) ( bytes / 8 ) ) { return TableOpenDamaged; } + const int64_t span = (int64_t) count * 8 + 8; + if ( span + 1 > bytes ) { return TableOpenDamaged; } + table.entries = buffer + bytes - span; + table.count = (int64_t) count; + // THE ENTRIES ARE DISTINCT: a table that carries one id twice is malformed + // for the whole wire, because no wire this schema writes carries a repeat + // and it would leave one more shape of table for a hostile writer to aim + // at (docs/SPEC-TABLES.md §3). + for ( int64_t i = 1; i < table.count; i++ ) + { + const uint64_t id = table.at( uint64_t( i ) + 1 ); + for ( int64_t j = 0; j < i; j++ ) + { + if ( table.at( uint64_t( j ) + 1 ) == id ) { return TableOpenDamaged; } + } + } + body_bytes = bytes - span - 1; + return TableOpenOk; +} + +// TableBodyExtent walks a body's framing to the zero reference that ends it, +// so a reader can tell a body that ENDED EARLY — leaving bytes no field claims +// — from one that is merely damaged. ANY BYTE BETWEEN THE ROOT'S TERMINATOR +// AND THE TABLE'S FIRST ENTRY IS MALFORMED, because no field claims it and the +// two ends of the file have met (docs/SPEC-TABLES.md §3). +inline bool TableBodyEndsEarly( const uint8_t * body, int64_t bytes, const TableIdTable & table ) +{ + TableReport ignored; + TableReader r( body, bytes, &ignored, &table ); + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { return false; } + if ( ref == 0 ) { return r.offset != bytes; } + if ( ref > (uint64_t) table.count ) { return false; } + if ( !r.has( 1 ) ) { return false; } + if ( !r.skip( r.get8() ) ) { return false; } + } +} + +// THE MESSAGE FORM (docs/SPEC-TABLES.md §3.3): a FILE carries its own id +// table and a MESSAGE STREAM announces one and then carries none. +// +// A form 2 wire is TWO PARTS, the form byte and the root body: the body ends +// at its own zero reference as it does in a file, there is no trailer, and the +// message's last byte is the body's terminator. Its references resolve against +// the CONNECTION's table, which is the unit's whole vocabulary in the order +// the compiler settled. +const uint8_t kTableWireMessageForm = 2; + +// The RESERVED build-version id, the second id the language holds back (§5, +// §11), beside the node table's. It is the announcement's one required field, +// and a reserved id in any body but the one whose transport it is, is +// malformed (§3.1). +static const uint64_t kTableBuildVersionFieldId = 0xFFFFFFFFFFFFFFFEull; + +// The reserved NODE-TABLE id's own slot in this unit's vocabulary (§3.3). A +// pointered message names the node table through it, exactly as every other +// field header names its id through a slot. +static const uint64_t kTableNodeTableFieldSlot = 38; + +// THE UNIT'S ANNOUNCEMENT, byte for byte: 61 entries and 508 bytes. It is an +// ordinary form 1 FILE — the form byte, a body carrying the BUILD VERSION +// under the reserved id at kind 9, and the trailer that IS the connection's +// table, slot 1 the reserved id and slots 2 and up the vocabulary under one +// numbering. +// +// The vocabulary is the unit's whole closure in the COOK PROJECTION's order +// (§20.2) — each record in the order the projection renders it and each +// record's fields in the order the projection renders them, then each enum's +// variants and each union's arms — followed by the tail the projection does +// not name: the reserved node-table id, the three blob type ids as bytes, +// string and wstring, and every table's own name id in the projection's sorted +// record order. The tail is UNCONDITIONAL, so an ordinary edit only ever grows +// it at its end and never moves a slot a generated field header carries as a +// literal. +static const int64_t kTableAnnounceBytes = 508; +static const uint8_t kTableAnnounce[ kTableAnnounceBytes ] = { + 0x01, 0x01, 0x09, 0xc7, 0x71, 0x45, 0xca, 0xda, 0x0e, 0x7c, 0x8d, 0x00, + 0xfe, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x30, 0xb1, 0x3a, 0xff, + 0x4a, 0xd9, 0xb1, 0x40, 0x20, 0xea, 0x4d, 0x40, 0x8e, 0xa7, 0x19, 0xaa, + 0x26, 0xa9, 0x02, 0x0c, 0x9b, 0x01, 0x48, 0x78, 0xe9, 0xea, 0x71, 0x6f, + 0x0f, 0x01, 0x82, 0xbf, 0x6f, 0x2c, 0x41, 0x4f, 0xbf, 0x84, 0x78, 0x3e, + 0xf3, 0xa4, 0x48, 0x44, 0x19, 0xab, 0xd7, 0x56, 0x05, 0x4a, 0xa3, 0x30, + 0x67, 0x55, 0x5b, 0x85, 0xc9, 0xe2, 0x4e, 0x30, 0x69, 0x6a, 0xb4, 0x81, + 0xfb, 0x67, 0x4d, 0x1a, 0xcf, 0x7b, 0x27, 0x21, 0x74, 0xa2, 0x79, 0x44, + 0x8e, 0xe2, 0xe5, 0xb1, 0x84, 0x76, 0xbc, 0x2e, 0xef, 0x83, 0x76, 0x1e, + 0xc5, 0x99, 0xf7, 0x82, 0x76, 0x4e, 0x0a, 0xd9, 0xa8, 0x2e, 0x86, 0x70, + 0x84, 0xed, 0xf2, 0x4a, 0xbb, 0xf0, 0x0c, 0x9b, 0xcc, 0xfb, 0x2d, 0x73, + 0x68, 0xb7, 0xf0, 0xae, 0x4c, 0x0c, 0xf6, 0x52, 0xbf, 0xe9, 0xd1, 0x2f, + 0x93, 0xcd, 0xda, 0xdb, 0x22, 0x72, 0x34, 0x7d, 0xf6, 0x0b, 0x72, 0x17, + 0x07, 0x17, 0x02, 0x86, 0x4c, 0xf5, 0x63, 0xaf, 0x54, 0x15, 0x02, 0x86, + 0x4c, 0xf4, 0x63, 0xaf, 0x3a, 0x70, 0x6e, 0x3e, 0x93, 0x43, 0xe5, 0x9d, + 0x3d, 0x62, 0xcb, 0x8f, 0xec, 0xfc, 0xf7, 0x39, 0x09, 0x06, 0x02, 0x86, + 0x4c, 0xeb, 0x63, 0xaf, 0x09, 0x4b, 0x4d, 0x57, 0xaa, 0x33, 0x47, 0xd2, + 0x31, 0x54, 0xaf, 0x1d, 0x19, 0x73, 0x50, 0x12, 0xb2, 0x0f, 0x40, 0x27, + 0x0b, 0x6b, 0x98, 0x01, 0x38, 0x81, 0x0a, 0xf1, 0x1f, 0x06, 0xa7, 0xa3, + 0x0f, 0x62, 0xad, 0x07, 0x77, 0x47, 0x82, 0x5f, 0x42, 0x4f, 0x4f, 0x30, + 0x0d, 0x39, 0x84, 0x1c, 0x86, 0x1b, 0x63, 0x8e, 0xba, 0xad, 0xbc, 0xc4, + 0xec, 0x10, 0x5b, 0x36, 0x19, 0x4a, 0xc9, 0x3d, 0xea, 0x0c, 0xe8, 0x30, + 0x94, 0xfd, 0xe4, 0x7c, 0xec, 0x22, 0x02, 0x86, 0x4c, 0xfc, 0x63, 0xaf, + 0x05, 0x28, 0x02, 0x86, 0x4c, 0xff, 0x63, 0xaf, 0x52, 0x26, 0x02, 0x86, + 0x4c, 0xfe, 0x63, 0xaf, 0xb1, 0x45, 0xc3, 0x44, 0x35, 0xab, 0xfe, 0x73, + 0xc0, 0x7f, 0xb3, 0x8a, 0xbe, 0x08, 0x63, 0x7f, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xe4, 0x4f, 0x1c, 0x4f, 0x47, 0xc0, 0x2e, 0x2f, + 0x58, 0xfc, 0xaf, 0xfa, 0xd8, 0xe0, 0x4b, 0x70, 0xc7, 0xd4, 0x7b, 0x26, + 0xb0, 0x9d, 0x29, 0x5f, 0xcc, 0x14, 0x15, 0x7f, 0xcb, 0xc2, 0x58, 0xd8, + 0x84, 0x9e, 0x3a, 0x55, 0x8b, 0x37, 0xe2, 0x06, 0x2a, 0x89, 0xf5, 0x12, + 0x09, 0xc1, 0xca, 0x0a, 0x44, 0xa2, 0x31, 0xc1, 0xad, 0xa7, 0xee, 0xee, + 0xe8, 0xcf, 0xbf, 0x43, 0x73, 0x18, 0x43, 0xd0, 0x42, 0xad, 0xf6, 0xf8, + 0x59, 0x86, 0x63, 0x91, 0xb7, 0xce, 0x00, 0x7c, 0xd1, 0xc5, 0x34, 0x20, + 0x06, 0x68, 0x47, 0x98, 0xd1, 0xa1, 0xcf, 0x52, 0x5f, 0x82, 0x58, 0xac, + 0x36, 0x15, 0x78, 0x5e, 0xb8, 0x8b, 0x59, 0x6f, 0xc9, 0xc6, 0x86, 0xbb, + 0xc3, 0x64, 0x89, 0x50, 0xd2, 0x8d, 0xa7, 0xf1, 0x80, 0xea, 0x3a, 0xb9, + 0xf1, 0x21, 0xf7, 0x41, 0x11, 0xed, 0xd9, 0xce, 0x96, 0x92, 0x43, 0x8a, + 0xfb, 0x06, 0xc9, 0xfe, 0x19, 0xe1, 0x13, 0xa0, 0xa7, 0x0a, 0xc7, 0x54, + 0x12, 0xd6, 0x40, 0xdc, 0x08, 0xf0, 0xf5, 0xc0, 0x24, 0x5f, 0xf8, 0x33, + 0xc8, 0xfb, 0x85, 0x9a, 0xaf, 0xe0, 0xc9, 0x0c, 0x91, 0x0a, 0x55, 0x60, + 0xf7, 0xa2, 0x07, 0xec, 0x8b, 0x6d, 0x02, 0x86, 0x43, 0xf3, 0xc2, 0x2e, + 0x87, 0x27, 0xcc, 0x86, 0xf0, 0xe0, 0x26, 0x8f, 0x3d, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, +}; + +// AnnounceMeasure is the announcement's byte count, which is a constant of the +// unit and not a walk. +inline int64_t AnnounceMeasure() { return kTableAnnounceBytes; } + +// Announce writes the announcement into the caller's buffer and answers the +// bytes written — exactly AnnounceMeasure's answer — or -1 when the buffer is +// too small. It allocates nothing and walks nothing. +inline int64_t Announce( uint8_t * buffer, int64_t capacity ) +{ + if ( buffer == NULL || capacity < kTableAnnounceBytes ) { return -1; } + memcpy( buffer, kTableAnnounce, (size_t) kTableAnnounceBytes ); + return kTableAnnounceBytes; +} + +// TableVocabulary is ONE DIRECTION of ONE CONNECTION's id table (§3.3): the +// entries an announcement carried, whole, under one numbering with slot 1 the +// reserved build-version id. +// +// A peer holds TWO of these for a connection, the one it writes with and the +// one it reads with, and neither is the other's. A restart opens a fresh +// connection with empty tables and nothing is cached across connections, so +// its whole life is one connection's. It BORROWS the announcement's bytes rather than +// copying them, so a receiver holds one table a direction and its memory is +// the bound below and nothing else. +struct TableVocabulary +{ + // THE CONFORMING DEFAULT BOUND (§3.3): 32 KiB a direction, eight times the + // 500-id unit that is already a large one. A connection's table is bounded + // by nothing the wire carries, so the receiver declares the maximum and an + // announcement above it is refused by name before an entry is touched. + static const int64_t kDefaultMaxEntries = 4096; + + TableIdTable table; + uint64_t build_version = 0; + bool announced = false; + int64_t max_entries = kDefaultMaxEntries; +}; + +// AnnounceRead reads an announcement into one direction's table (§3.3). +// +// THE BOUND IS CHECKED BEFORE ANYTHING IS ALLOCATED: the entry count is a +// fixed little-endian u64 at the end, so a receiver reads it, compares it and +// refuses without touching an entry. After that it is §3's ordinary FILE read, +// because the announcement IS a file, with EXACTLY ONE STRICT CHECK over its +// body: the reserved build-version field present, exactly once, under kind 9, +// eight bytes wide. Everything else is an ordinary field under §4's tolerance, +// so an unknown one is skipped and counted and the announcement can GAIN a +// field in a later minor without a lockstep redeploy. +// +// The FIRST announcement sets the table and it is the only one that can. A +// SECOND is refused by name: it does not replace the table, it does not amend +// it and it changes nothing. A refused announcement sets NO TABLE. +inline bool AnnounceRead( TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + if ( vocabulary.announced ) + { + to->refused = true; + to->reason = second_announcement; + return false; + } + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireForm ) + { + to->refused = true; + to->reason = buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + return false; + } + if ( bytes < 9 ) { to->malformed = true; return false; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + if ( ( lo | ( hi << 32 ) ) > (uint64_t) vocabulary.max_entries ) + { + to->refused = true; + to->reason = vocabulary_too_large; + return false; + } + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else { to->refused = true; to->reason = newer_form; } + return false; + } + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) { to->malformed = true; return false; } + // the body, under §4's tolerance and this form's one strict check + TableReader r( buffer + 1, body_bytes, to, &table ); + uint64_t version = 0; + int32_t seen = 0; + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { to->malformed = true; return false; } + if ( ref == 0 ) { break; } + if ( ref > (uint64_t) table.count || !r.has( 1 ) ) { to->malformed = true; return false; } + const uint64_t id = table.at( ref ); + const uint8_t kind = r.get8(); + if ( id != kTableBuildVersionFieldId ) + { + to->unknown++; + if ( !r.skip( kind ) ) { to->malformed = true; return false; } + continue; + } + if ( kind != 9 || !r.has( 8 ) ) { to->refused = true; to->reason = no_vocabulary; return false; } + version = r.get64(); + seen++; + } + if ( seen != 1 ) { to->refused = true; to->reason = no_vocabulary; return false; } + vocabulary.table = table; + vocabulary.build_version = version; + vocabulary.announced = true; + return true; +} + +inline float table_bits_to_float( uint32_t bits ) { float f; memcpy( &f, &bits, 4 ); return f; } +inline uint32_t table_float_to_bits( float f ) { uint32_t b; memcpy( &b, &f, 4 ); return b; } +inline double table_bits_to_double( uint64_t bits ) { double d; memcpy( &d, &bits, 8 ); return d; } +inline uint64_t table_double_to_bits( double d ) { uint64_t b; memcpy( &b, &d, 8 ); return b; } + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +#ifndef LISTDEMO_SCHEMA_TABLE_ARENA +#define LISTDEMO_SCHEMA_TABLE_ARENA + +namespace listdemo { + +// ---- variable-length tables: tuning constants (docs/SPEC-TABLES.md) ---- +// +// The segment size and the count multiply to exactly 2^32: the u32 reference +// is the arena's hard ceiling, and these constants saturate it rather than +// leaving address space unreachable. Slab handout costs one atomic per slab, +// so per-node allocation costs no synchronization at all. + +static const uint32_t kTableSegmentBits = 22; // 4 MiB segments +static const uint32_t kTableSegmentSize = 1u << kTableSegmentBits; +static const uint32_t kTableSegmentMask = kTableSegmentSize - 1u; +static const uint32_t kTableMaxSegments = 1u << ( 32 - kTableSegmentBits ); // 1024 -> 4 GiB +static const uint32_t kTableSlabBytes = 64u * 1024u; // one atomic per slab +static const uint32_t kTableAlign = 8; // every node starts 8-aligned +static const uint32_t kTableAllocFailed = 0xFFFFFFFFu; + +// ---- THE CALLER'S ALLOCATOR (docs/SPEC-TABLES.md §6.5) ---- +// +// Every allocation the variable-length runtime makes goes through one of +// these — the arena's segments, the pack walk's identity map, the numbering's +// entry array, the packed region, and the tool path's node directory. There is +// no other call to the C library on this path, so a counting allocator sees +// every byte and a game's own heap can own all of it. +// +// It is the shape TableBlockAllocator already has (§19.1): two function +// pointers and a context the caller carries. What it adds is a CONTRACT ON +// alloc — the bytes come back ZEROED. Lock copies whole nodes, PADDING +// INCLUDED, so anything left uninitialized reaches a packed region; the default +// pair reaches that through calloc, which costs nothing measurable because a +// fresh segment is untouched pages either way. +struct TableAllocator +{ + void * ( *alloc )( void * context, int64_t bytes ); // ZEROED bytes, NULL on failure + void ( *free )( void * context, void * pointer ); + void * context; +}; + +// The default pair, and it is the one every entry point takes when the caller +// names none. It calls schema_allocate / schema_release, so a program with its +// own C-library replacement can move the floor without writing a struct at all. +inline void * table_default_alloc( void * context, int64_t bytes ) { (void) context; return schema_allocate( bytes ); } +inline void table_default_free( void * context, void * pointer ) { (void) context; schema_release( pointer ); } + +inline TableAllocator TableDefaultAllocator() +{ + TableAllocator allocator; + allocator.alloc = table_default_alloc; + allocator.free = table_default_free; + allocator.context = NULL; + return allocator; +} + +// ---- TableRef: a relocatable reference (never a machine pointer) ---- +// +// Two encodings, one slot, and the FORM says which is in force: +// +// in the arena — the node's arena offset (segment index in the high bits) +// in a region — the SELF-RELATIVE byte delta from this slot's own address, +// so a deref is one add, needs no base pointer, and a whole +// region relocates by memcpy with zero fix-up +// +// 0 is null in both, and a slot can never name the node that contains it, so +// zero names nothing real in either form. +// +// A REGION DELTA HAS NO REQUIRED SIGN (§6.3). A region is packed depth-first, +// so a node's FIRST reference points forward; every LATER reference to that +// same node points BACK at the one body it already has, which is exactly what +// makes one node one node in a region. Sharing and a back-reference are the +// same fact, and nothing validates a reference by its sign. +// +// IT IS EIGHT BYTES, SIGNED, so ONE REGION REACHES EVERYTHING (§6.3, §7): a +// four-byte slot bounded a region at 2 GiB, and the scale a cook exists for is +// *"100mbs or many gigabytes of data in Assets.bin"*. +struct TableRef +{ + int64_t value = 0; + bool null() const { return value == 0; } +}; + +// TableSlot is what Alloc hands back: usable as the node pointer (write +// fields through it) AND as the reference to store in a pointer field. +template struct TableSlot +{ + T * ptr = NULL; + TableRef ref; + T * operator->() const { return ptr; } + T & operator*() const { return *ptr; } + operator T *() const { return ptr; } + operator TableRef() const { return ref; } + bool null() const { return ptr == NULL; } +}; + +inline uint32_t TableAlignUp( uint32_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( kTableAlign - 1 ); } +inline int64_t TableAlignUp64( int64_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( int64_t( kTableAlign ) - 1 ); } + +// ---- a BYTE BUFFER's node (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// A *bytes or *string slot is a TableRef like every pointer slot, and it names +// a BLOB NODE: this eight-byte header and then the bytes, at offset eight so +// the data is eight-aligned. A *string blob carries one more zero byte after +// its data, so a region hands back a C string with no copy. The node's extent +// is the header plus its bytes, rounded to the arena's alignment like every +// node's; on the wire it is a record whose body is the bytes (§3.1). +struct TableBlob +{ + uint32_t length; + uint32_t zero; +}; + +static const int64_t kTableBlobHeader = 8; // length (u32), then four zero bytes +static const int64_t kTableBlobMaxLength = 0xFFFFFFFF; // a record's length is a u32 (§3.1) + +// the node's storage: the header, the bytes, a string's terminator, rounded +// to the arena's alignment like every node +inline int64_t TableBlobStorage( int64_t length, bool terminated ) +{ + return TableAlignUp64( kTableBlobHeader + length + ( terminated ? 1 : 0 ) ); +} + +// What a read answers: a pointer INTO the region and the length, NULL and +// zero for a null slot. Off a locked region, a loaded one or an opened cook +// the pointer is one add from the slot, and nothing is copied. +struct TableBytesView +{ + const uint8_t * data; + int64_t length; +}; + +struct TableStringView +{ + const char * data; // zero-terminated + int64_t length; +}; + +// What AllocBytes and AllocString hand back: the bytes to write through, the +// length asked for, and the reference to store in the slot — the three +// answers TableSlot gives for a table node. +struct TableBytesSlot +{ + uint8_t * data = NULL; + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +struct TableStringSlot +{ + char * data = NULL; // room for length bytes and the terminator, already zero + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +// ---- the arena: segmented, slab-handed, lock-free by ownership ---- +// +// Allocation is thread-local inside a worker's slab — no atomics on the node +// path. A worker takes its next slab with ONE compare-exchange, and a new +// segment is published with one more. Nothing ever moves: a segment, once +// allocated, lives untouched until the arena is torn down, so a T* obtained +// from Alloc stays valid while other workers allocate, and an offset stays +// correct while the arena grows. +// +// The model this DELIBERATELY refuses: one buffer under a lock, grown by +// realloc. A realloc moves the buffer under workers mid-write; offsets fix +// identity but not the raw references already resolved from them, and the +// resulting corruption is invisible until much later. Segments never move, so +// that bug class cannot be written here. +// +// Slack: at most one slab tail per worker plus one slab per segment (a slab +// that will not fit is skipped rather than split), i.e. under 2% of a segment +// plus threads x 64 KiB. That is the price of never synchronizing per node. +struct TableArena +{ + std::atomic segments[ kTableMaxSegments ]; + std::atomic cursor; // (segment << kTableSegmentBits) | bytes handed out + bool locked = false; // MONOTONIC: Lock() is one-way, there is no unlock + // THE ARENA CARRIES ITS OWN, so everything downstream of a builder — + // segments, pack map, numbering, region, node directory — allocates through + // the one pair the caller named, with nothing to thread by hand. + TableAllocator allocator; +}; + +inline void TableArenaInit( TableArena & arena, TableAllocator allocator ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + arena.segments[i].store( NULL, std::memory_order_relaxed ); + } + arena.cursor.store( 0, std::memory_order_relaxed ); + arena.locked = false; + arena.allocator = allocator; +} + +inline void TableArenaShutdown( TableArena & arena ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + uint8_t * segment = arena.segments[i].exchange( NULL, std::memory_order_acq_rel ); + if ( segment != NULL ) { arena.allocator.free( arena.allocator.context, segment ); } + } + arena.cursor.store( 0, std::memory_order_relaxed ); +} + +// one L1 load plus an add: the segment table is 8 KiB and stays hot +inline uint8_t * TableArenaAt( const TableArena & arena, uint32_t offset ) +{ + return arena.segments[ offset >> kTableSegmentBits ].load( std::memory_order_relaxed ) + ( offset & kTableSegmentMask ); +} + +// TableArenaGrabSlab hands one worker its next private slab. Returns +// kTableAllocFailed when the arena's address space or the allocator is +// exhausted — a loud refusal, never a silent smaller slab. +inline uint32_t TableArenaGrabSlab( TableArena & arena ) +{ + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t segment = cursor >> kTableSegmentBits; + uint32_t used = cursor & kTableSegmentMask; + // strictly less: a slab is never split across segments, and the tail + // is the documented slack + if ( used + kTableSlabBytes < kTableSegmentSize ) + { + if ( arena.segments[segment].load( std::memory_order_acquire ) == NULL ) + { + // THE SEGMENT COMES BACK ZEROED, which is the allocator's + // contract and not an extra pass here: Lock copies whole nodes, + // PADDING INCLUDED, so anything uninitialized reaches a packed + // region. Value-initializing a node with placement new zeroes + // its MEMBERS and not its padding, so the zeroing has to happen + // at the segment or not at all. It costs nothing measurable: a + // fresh segment is untouched pages either way, and the default + // pair's calloc has the kernel hand them over zeroed. + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, (int64_t) kTableSegmentSize ); + if ( memory == NULL ) { return kTableAllocFailed; } + uint8_t * expected = NULL; + if ( !arena.segments[segment].compare_exchange_strong( expected, memory, std::memory_order_acq_rel ) ) + { + // another worker published this segment first + arena.allocator.free( arena.allocator.context, memory ); + } + } + if ( arena.cursor.compare_exchange_weak( cursor, cursor + kTableSlabBytes, std::memory_order_acq_rel ) ) + { + return ( segment << kTableSegmentBits ) | used; + } + continue; + } + uint32_t next_segment = segment + 1; + if ( next_segment >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + arena.cursor.compare_exchange_weak( cursor, next_segment << kTableSegmentBits, std::memory_order_acq_rel ); + } +} + +// TableArenaGrabSpan reserves a SPAN of the arena's address space for one node +// larger than a slab — a BYTE BUFFER of any size (docs/SPEC-TABLES.md §2.5) — +// and allocates it as one contiguous block. It takes whole segment indices +// from the cursor, starting at the index after the cursor's so nothing else +// is ever handed out inside the span, and publishes the block under the first +// of them; the indices the span covers past that one stay NULL, which is +// enough, because only a node's START is ever resolved through the segment +// table and a blob's bytes follow its header inside the one allocation. The +// unused tail of the segment the cursor was in is slack, like a slab tail. +// Returns kTableAllocFailed when the address space or the allocator is +// exhausted — a loud refusal, never a smaller blob. +inline uint32_t TableArenaGrabSpan( TableArena & arena, int64_t bytes ) +{ + if ( bytes <= 0 || bytes > ( (int64_t) kTableMaxSegments - 2 ) * (int64_t) kTableSegmentSize ) { return kTableAllocFailed; } + const uint32_t spanned = (uint32_t) ( ( bytes + kTableSegmentSize - 1 ) >> kTableSegmentBits ); + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t start = ( cursor >> kTableSegmentBits ) + 1; + if ( start + spanned >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + uint32_t next = ( start + spanned ) << kTableSegmentBits; + if ( !arena.cursor.compare_exchange_weak( cursor, next, std::memory_order_acq_rel ) ) { continue; } + // the span is this worker's now: nothing else can publish under its + // first index, so a plain store suffices, and the block comes back + // ZEROED like every segment — the blob's bytes and its tail are zeros + // until written + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, bytes ); + if ( memory == NULL ) { return kTableAllocFailed; } + arena.segments[start].store( memory, std::memory_order_release ); + return start << kTableSegmentBits; + } +} + +// ---- TableWorker: one thread's allocation front ---- +// +// The threading contract, stated plainly: +// * Alloc on YOUR OWN worker is safe concurrently with any other worker's. +// No locks, no atomics per node. +// * Writing fields of a node ANOTHER worker allocated is your own +// synchronization problem — this runtime does not arbitrate it. +// * Lock and Save are single-threaded: call them after the workers have +// joined. +struct TableWorker +{ + TableArena * arena = NULL; + uint32_t next = 0; + uint32_t end = 0; + + template TableSlot Alloc() + { + static_assert( alignof( T ) <= kTableAlign, "a table node's alignment must fit the arena's" ); + TableSlot slot; + if ( arena == NULL || arena->locked ) { return slot; } + uint32_t bytes = TableAlignUp( (uint32_t) sizeof( T ) ); + if ( bytes > kTableSlabBytes ) { return slot; } // a node larger than a slab: refused, never split + if ( end == 0 || next + bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return slot; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + uint32_t at = next; + next += bytes; + // A NODE IS BORN IN TWO HALVES: start its lifetime in the raw + // storage, then write the declared defaults ONE MEMBER AT A TIME. + // + // It is "T", not "T{}". Value-initialising the whole aggregate says + // the same thing and costs cl O(BYTES) TO COMPILE — it expands element + // by element in its front end — while both halves here cost + // O(declarations). The slab cap below refuses a large node at RUN + // TIME and bounds nothing at compile time: the cost is paid by + // whatever T a caller instantiates this with. + // Padding is not the difference: value-initialisation zeroes MEMBERS + // and not padding either way, which is why the segment is calloc'd. + // + // TableReset is an OVERLOAD SET, one per closure member, reached from + // this template by argument-dependent lookup on T's own namespace — + // Alloc is a template and cannot spell Reset. + // + // The reset is here because ONE DEFINITION SAYS WHAT THE DECLARED + // DEFAULTS ARE, and it is Reset. Default-initialisation lands on + // the same values today, because a member with a non-zero default + // carries a member initializer that says so — but that is the class + // definition agreeing with Reset, not the arena reading it, and #320's + // fix was itself a pass that MOVED initialisation between the two. + // The arena reads the definition. + slot.ptr = new ( TableArenaAt( *arena, at ) ) T; + TableReset( *slot.ptr ); + slot.ref.value = at; + return slot; + } + + // Alloc a BYTE BUFFER's node of exactly length bytes (docs/SPEC-TABLES.md + // §2.5): the blob header and its bytes, zeroed, in this thread's slab when + // it fits and in a span of the arena's own when it does not. NULL is the + // arena locked, a length below zero or past a record's u32, or the + // allocator refusing. The offset comes back for the reference. + TableBlob * AllocBlob( int64_t length, bool terminated, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( length < 0 || length > kTableBlobMaxLength ) { return NULL; } + const int64_t bytes = TableBlobStorage( length, terminated ); + if ( bytes > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, bytes ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + } + else + { + if ( end == 0 || next + (uint32_t) bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) bytes; + } + TableBlob * blob = (TableBlob *) TableArenaAt( *arena, at ); + blob->length = (uint32_t) length; // the bytes after it are the segment's zeros + blob->zero = 0; + return blob; + } + + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries + // no type id, takes no index and has no Reset, so it goes through the same + // slab and span the blob path uses rather than through Alloc. + uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( bytes <= 0 || align > (int64_t) kTableAlign ) { return NULL; } + const int64_t rounded = TableAlignUp64( bytes ); + if ( rounded > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, rounded ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + return TableArenaAt( *arena, at ); + } + if ( end == 0 || next + (uint32_t) rounded > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) rounded; + return TableArenaAt( *arena, at ); // the segment came back zeroed + } + // a *bytes node: the bytes to write through, and the reference to store + TableBytesSlot AllocBytes( int64_t length ) + { + TableBytesSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, false, at ); + if ( blob == NULL ) { return slot; } + slot.data = (uint8_t *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } + + // a *string node: room for length bytes and the zero byte after them + TableStringSlot AllocString( int64_t length ) + { + TableStringSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, true, at ); + if ( blob == NULL ) { return slot; } + slot.data = (char *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } +}; + +// ---- TablePackMap: the pack walk's identity map (docs/SPEC-TABLES.md §3.1, §6.2) ---- +// +// ONE ENTRY PER REACHABLE NODE, and that map IS identity: a node must know +// where it landed to be named a second time, so Lock packs a shared node ONCE +// and every later reference resolves to the one body it already has. That is +// the same first-visit numbering the wire uses, so the pack order and the node +// order are one order. +// +// COLOURING AN ENTRY WHILE ITS DESCENT IS OPEN COSTS ONE BIT, and it is what +// makes a data cycle free to refuse: a reference to an entry still open is a +// cycle, and Lock returns failure rather than recursing away. The ROOT's entry +// is open for the whole walk. +// +// The map is proportional to NODES, never to bytes, and it lives on the +// AUTHORING side, where §6.5 licenses allocation. Nothing on the reading path +// ever builds one. +struct TablePackEntry +{ + const void * key; // the node's address in the graph being packed + int64_t offset; // where that node landed in the region + uint8_t open; // its descent is still open: a reference here is a cycle +}; + +struct TablePackMap +{ + TablePackEntry * entries = NULL; + int64_t capacity = 0; // a power of two, or zero while empty + int64_t count = 0; + TableAllocator allocator; // the caller's, carried from the walk that built it +}; + +inline void TablePackMapInit( TablePackMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TablePackMapShutdown( TablePackMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TablePackMapInit( map, map.allocator ); +} + +// The two walks behind Lock re-derive the SAME map from the same graph — the +// numbering is never carried between them (§3.1) — so the second starts from +// an empty map and keeps the capacity the first paid for. +inline void TablePackMapReset( TablePackMap & map ) +{ + if ( map.entries != NULL ) { memset( map.entries, 0, (size_t) map.capacity * sizeof( TablePackEntry ) ); } + map.count = 0; +} + +// open addressing, linear probing, a multiply-shift hash over the address: a +// node key is a pointer and its low bits are alignment, so the low bits alone +// would collide on every node of one type +inline int64_t TablePackMapSlot( const TablePackMap & map, const void * key ) +{ + uint64_t hash = (uint64_t) (uintptr_t) key; + hash *= 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != NULL && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TablePackEntry * TablePackMapFind( TablePackMap & map, const void * key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TablePackEntry * entry = &map.entries[ TablePackMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +// QUADRUPLING, not doubling, and the reason is measured: growth rehashes every +// entry, and on a graph of 131,071 nodes the doubling schedule spent 45% of +// Lock in rehashing alone. Quadrupling from 1024 buys 1.35x on that graph and +// keeps the map NODE-proportional (§6.2) — under 128 bytes a node at its +// worst, right after a grow, and about 64 on average. +inline bool TablePackMapGrow( TablePackMap & map ) +{ + TablePackMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 1024; + grown.entries = (TablePackEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TablePackEntry ) ); + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == NULL ) { continue; } + grown.entries[ TablePackMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// REACH a node: one probe answers both questions the walk has. A true "taken" +// says this is a FIRST visit, and the entry is now the node's, coloured open +// at "offset"; otherwise the entry is the one the node already has, and its +// open bit says cycle or sharing. NULL is an allocation failure, and it is a +// refusal like any other: Lock fails rather than packing a graph it cannot +// track. +// +// It is one call and not a find followed by an insert because the walk asks +// this question twice per node — once to measure, once to pack — and every +// probe is a miss into a table larger than L2. +inline TablePackEntry * TablePackMapReach( TablePackMap & map, const void * key, int64_t offset, bool & taken, int64_t & slot ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TablePackMapGrow( map ) ) { return NULL; } + } + slot = TablePackMapSlot( map, key ); + TablePackEntry * entry = &map.entries[slot]; + taken = entry->key != key; // an empty slot is a first visit; the key is never NULL + if ( taken ) + { + entry->key = key; + entry->offset = offset; + entry->open = 1; + map.count++; + } + return entry; +} + +// The descent finished: the node keeps its entry — identity outlives the +// descent — and stops being a cycle. The "hint" is the slot Reach returned, and it +// is checked against the key rather than trusted, so a rehash between the two +// costs a second probe instead of correctness. +inline void TablePackMapClose( TablePackMap & map, const void * key, int64_t hint ) +{ + if ( hint >= 0 && hint < map.capacity && map.entries[hint].key == key ) + { + map.entries[hint].open = 0; + return; + } + TablePackEntry * entry = TablePackMapFind( map, key ); + if ( entry != NULL ) { entry->open = 0; } +} + +// ---- resolution contexts: which encoding a walk is reading ---- + +struct TableArenaCtx { const TableArena * arena; }; +struct TableRegionCtx {}; + +// ---- a BYTE BUFFER's resolution (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// The same two encodings a table pointer has, resolved the same way: a +// self-relative delta in a region — one add, no base — and an arena offset +// while the builder is mutable. The blob is reached through its header, and a +// view is the header plus eight and the header's first word. Nothing here +// allocates and nothing copies: off a locked region, a loaded one or an +// opened cook the view points INTO the region. +inline const TableBlob * TableBlobAt( const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableRegionCtx &, const TableRef & ref ) { return TableBlobAt( ref ); } +inline const TableBlob * TableBlobAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} + +inline TableBytesView TableBytesViewOf( const TableBlob * blob ) +{ + TableBytesView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const uint8_t *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} +inline TableStringView TableStringViewOf( const TableBlob * blob ) +{ + TableStringView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const char *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} + +// the const form's hot path: one add, no base +inline TableBytesView TableBytesAt( const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ref ) ); } +inline TableStringView TableStringAt( const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ref ) ); } +// and the context forms a walk uses: a region context, an arena context, or +// the arena itself while the builder is mutable +template inline TableBytesView TableBytesAt( const Ctx & ctx, const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ctx, ref ) ); } +template inline TableStringView TableStringAt( const Ctx & ctx, const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ctx, ref ) ); } + +// allocate a blob in the arena and point the slot at it; the slot holds the +// arena offset, as every slot does while the builder is mutable +inline uint8_t * TableBytesEmplace( TableWorker & worker, TableRef & slot, int64_t length ) +{ + TableBytesSlot allocated = worker.AllocBytes( length ); + slot = allocated.ref; + return allocated.data; +} +// the text is copied in when one is given; a NULL text leaves the zeros for +// the caller to fill +inline char * TableStringEmplace( TableWorker & worker, TableRef & slot, const char * text, int64_t length ) +{ + TableStringSlot allocated = worker.AllocString( length ); + slot = allocated.ref; + if ( allocated.data != NULL && text != NULL && length > 0 ) { memcpy( allocated.data, text, (size_t) length ); } + return allocated.data; +} + +// ---- the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table, and a +// pointer field rides as an INDEX into it under kind 17. The encoding is +// flat: no pointer edge is a nesting level, so a chain's length is not a depth, +// and two references to one node are one node. +// +// THE FIELD RIDES ONCE: an L with sixty-four bits of capability frames a +// numbering of any size, so the whole numbering is one contiguous payload and a +// save's node bodies have no aggregate ceiling. + +static const uint64_t kTableNodeIndexNull = 0; // absence and null are one value +static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts the table + +// The not-materialized sentinel (§6.3): a record whose type id this build could +// not name. Distinct from every real offset including the root's 0, so an index +// resolving through it yields NULL and can never fabricate the root. +static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; + +// ---- the numbering, on the SAVE side ---- +// +// One entry per reachable node in FIRST-VISIT order, so entry k is node index +// k + 2. The two thunks are what let one loop write a table of mixed types: the +// numbering walk knows each target's type STATICALLY at the site it numbers it, +// so it stores the instantiation there and the loop never asks what a node is. +struct TableNumbering; + +struct TableNodeEntry +{ + const void * node; + uint64_t type_id; + // the type id's MESSAGE-FORM SLOT (docs/SPEC-TABLES.md §3.3), stored where + // the numbering walk stores the id itself and for the same reason: the + // target's type is known STATICALLY at the site that numbers it, so a + // form 2 save reads the slot out of the entry instead of looking an id up. + // Every pointer target's type id is an entry of the announcement, which is + // what makes the slot a compile-time fact of a POINTERED message too. + uint64_t type_slot; + int64_t ( * measure )( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ); + bool ( * save )( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ); +}; + +struct TableNumbering +{ + TablePackMap seen; // node -> index; the ROOT is index 1, open for the whole walk + TableNodeEntry * entries = NULL; + int64_t count = 0; + int64_t capacity = 0; +}; + +// The numbering allocates through the map's pair rather than carrying a second +// copy of it: one numbering is one walk, and a walk has one allocator. +inline void TableNumberingInit( TableNumbering & n, TableAllocator allocator ) +{ + TablePackMapInit( n.seen, allocator ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +inline void TableNumberingShutdown( TableNumbering & n ) +{ + TableAllocator allocator = n.seen.allocator; + TablePackMapShutdown( n.seen ); + allocator.free( allocator.context, n.entries ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +// The index a numbered node was given, for the save that writes it into a +// pointer slot. False means the two walks disagree about the graph, which is a +// refusal and never a guess. +inline bool TableNumberingIndex( const TableNumbering & n, const void * node, uint64_t & index ) +{ + if ( n.seen.capacity == 0 ) { return false; } + const TablePackEntry & entry = n.seen.entries[ TablePackMapSlot( n.seen, node ) ]; + if ( entry.key != node ) { return false; } + index = (uint64_t) entry.offset; + return true; +} + +inline bool TableNumberingAppend( TableNumbering & n, const TableNodeEntry & entry ) +{ + if ( n.count == n.capacity ) + { + // GROW BY COPY, never by realloc: the allocator hook is a PAIR, and a + // game's heap is not required to have a resize primitive at all. The + // schedule quadruples, so the copying is amortized to a constant per + // entry and the growth is the same growth it always was. + int64_t capacity = n.capacity != 0 ? n.capacity * 4 : 256; + TableAllocator allocator = n.seen.allocator; + TableNodeEntry * grown = (TableNodeEntry *) allocator.alloc( allocator.context, capacity * (int64_t) sizeof( TableNodeEntry ) ); + if ( grown == NULL ) { return false; } + if ( n.entries != NULL ) + { + memcpy( grown, n.entries, (size_t) n.count * sizeof( TableNodeEntry ) ); + allocator.free( allocator.context, n.entries ); + } + n.entries = grown; + n.capacity = capacity; + } + n.entries[n.count++] = entry; + return true; +} + +// The thunks the numbering stores. Each resolves to the closure member's own +// MeasureBody / SaveBodyFields through an overload set in the member's DECLARING +// file, reached by argument-dependent lookup at instantiation — the same bridge +// the arena's TableReset uses, and the reason a numbering may span the files of +// one unit without any file naming another's members. +template +inline int64_t TableNodeMeasureThunk( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ) +{ + return TableNodeMeasure( *(const Ctx *) ctx, numbering, ids, *(const T *) node ); +} + +template +inline bool TableNodeSaveThunk( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ) +{ + return TableNodeSave( *(const Ctx *) ctx, numbering, w, ids, *(const T *) node ); +} + +// ---- a BYTE BUFFER's record (docs/SPEC-TABLES.md §2.5, §3.1) ---- +// +// A blob rides as a node record under one of two RESERVED type ids — the fold +// a table's name takes, over the keywords "bytes" and "string", which no table +// can be named — with the bytes as its body and nothing framed inside. These +// two thunks are what the numbering stores for a blob, as it stores a +// member's codec for a table: the length, and the bytes verbatim. +static const uint64_t kTableBytesTypeId = 0x2f2ec0474f1c4fe4ull; // fnv1a64( "bytes" ) +static const uint64_t kTableStringTypeId = 0x704be0d8faaffc58ull; // fnv1a64( "string" ) + +template +inline int64_t TableBlobMeasureThunk( const void *, const TableNumbering &, TableIds &, const void * node ) +{ + return (int64_t) ( (const TableBlob *) node )->length; +} + +template +inline bool TableBlobSaveThunk( const void *, const TableNumbering &, TableWriter & w, TableIds &, const void * node ) +{ + const TableBlob * blob = (const TableBlob *) node; + w.raw( (const void *) ( blob + 1 ), (int64_t) blob->length ); + return true; +} + +// TableNodeTableMeasure and TableNodeTableSave are the framing, and they are +// ONE fill rule written twice — measure derives it from the graph and save +// derives the same one, which is what makes measure == save hold across a +// pointer graph (§3.1). +// +// The field rides ONCE, under the reserved id, kind 12: the payload opens with +// the count and then carries the records back to back, each a type id +// REFERENCE, a length and a body. The reserved id is interned BEFORE the +// records, and a record's type id before its body, which is the first-use order +// the trailer is written in (§3). +template +inline int64_t TableNodeTablePayload( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + int64_t payload = TableLebBytes( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + payload += TableLebBytes( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return -1; } + payload += TableLebBytes( (uint64_t) body ) + body; + } + return payload; +} + +template +inline int64_t TableNodeTableMeasure( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return 0; } // a root that reaches no nodes writes none of them + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return -1; } + return TableLebBytes( ref ) + 1 + TableLebBytes( (uint64_t) payload ) + payload; +} + +template +inline bool TableNodeTableSave( const Ctx & ctx, TableWriter & w, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return true; } + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return false; } + w.putleb( ref ); + w.put8( 12 ); // kind 12 is the opaque byte payload: a reader that cannot name the id skips by L + w.putleb( (uint64_t) payload ); + w.putleb( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + w.putleb( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return false; } + w.putleb( (uint64_t) body ); + if ( !n.entries[k].save( (const void *) &ctx, n, w, ids, n.entries[k].node ) ) { return false; } + } + return true; +} + +// ---- the numbering, on the LOAD side: a region's NODE DIRECTORY (§6.3) ---- +// +// The wire's numbering made resident: one entry per numbered node, in index +// order, position i describing node index i + 1 — so position 0 is the ROOT at +// offset 0. It is ATTRIBUTION, and attribution is separable: nothing that reads +// a structure touches it, a deref is one add on a self-relative offset, and a +// caller may release it once Load returns. +struct TableNodeDirEntry +{ + uint64_t offset; + uint64_t type_id; +}; + +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; + +// TableNodeMap is what a pointer slot resolves through while a body decodes. +struct TableNodeMap +{ + uint8_t * base = NULL; + const TableNodeDirEntry * entries = NULL; + int64_t count = 0; // the ROOT's entry included, so it is records + 1 + bool good = false; // the node table read whole; a numbering that failed resolves nothing + // WHERE THE NODES LIVE, and therefore what a resolved slot holds: a region + // takes the SELF-RELATIVE delta so a deref is one add, and the tool's + // builder path takes the node's ARENA OFFSET (§6.3). + bool arena = false; + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. + TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; +}; + +// TableNodeResolve places one node index in a pointer slot, and every failure +// is one of §4's events with the pointer left null. The declared TARGET type id +// is checked at every index, the root's included: the root carries no record +// and therefore no wire type id, so the READER'S OWN root type is what the +// claim is checked against. +inline void TableNodeResolve( const TableNodeMap & map, TableRef & slot, uint64_t index, uint64_t target, TableReport * report ) +{ + slot.value = 0; + if ( index == kTableNodeIndexNull || !map.good ) { return; } + if ( index - 1 >= (uint64_t) map.count ) + { + report->malformed = true; // an index above node_count + 1 + return; + } + const TableNodeDirEntry & entry = map.entries[index - 1]; + if ( entry.offset == kTableNodeAbsent ) + { + // a node whose type id this build could not name KEEPS ITS INDEX, and + // every pointer naming it reads null. The unknown was counted once, at + // the node, not once per pointer. + return; + } + if ( entry.type_id != target ) + { + report->kind_mismatch++; + return; + } + slot.value = map.arena ? (int64_t) entry.offset + : (int64_t) ( ( map.base + entry.offset ) - (const uint8_t *) &slot ); +} + +// ---- the record SCAN, and it is the whole of load's bound (§3.1) ---- +// +// Reading follows no reference. The scan walks the root body's top-level fields, +// finds the ONE under the reserved id, and reads records out of its payload in +// order — the field rides once, so nothing is copied to make a body contiguous +// and the generated body decoder never learns the transport exists. +struct TableNodeScan +{ + TableReader fields; // over the ROOT body, skipping past everything else + const uint8_t * payload; // the node-table field's payload + int64_t payload_size; + int64_t payload_offset; + bool opened; // the root body has been walked for the field + uint64_t declared; + int64_t records; + bool present; // the root body carries a node table at all + bool malformed; + const TableIdTable * ids; +}; + +inline TableNodeScan TableNodeScanBegin( const uint8_t * body, int64_t size, TableReport * report, const TableIdTable * ids ) +{ + TableNodeScan s = { TableReader( body, size, report, ids ), NULL, 0, 0, false, 0, 0, false, false, ids }; + return s; +} + +// find the node-table field, or answer false when the root body has none. A +// body carrying an id more than once is legal input and THE LAST OCCURRENCE +// WINS (docs/SPEC-TABLES.md §3), so the walk runs to the terminator and keeps +// the last rather than stopping at the first. +inline bool TableNodeScanOpen( TableNodeScan & s ) +{ + if ( s.opened ) { return false; } + s.opened = true; + for ( ;; ) + { + uint64_t ref = 0; + if ( !s.fields.getleb( ref ) ) { break; } + if ( ref == 0 ) { break; } // the terminator + if ( s.ids == NULL || ref > (uint64_t) s.ids->count ) { break; } + const uint64_t id = s.ids->at( ref ); + if ( !s.fields.has( 1 ) ) { break; } + const uint8_t kind = s.fields.get8(); + if ( id == kTableNodeTableFieldId ) + { + s.present = true; + if ( kind != 12 ) { s.malformed = true; return false; } + uint64_t length = 0; + if ( !s.fields.getleb( length ) || !s.fields.room( length ) ) { s.malformed = true; return false; } + s.payload = s.fields.buffer + s.fields.offset; + s.payload_size = (int64_t) length; + s.fields.offset += (int64_t) length; + continue; + } + if ( !s.fields.skip( kind ) ) { break; } + } + if ( s.payload == NULL ) { return false; } + TableReader head( s.payload, s.payload_size, s.fields.report, s.ids ); + if ( !head.getleb( s.declared ) ) { s.malformed = true; return false; } + s.payload_offset = head.offset; + return true; +} + +// the next record, or false at the end of the table — s.malformed says whether +// the end was the end or the framing giving out +inline bool TableNodeScanNext( TableNodeScan & s, uint64_t & type_id, const uint8_t * & body, int64_t & length ) +{ + if ( !s.opened && !TableNodeScanOpen( s ) ) { return false; } + if ( s.payload == NULL || s.payload_offset >= s.payload_size ) { return false; } + TableReader rec( s.payload, s.payload_size, s.fields.report, s.ids ); + rec.offset = s.payload_offset; + uint64_t ref = 0; + if ( !rec.getleb( ref ) || ref == 0 || s.ids == NULL || ref > (uint64_t) s.ids->count ) + { + s.malformed = true; // a type id reference of 0, or one past the table + return false; + } + type_id = s.ids->at( ref ); + uint64_t declared_length = 0; + if ( !rec.getleb( declared_length ) ) + { + s.malformed = true; // a record whose length is damaged + return false; + } + if ( declared_length > (uint64_t) ( s.payload_size - rec.offset ) ) + { + s.malformed = true; // a record whose length runs past its field + return false; + } + body = s.payload + rec.offset; + length = (int64_t) declared_length; + s.payload_offset = rec.offset + length; + s.records++; + return true; +} + +// The record scan is AUTHORITATIVE: node_count is data from the wire, and a +// count that disagrees with the scan is malformed. Nothing is sized from it +// before the scan has confirmed it. +inline bool TableNodeScanWhole( TableNodeScan & s ) +{ + if ( s.malformed ) { return false; } + if ( !s.present ) { return true; } // no node table at all is not a broken one + return s.declared == (uint64_t) s.records; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_ARENA + +#ifndef LISTDEMO_SCHEMA_TABLE_EXTENT +#define LISTDEMO_SCHEMA_TABLE_EXTENT + +namespace listdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_EXTENT + +#ifndef LISTDEMO_SCHEMA_TABLE_MAP +#define LISTDEMO_SCHEMA_TABLE_MAP + +namespace listdemo { + +// ---- a MAP: a sorted entry array, and the lookup over it (§2.8) ---- +// +// On the wire, in a region and in a cook a map is an array of one generated +// ENTRY table held in ascending key order. What this adds is Find — a binary +// search over that array where it lies — and a builder that inserts, replaces +// and erases by key. Nothing here is stored: a region and a cook carry the +// array and the count, and not one byte about a hash or a probe. + +// entries carved from ONE call to the allocator pair; a new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableMapSegmentEntries = 32; + +// TableDeclRef names a type in an unevaluated context and is never defined — +// what 's declval is for, without the include the generated corpus +// refuses to pay for (the iterator_traits note, §13.9). +template T & TableDeclRef(); + +// THE ORDER IS TOTAL, AND IT IS THE SAME IN NINE LANGUAGES (§2.8). Integers +// compare by VALUE, signed for the signed kinds and unsigned for the unsigned. +// Strings compare by BYTES, unsigned, a shorter string that is a prefix of a +// longer one first: memcmp over the common length, then the lengths. Never a +// locale, never a code point, never a case fold. +inline int TableKeyOrder( uint64_t a, uint64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( int64_t a, int64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( const char * a, int32_t a_length, const char * b, int32_t b_length ) +{ + const int32_t common = a_length < b_length ? a_length : b_length; + if ( common > 0 ) + { + const int order = memcmp( (const void *) a, (const void *) b, (size_t) common ); + if ( order != 0 ) { return order < 0 ? -1 : 1; } + } + return a_length < b_length ? -1 : ( a_length > b_length ? 1 : 0 ); +} + +// the length of a NUL-terminated key at a call site, bounded by the storage it +// has to fit: a key one byte longer than the bound is refused, never truncated +inline int32_t TableKeyLength( const char * key, int32_t bound ) +{ + if ( key == NULL ) { return 0; } + for ( int32_t i = 0; i <= bound; i++ ) { if ( key[i] == 0 ) { return i; } } + return bound + 1; // longer than the bound: the caller refuses it +} + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.8, §7.2) ---- +// +// An int64 self-relative reference to the entry array and an int32 count, then +// padding to eight. The reference is a TableRef like a pointer's: in the arena +// it names the builder's HEAD, in a region it is the delta from the slot to +// the first entry, and 0 is the empty map in both. +template struct TableMap +{ + TableRef entries; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Entry * Entries() const + { + return entries.value != 0 ? (const Entry *) ( (const uint8_t *) &entries + entries.value ) : NULL; + } + int32_t size() const { return count; } + + // FIND: floor( log2 n ) + 1 key compares, in place, no allocation. NULL + // when absent, and on a map[K]*T the RESOLVED pointer, which is what a + // pointer field's accessor answers. + template const Entry * FindEntry( Key key ) const + { + const Entry * base = Entries(); + int32_t low = 0, high = count; + while ( low < high ) + { + const int32_t mid = low + ( high - low ) / 2; + const int order = TableEntryOrder( base[mid], key ); + if ( order == 0 ) { return base + mid; } + if ( order < 0 ) { low = mid + 1; } else { high = mid; } + } + return NULL; + } + // the return type is DEDUCED, so it is worked out when a call site + // instantiates Find and not when the holder's record declares the slot — + // which is what lets the entry's own overloads be declared after it + template auto Find( Key key ) const + { + return TableEntryFound( FindEntry( key ) ); + } + + // ---- iteration: ASCENDING key order, the key beside the value ---- + // + // A proxy BY VALUE, the keyed array's shape (§2.4): for ( auto [ key, + // value ] : map ). It carries no iterator_traits, for the reason + // TableKeyed's does not (§13.9). + struct ConstEntry + { + decltype( TableEntryKey( TableDeclRef() ) ) key; + decltype( TableEntryFound( (const Entry *) NULL ) ) value; + }; + + struct ConstIterator + { + const Entry * at; + ConstEntry operator*() const { return ConstEntry{ TableEntryKey( *at ), TableEntryFound( at ) }; } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Entries() }; } + ConstIterator end() const { return ConstIterator{ Entries() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.8, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first entry is inserted. Each +// segment is a fixed number of entries carved from one call to the allocator +// pair. An entry's address is stable for the arena's life, so a value handed +// back by an insert stays valid while other entries arrive. +struct TableMapHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an insert appends into + int32_t live; + int32_t dead; +}; + +template struct TableMapSegment +{ + TableRef next; + int32_t used; // entries carved from this segment + int32_t padding; + uint32_t dead[ ( kTableMapSegmentEntries + 31 ) / 32 ]; // Erase marks one bit, never the entry + Entry entries[ kTableMapSegmentEntries ]; +}; + +inline bool TableMapSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// ---- the ORDERED CURSOR the four writing walks read (§2.8) ---- +// +// Measure, Save, Lock and Cook each write a map's entries in ascending key +// order with no key twice, deriving the order from the builder's entries as +// each walk derives the numbering (§3.1). Nothing passes between them, so +// measure == save over a map is a real check on two sorts agreeing. +// +// A REGION is already sorted, so its cursor is the array in place and +// allocates nothing. The BUILDER's is the sort: an array of entry pointers +// allocated through the pair and released before the walk returns, because +// sorting the segments themselves would move entries whose addresses a caller +// holds. +template struct TableMapCursor +{ + const Entry * const * order = NULL; // the builder's form: sorted pointers + const Entry * entries = NULL; // the region's form: the array in place + int32_t count = 0; + TableAllocator allocator; + bool ok = false; + const Entry * operator[]( int32_t index ) const + { + return order != NULL ? order[index] : entries + index; + } +}; + +// heapsort: O( n log n ) once per map, no recursion, no allocation past the +// pointer array the caller already paid for +template inline void TableMapSort( const Entry ** order, int32_t count ) +{ + for ( int32_t start = count / 2 - 1; start >= 0; start-- ) + { + int32_t root = start; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= count ) { break; } + if ( child + 1 < count && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * swap = order[root]; order[root] = order[child]; order[child] = swap; + root = child; + } + } + for ( int32_t end = count - 1; end > 0; end-- ) + { + const Entry * swap = order[0]; order[0] = order[end]; order[end] = swap; + int32_t root = 0; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= end ) { break; } + if ( child + 1 < end && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * hold = order[root]; order[root] = order[child]; order[child] = hold; + root = child; + } + } +} + +// the REGION form: the array is already sorted, so the cursor is the array +template +inline TableMapCursor TableMapOrder( const TableRegionCtx &, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.entries = map.Entries(); + cursor.count = map.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: gather the LIVE entries out of the segment chain in +// insertion order, then sort. A dead entry costs nothing on any wire (§2.8). +template +inline TableMapCursor TableMapOrder( const TableArena & arena, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.allocator = arena.allocator; + cursor.count = map.count; + if ( map.entries.value == 0 || map.count <= 0 ) { cursor.ok = map.count == 0; cursor.count = 0; return cursor; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + if ( head->live != map.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + const Entry ** order = (const Entry **) arena.allocator.alloc( arena.allocator.context, (int64_t) map.count * (int64_t) sizeof( const Entry * ) ); + if ( order == NULL ) { return cursor; } + int32_t at = 0; + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 && at < map.count ) + { + const TableMapSegment * segment = (const TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used && at < map.count; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + order[at++] = segment->entries + i; + } + segment_ref = segment->next; + } + if ( at != map.count ) + { + arena.allocator.free( arena.allocator.context, order ); + return cursor; + } + TableMapSort( order, map.count ); + cursor.order = order; + cursor.ok = true; + return cursor; +} + +template +inline TableMapCursor TableMapOrder( const TableArenaCtx & ctx, const TableMap & map ) +{ + return TableMapOrder( *ctx.arena, map ); +} + +template inline void TableMapRelease( TableMapCursor & cursor ) +{ + if ( cursor.order != NULL ) { cursor.allocator.free( cursor.allocator.context, (void *) cursor.order ); } + cursor.order = NULL; +} + +// ---- the builder's five (§2.8) ---- +// +// Insert APPENDS after one LINEAR SCAN of the live entries for the key it may +// replace, Find is that same scan, and Erase is the scan and one bit. The +// builder builds NO INDEX, and that is a rule: the sort happens once, at Lock, +// Save or Cook, and every lookup that matters runs over the sorted region. + +// the head, allocated when the first entry is inserted +template +inline TableMapHead * TableMapReach( TableWorker & worker, TableMap & map ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( map.entries.value != 0 ) { return (TableMapHead *) TableArenaAt( *worker.arena, (uint32_t) map.entries.value ); } + uint32_t at = 0; + TableMapHead * head = (TableMapHead *) worker.AllocRaw( (int64_t) sizeof( TableMapHead ), (int64_t) alignof( TableMapHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + map.entries.value = (int64_t) at; + return head; +} + +// one entry's storage, appended: the current segment when it has room, a new +// one carved from one call to the pair when it does not +template +inline Entry * TableMapAppend( TableWorker & worker, TableMapHead * head, TableMap & map ) +{ + TableMapSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableMapSegmentEntries ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableMapSegment *) worker.AllocRaw( (int64_t) sizeof( TableMapSegment ), (int64_t) alignof( TableMapSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableMapSegment * previous = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Entry * entry = segment->entries + segment->used; + segment->used++; + head->live++; + map.count++; + return entry; +} + +// the LINEAR SCAN: the live entries in insertion order, O( n ) key compares +template +inline Entry * TableMapScan( const TableArena & arena, const TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) == 0 ) { return segment->entries + i; } + } + segment_ref = segment->next; + } + return NULL; +} + +// ERASE marks the entry DEAD, one bit in the segment's slot and not in the +// entry table, and decrements the live count. Its storage is reclaimed at +// RESET and never reused mid-build, because reusing a slot would make "an +// entry's address is stable" false for exactly one case. +template +inline bool TableMapErase( TableArena & arena, TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return false; } + TableMapHead * head = (TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) != 0 ) { continue; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + map.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INSERTION order, live entries only (§2.8) ---- +template struct TableMapEach +{ + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableMapSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableMapSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + auto operator*() const { return TableEntryEach( segment->entries + index ); } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableMapEach TableMapEachOf( const TableArena & arena, const TableMap & map ) +{ + TableMapEach each = { &arena, TableRef() }; + if ( map.entries.value != 0 ) + { + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + each.first = head->first; + } + return each; +} + +// ---- the LOAD side: where a decoded entry lands (§2.8) ---- +// +// THE READER TRUSTS NOTHING and spends one compare per entry. Every load path +// applies the same rules and produces one report (§4), so the region load of +// §6.5 and LoadBuilder never disagree about a wire. These two shapes are what +// makes that true with one generated decoder: a REGION carves the entry array +// out of the holder node's own extent, and the TOOL's path appends into the +// builder's arena, and the decoder above them cannot tell which it has. + +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. + +// TableMapFill is one map field being decoded: where the next entry lands, and +// the entry that last LANDED, which is what the ascending check compares +// against. +template struct TableMapFill +{ + TableMap * map = NULL; + Entry * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; +}; + +template +inline TableMapFill TableMapFillBegin( const TableNodeMap & nodes, TableMap & map, uint32_t n ) +{ + TableMapFill fill; + fill.map = ↦ + map.entries.value = 0; + map.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Entry ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Entry ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Entry *) base; + fill.capacity = (int32_t) n; + map.entries.value = (int64_t) ( base - (const uint8_t *) &map.entries ); + fill.ok = true; + return fill; +} + +// the entry that last LANDED — NULL before the first +template inline Entry * TableMapFillLast( TableMapFill & fill ) +{ + if ( fill.map->count <= 0 ) { return NULL; } + if ( fill.array != NULL ) { return fill.array + ( fill.map->count - 1 ); } + return TableMapLive( *fill.worker->arena, *fill.map, fill.map->count - 1 ); +} + +// the next slot, at the entry type's declared defaults +template inline Entry * TableMapFillNext( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + if ( fill.map->count >= fill.capacity ) { return NULL; } + Entry * entry = fill.array + fill.map->count; + TableReset( *entry ); + fill.map->count++; + return entry; + } + TableMapHead * head = TableMapReach( *fill.worker, *fill.map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( *fill.worker, head, *fill.map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// A MAP WITH HALF ITS KEYS IS NOT A MAP (§2.8): at the first entry whose key +// kind disagrees with the reader's declaration the map resets to EMPTY, one +// kind_mismatch is counted for the map, and its remaining bytes are skipped. +template inline void TableMapFillReset( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + fill.map->entries.value = 0; + fill.map->count = 0; + return; + } + if ( fill.map->entries.value != 0 ) + { + TableMapHead * head = (TableMapHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.map->entries.value ); + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + } + fill.map->count = 0; +} + +// an EMPTY map's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableMapFillEnd( TableMapFill & fill ) +{ + if ( fill.array != NULL && fill.map->count == 0 ) { fill.map->entries.value = 0; } +} + +// the k-th LIVE entry of a builder map, in insertion order — what the tool +// path's ascending check compares against +template +inline Entry * TableMapLive( const TableArena & arena, const TableMap & map, int32_t index ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + int32_t at = 0; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( at == index ) { return segment->entries + i; } + at++; + } + segment_ref = segment->next; + } + return NULL; +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.8, §6.5) ---- +// +// LoadMeasure's term for a map is N x sizeof( Entry ) rounded to +// alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this +// reads no field: it walks the map's own header and, where an entry's value +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect, +// and a refusal carries its reason (§6.5). +// A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its +// own L and the body's terminator, and under this form's variable lengths that +// footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a +// map's L can carry, and therefore what a LoadMeasure may be asked for. +static const int64_t kTableMapEntryFloor = 2; + +inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry + at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); + at += (int64_t) n * entry_size; + if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// ---- the TEXT form's placement (docs/SPEC-TABLES.md §2.8, §16) ---- +// +// The text is a plain JSON object keyed by the KEY, and the generic walk fills +// it through the ENTRY'S OWN descriptor — so all it needs from here is one +// entry at one key, handed back at its defaults. It is the builder's Insert +// with the ENTRY returned rather than its value, because the walk writes the +// value through a field row and not through a typed pointer. +template +inline Entry * TableMapPlace( TableWorker & worker, TableMap & map, Key key ) +{ + if ( worker.arena == NULL ) { return NULL; } + Entry * found = TableMapScan( *worker.arena, map, key ); + if ( found != NULL ) + { + TableResetMapValue( *found ); // a repeated key is LAST-WINS, whole + return found; + } + TableMapHead * head = TableMapReach( worker, map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( worker, head, map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// ---- the OPTIONAL RUNTIME INDEX (§2.8) ---- +// +// Open addressing with LINEAR PROBING over the sorted array, built AT LOAD for +// a map large enough that log n compares over a cold array cost more than one +// hash and a probe. IT IS NEVER STORED: the caller measures it, owns its +// storage, builds it in one pass and releases it whenever. +// +// ITS HASH AND ITS LOAD FACTOR ARE NOT A CROSS-PORT CONTRACT, and that is a +// rule. What a port is held to is the CONTRACT of the lookup: the same value +// the sorted array's Find returns for the same key, and no allocation past the +// storage the caller handed in. +struct TableMapIndex +{ + int32_t * slots = NULL; // entry indices, +1; 0 is an empty slot + int32_t capacity = 0; + bool good = false; +}; + +// this runtime's own, and no port reproduces it: fnv1a64 over the key's bytes +inline uint64_t TableMapHash( const void * bytes, int32_t length ) +{ + uint64_t hash = 0xCBF29CE484222325ull; + const uint8_t * at = (const uint8_t *) bytes; + for ( int32_t i = 0; i < length; i++ ) { hash ^= (uint64_t) at[i]; hash *= 0x100000001B3ull; } + return hash; +} +inline uint64_t TableMapHash( uint64_t key ) { return TableMapHash( (const void *) &key, (int32_t) sizeof( key ) ); } + +// this runtime's own load factor, and no port reproduces it either: the next +// power of two at or above twice the count, so a probe run stays short +inline int32_t TableMapIndexSlots( int32_t count ) +{ + int32_t slots = 8; + while ( slots < count * 2 ) { slots *= 2; } + return slots; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_MAP + +#ifndef LISTDEMO_SCHEMA_TABLE_LIST +#define LISTDEMO_SCHEMA_TABLE_LIST + +namespace listdemo { + +// ---- an UNBOUNDED ARRAY: a counted array whose count the data decides (§2.9) ---- +// +// On the wire, in a region and in a cook a list is the kind 14 body a [..N]T +// writes, its elements by-value records inside the holder's node extent. What +// this adds is the slot, a builder that appends into segments that never +// move, and a const surface that indexes and iterates in place. There is no +// sort, no key and no lookup: the order is INSERTION order, and it is +// identity the way position is identity in a fixed array. + +// elements carved from ONE call to the allocator pair. A new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableListSegmentElements = 32; + +// THE ELEMENT STORAGE: T itself, and a TableRef slot for a []*T, whose +// elements are references exactly as a pointer field's slot is (§2.1) +template struct TableListStorage { typedef T Element; }; +template struct TableListStorage { typedef TableRef Element; }; + +// WHAT THE CONST FORM ANSWERS: the element by reference, and on a []*T the +// RESOLVED pointer, one add on the self-relative delta, NULL for a null slot, +// exactly as At answers it (§6.2, §6.3) +template struct TableListConst +{ + typedef const T & Result; + static Result At( const T * element ) { return *element; } +}; +template struct TableListConst +{ + typedef const T * Result; + static Result At( const TableRef * element ) + { + return element->value != 0 ? (const T *) ( (const uint8_t *) element + element->value ) : NULL; + } +}; + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.9, §7.2) ---- +// +// An int64 self-relative reference to the element array and an int32 count, +// then padding to eight. The reference is a TableRef like a pointer's: in the +// arena it names the builder's HEAD, in a region it is the delta from the slot +// to the first element, and 0 is the empty list in both. It is the map's slot +// exactly, because it is the same two facts. +template struct TableList +{ + typedef typename TableListStorage::Element Element; + + TableRef elements; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Element * Elements() const + { + return elements.value != 0 ? (const Element *) ( (const uint8_t *) &elements + elements.value ) : NULL; + } + int32_t size() const { return count; } + + // INDEXING IS BOUNDS-CHECKED IN EVERY BUILD (§2.4, §2.9): the extent is a + // number that CAME FROM A FILE, so an index past it is not a mistake a + // release build gets to make cheaply. There is no undefined-behavior path + // here in any configuration. The assert carries the message where a + // debugger can read it and NDEBUG removes that. The fatal is what stands + // after it. Both go through the hooks: define schema_assert and + // schema_fatal and this refusal lands in your own handler. + void RefuseIndex( int32_t index ) const + { + if ( (uint32_t) index >= (uint32_t) count ) + { + schema_assert( false && "an unbounded array is indexed inside its count, which came from a file" ); + schema_fatal(); + } + } + typename TableListConst::Result operator[]( int32_t index ) const + { + RefuseIndex( index ); + return TableListConst::At( Elements() + index ); + } + + // ---- iteration: INDEX order, the element and no key ---- + // + // It carries no iterator_traits, for the reason TableKeyed's does not + // (§13.9). + struct ConstIterator + { + const Element * at; + typename TableListConst::Result operator*() const { return TableListConst::At( at ); } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Elements() }; } + ConstIterator end() const { return ConstIterator{ Elements() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.9, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first element is added. Each +// segment is a fixed number of elements carved from one call to the allocator +// pair. An element's address is stable for the arena's life, so a T * handed +// back by Add stays valid while other elements arrive. +struct TableListHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an Add appends into + int32_t live; + int32_t dead; +}; + +template struct TableListSegment +{ + TableRef next; + int32_t used; // elements carved from this segment + int32_t padding; + uint32_t dead[ ( kTableListSegmentElements + 31 ) / 32 ]; // Erase marks one bit, never the element + Element elements[ kTableListSegmentElements ]; +}; + +inline bool TableListSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// the head, allocated when the first element is added +template +inline TableListHead * TableListReach( TableWorker & worker, TableList & list ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( list.elements.value != 0 ) { return (TableListHead *) TableArenaAt( *worker.arena, (uint32_t) list.elements.value ); } + uint32_t at = 0; + TableListHead * head = (TableListHead *) worker.AllocRaw( (int64_t) sizeof( TableListHead ), (int64_t) alignof( TableListHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + list.elements.value = (int64_t) at; + return head; +} + +// one element's storage, appended: the current segment when it has room, a +// new one carved from one call to the pair when it does not. NULL means NOT +// ADDED: an arena that cannot carve another segment, or a count at the int32 +// cap (§2.2, §2.9). +template +inline typename TableList::Element * TableListAppend( TableWorker & worker, TableListHead * head, TableList & list ) +{ + typedef typename TableList::Element Element; + if ( list.count >= INT32_MAX ) { return NULL; } // the int32 storage cap + TableListSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableListSegmentElements ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableListSegment *) worker.AllocRaw( (int64_t) sizeof( TableListSegment ), (int64_t) alignof( TableListSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableListSegment * previous = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Element * element = segment->elements + segment->used; + segment->used++; + head->live++; + list.count++; + return element; +} + +// ADD, whole: the head, the append, and the element at its declared defaults +// (§2.9). The text form's placement is this same call, because a list has no +// key to place under (§16). +template +inline typename TableList::Element * TableListPlace( TableWorker & worker, TableList & list ) +{ + typedef typename TableList::Element Element; + TableListHead * head = TableListReach( worker, list ); + if ( head == NULL ) { return NULL; } + Element * element = TableListAppend( worker, head, list ); + if ( element == NULL ) { return NULL; } + new ( element ) Element(); // value-init: the declared defaults, and null for a slot + return element; +} + +// ERASE, ADDRESSED BY THE POINTER (§2.9): the element Add handed back is the +// handle, because a list has no key and the address is the one thing the +// builder promises never moves (§6.4). It marks the element DEAD, one bit in +// the segment's slot and not in the element storage, and decrements the live +// count. False when the pointer is not this list's. Its storage is reclaimed +// at RESET and never reused mid-build, the map's rule for the map's reason. +template +inline bool TableListErase( TableArena & arena, TableList & list, const typename TableList::Element * element ) +{ + typedef typename TableList::Element Element; + if ( list.elements.value == 0 || element == NULL ) { return false; } + TableListHead * head = (TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableListSegment * segment = (TableListSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + if ( element >= segment->elements && element < segment->elements + segment->used ) + { + const int32_t i = (int32_t) ( element - segment->elements ); + if ( TableListSegmentDead( segment->dead, i ) ) { return false; } // already erased + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + list.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INDEX order, live elements only (§2.9) ---- +template struct TableListEach +{ + typedef typename TableList::Element Element; + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableListSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableListSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + Element * operator*() const { return segment->elements + index; } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableListEach TableListEachOf( const TableArena & arena, const TableList & list ) +{ + TableListEach each = { &arena, TableRef() }; + if ( list.elements.value != 0 ) + { + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + each.first = head->first; + } + return each; +} + +// ---- the INDEX-ORDER CURSOR the four writing walks read (§2.9) ---- +// +// Measure, Save, Lock and Cook each visit a list's live elements in the order +// they were added, and they allocate nothing to do it: a region's cursor is +// the array in place, and the builder's walks the segment chain. Indexing the +// builder's form is SEQUENTIAL by construction, every walk steps i, i + 1, +// i + 2, so the cursor remembers where the last access landed and moves one +// live slot per step. An access behind the memo restarts from the first +// segment, which no walk here does. +template struct TableListCursor +{ + const Element * elements = NULL; // the region's form: the array in place + const TableArena * arena = NULL; // the builder's form: the segments + TableRef first; + int32_t count = 0; + bool ok = false; + // the memo: the segment and slot the last access landed on, and the live + // index that slot holds + mutable const TableListSegment * segment = NULL; + mutable int32_t within = -1; + mutable int32_t logical = -1; + + const Element * At( int32_t index ) const + { + if ( elements != NULL ) { return elements + index; } + if ( segment == NULL || index < logical ) + { + segment = first.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL; + within = -1; + logical = -1; + } + while ( logical < index ) + { + for ( ;; ) + { + within++; + while ( segment != NULL && within >= segment->used ) + { + segment = segment->next.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + within = 0; + } + if ( segment == NULL ) { return NULL; } // the slot and the head disagree + if ( !TableListSegmentDead( segment->dead, within ) ) { break; } + } + logical++; + } + return segment->elements + within; + } + const Element & operator[]( int32_t index ) const { return *At( index ); } +}; + +// the REGION form: the array is the cursor +template +inline TableListCursor::Element> TableListElements( const TableRegionCtx &, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.elements = list.Elements(); + cursor.count = list.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: the live elements out of the segment chain, in the +// order they were added. A dead element costs nothing on any wire (§2.9). +template +inline TableListCursor::Element> TableListElements( const TableArena & arena, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.arena = &arena; + cursor.count = list.count; + if ( list.elements.value == 0 || list.count <= 0 ) { cursor.ok = list.count == 0; cursor.count = 0; return cursor; } + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + if ( head->live != list.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + cursor.first = head->first; + cursor.ok = true; + return cursor; +} + +template +inline TableListCursor::Element> TableListElements( const TableArenaCtx & ctx, const TableList & list ) +{ + return TableListElements( *ctx.arena, list ); +} + +// ---- the LOAD side: where a decoded element lands (§2.9) ---- +// +// The same two shapes the map's fill takes, because the decoder above them +// cannot tell which it has: a REGION carves the element array out of the +// holder node's own extent, PRE-ORDER, and the TOOL's path appends into the +// builder's arena. +template struct TableListFill +{ + typedef typename TableList::Element Element; + TableList * list = NULL; + Element * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; + bool refused = false; // a count above the int32 cap on the tool's path: LoadBuilder answers NULL +}; + +template +inline TableListFill TableListFillBegin( const TableNodeMap & nodes, TableList & list, uint64_t n ) +{ + typedef typename TableList::Element Element; + TableListFill fill; + fill.list = &list; + list.elements.value = 0; + list.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( n > (uint64_t) INT32_MAX ) + { + // A COUNT ABOVE THE int32 STORAGE CAP (§2.2, §2.9): into a region it was + // refused by LoadMeasure before this ran, and into a builder it is the + // refusal LoadBuilder answers NULL for, moving no counter + fill.refused = nodes.carve->worker != NULL; + return fill; + } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Element ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Element ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Element *) base; + fill.capacity = (int32_t) n; + list.elements.value = (int64_t) ( base - (const uint8_t *) &list.elements ); + fill.ok = true; + return fill; +} + +// the next slot, at the element's declared defaults. NULL when the arena +// could not carve, which the decoder reports as framing damage +template inline typename TableList::Element * TableListFillNext( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count >= fill.capacity ) { return NULL; } + Element * element = fill.array + fill.list->count; + new ( element ) Element(); + fill.list->count++; + return element; + } + return TableListPlace( *fill.worker, *fill.list ); +} + +// A SLOT WHOSE ELEMENT NEVER LANDED is given back (§2.9, §4): the array keeps +// what it decoded, and an element whose own framing gave out before one byte +// of it decoded was not decoded. The region's form uncounts it, and the builder's +// marks it dead, which is what the storage rule allows mid-build. +template inline void TableListFillDrop( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count > 0 ) { fill.list->count--; } + return; + } + if ( fill.list->elements.value == 0 ) { return; } + TableListHead * head = (TableListHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.list->elements.value ); + if ( head->last.value == 0 ) { return; } + TableListSegment * segment = (TableListSegment *) TableArenaAt( *fill.worker->arena, (uint32_t) head->last.value ); + if ( segment->used <= 0 ) { return; } + const int32_t i = segment->used - 1; + if ( TableListSegmentDead( segment->dead, i ) ) { return; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + fill.list->count--; +} + +// an EMPTY list's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableListFillEnd( TableListFill & fill ) +{ + if ( fill.array != NULL && fill.list->count == 0 ) { fill.list->elements.value = 0; } +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.9, §6.5) ---- +// +// N x sizeof( T ) rounded to alignof( T ), AT EVERY DEPTH. N is framing and +// not a value, so this reads no field: it walks the list's own header and, +// where a table element holds a list or a map of its own, the elements' +// headers under it. Every -1 carries its REASON (§6.5): the int32 cap first, +// because a count past it cannot fit any body, and then the body's own L. +inline bool TableListWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t elem_size, int64_t elem_align, uint8_t elem_kind, int64_t elem_floor, + TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != elem_kind ) { return true; } // another element kind: §4's ordinary kind mismatch, the field reads empty + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; } // an N the list's L cannot carry + at = ( at + elem_align - 1 ) & ~( elem_align - 1 ); + at += (int64_t) n * elem_size; + if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_LIST + +#ifndef LISTDEMO_SCHEMA_BUILD_VERSION +#define LISTDEMO_SCHEMA_BUILD_VERSION + +namespace listdemo { + +// THE BUILD VERSION (docs/SPEC-TABLES.md §20): one digest over every fact the bytes +// this build produces depend on — the type wire's protocol id, every record's +// layout as the compiler's own C ABI model computes it, and the facts that +// decide what a load PUTS in those slots. It is the number a cook's header +// carries and the number Open compares, and the number a block's prologue +// carries and BlockOpen compares: a build version answers "which build?" and +// not "which form?", and what separates the two forms is their MAGIC. +// +// There are TWO ids in the design and they are not interchangeable: the +// PROTOCOL ID is the type wire's and nothing else, and the BUILD VERSION is +// what everything cooked or blocked is keyed by. A table edit moves this and +// never the protocol id; a type edit moves both. +static const uint64_t BuildVersion = 0x8d7c0edaca4571c7ull; + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_BUILD_VERSION + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK +#define LISTDEMO_SCHEMA_TABLE_COOK + +namespace listdemo { + +// ---- the cooked form (docs/SPEC-TABLES.md §7) ---- +// +// A cooked file is a HEADER, a DATA part and an ATTRIBUTION part, in that +// order. Every word of the header is a u64 written in the byte order the cook +// was produced in, and the header is 64 bytes: +// +// 0 magic 0x4b4f4f434d484353, read BYTEWISE before anything else +// 8 build_version the unit's id (docs/SPEC-TABLES.md §20) +// 16 byte_order 1 little, 2 big — the order that WROTE the file +// 24 data_length the region's bytes, rounded up to alignment +// 32 attribution_length the directory's bytes, or 0 +// 40 alignment the region's alignment, never below eight +// 48 reserved zero +// 56 reserved zero +// +// The DATA part is Lock's region written verbatim (§7.2) — the root at its +// base — and it is what a runtime points at. The ATTRIBUTION part is the node +// directory (§6.3), and NOTHING THAT READS THE STRUCTURE TOUCHES IT: it is +// written beside the data for schema cook-check, so a build that ships no +// tooling need not carry it at all. +static const int64_t kTableCookHeaderBytes = 64; + +// THE MAGIC'S VALUE, and a consumer written from the page needs the constant +// rather than a description of one. It is "SCHMCOOK" read as ASCII in the byte +// order a little-endian store produces — the same shape the block form's +// SCHMABLK takes, so a hex dump of a little-endian cook is legible and the two +// accelerators sit in one vocabulary. +// +// IT IS STORED IN THE PRODUCER'S ORDER, which is what makes it the byte-order +// check as well as the form check: a consumer reads back this build's +// constant, or that constant byte-reversed — which identifies a cook of the +// OTHER order — or something that is not a cook. All three answers but the +// first refuse, and a cook and a BLOCK are separated here too, because a +// form's identity belongs in its magic rather than in a second digest. +static const uint64_t TableCookMagic = 0x4b4f4f434d484353ull; + +// THIS BUILD's byte order, as the header's own word carries it. The magic is +// what REFUSES a foreign order; this word is what RECORDS which order wrote +// the file, so a refusal names the order rather than inferring it and a tool +// dumping a cook reads the fact. A file whose magic matched and whose order +// word did not is corrupt, and there is no reading that recovers it. +// +// The BUILD VERSION cannot do either job: §20.1 digests byteorder as a +// GENERATION input, little for every target schema generates for today, so +// two builds of one schema for two orders emit the same id. +#if defined( __BYTE_ORDER__ ) && defined( __ORDER_BIG_ENDIAN__ ) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__ +static const uint64_t TableCookByteOrder = 2; // big +#else +static const uint64_t TableCookByteOrder = 1; // little +#endif + +// The greatest region alignment a cooked file may name. The DATA part begins +// at align_up( 64, alignment ), which is 64 for every unit this language can +// declare — the largest alignment it has is sixteen — so a word past this cap +// describes a file no build of this schema wrote (docs/SPEC-TABLES.md §7.1). +static const uint64_t TableCookMaxAlign = 64; + +// The header read, BYTEWISE. memcpy is the portable spelling of "these eight +// bytes, in this machine's order"; every compiler this repo builds under folds +// it to one load, and it is the only read in the whole of Open that is not a +// comparison. +inline uint64_t table_cook_read64( const uint8_t * p ) +{ + uint64_t v; + memcpy( &v, p, sizeof( v ) ); + return v; +} + +// TableCookOpen: THE WHOLE CHECK, in one place, because §7 states the +// enumeration once and every generated Open is that one enumeration plus +// its own root's two layout facts. +// +// THE CHECK, in order: the magic read bytewise, the byte order it establishes, +// the build version against this build's own, both RESERVED words zero, the +// region alignment the header names, the two part lengths against the length +// the caller passed — a truncated file and a file with trailing bytes are the +// same refusal — the root's own storage inside the data part, and the +// alignment of the base. +// +// AND THAT IS ALL OF IT. On a match the bytes ARE what this build wrote, in +// this build's layout and this build's byte order, so there is nothing to +// validate and nothing to fix up: the caller gets the root. Nothing per node +// happens here, which is what makes open O(1) in the file's size; a walk of +// any shape would forfeit that, and validating an untrusted file is schema +// cook-check's job and a person's decision (§7.4). +// +// EVERY NUMBER BELOW COMES OUT OF THE FILE, so the arithmetic is unsigned and +// each term is BOUNDED BEFORE IT IS ADDED: a forged length near 2^64 must +// refuse, and an addition that wrapped would be the defect the comparison +// after it was supposed to catch. Nothing past length is read on any path, +// including every refusing one. +inline const uint8_t * TableCookOpen( const void * bytes, uint64_t length, uint64_t root_size, uint64_t root_align ) +{ + if ( bytes == NULL ) { return NULL; } + if ( length < (uint64_t) kTableCookHeaderBytes ) { return NULL; } + const uint8_t * raw = (const uint8_t *) bytes; + // the MAGIC, bytewise and first: it is what establishes the byte order + // every other header word is read in, so nothing else may be read before + // it. A byte-reversed constant is a cook of the other order and refuses + // here, which is why the order never reaches a fix-up pass. + if ( table_cook_read64( raw ) != TableCookMagic ) { return NULL; } + if ( table_cook_read64( raw + 16 ) != TableCookByteOrder ) { return NULL; } + if ( table_cook_read64( raw + 8 ) != BuildVersion ) { return NULL; } + // the RESERVED words: a non-zero one means a writer used a form this build + // does not understand, and Open refuses rather than ignoring it. + if ( table_cook_read64( raw + 48 ) != 0 ) { return NULL; } + if ( table_cook_read64( raw + 56 ) != 0 ) { return NULL; } + const uint64_t data_length = table_cook_read64( raw + 24 ); + const uint64_t attribution_length = table_cook_read64( raw + 32 ); + const uint64_t alignment = table_cook_read64( raw + 40 ); + // THE ALIGNMENT WORD IS DATA, and it is the one header field the rest of + // the check does arithmetic WITH rather than only comparison against. A + // region's alignment is a power of two, never below eight (the floor that + // puts the attribution part on an eight-byte boundary without a second + // padding rule) and never past the cap above; a word that is none of those + // rounds nothing and aligns nothing, so it is refused before it is used. + if ( alignment < 8 || alignment > TableCookMaxAlign ) { return NULL; } + if ( ( alignment & ( alignment - 1 ) ) != 0 ) { return NULL; } + // and it must be an alignment THE ROOT CAN SIT AT, since the root is at + // the region's base: both are powers of two, so "at least the root's" + // is one division. + if ( ( alignment % root_align ) != 0 ) { return NULL; } + // The DATA part begins at align_up( 64, alignment ). It is DERIVED and not + // a header field, because a fact a reader computes is a fact two writers + // cannot disagree about. + const uint64_t data_offset = ( (uint64_t) kTableCookHeaderBytes + alignment - 1 ) & ~( alignment - 1 ); + if ( length < data_offset ) { return NULL; } + // the two part lengths against the length the caller passed. The whole + // file is data_offset + data_length + attribution_length, and a length + // that is not EXACTLY that refuses — truncation and trailing bytes are one + // refusal, and both terms are subtracted rather than added so no sum can + // carry. + if ( data_length > length - data_offset ) { return NULL; } + if ( attribution_length != length - data_offset - data_length ) { return NULL; } + // the ROOT sits at the region's base, so the region has to hold it: a + // shorter data part describes a root partly outside the file, which is the + // one way a match-and-point reader could hand back storage it never + // received. + if ( data_length < root_size ) { return NULL; } + const uint8_t * base = raw + data_offset; + // the alignment of the BASE. The header pads the data part to the region's + // alignment, so a base an allocator or mmap gave you is already aligned — + // mmap gives page alignment for free — and a base that is not is a caller's + // buffer this form cannot be read out of. + if ( ( (uintptr_t) base % (uintptr_t) alignment ) != 0 ) { return NULL; } + return base; +} + +// ---- the cooked form, the WRITE side (docs/SPEC-TABLES.md §7.6) ---- +// +// THE BYTE ORDER IS THE TARGET'S, NOT THE HOST'S. A cook is produced in the +// byte order of the build that will read it (§7), so the fixing happens here — +// offline, once, on the writing side — and never at Open. Passing +// TableByteOrder::Big on a little-endian machine produces a big-endian build's +// file, and nothing about the writing host reaches the bytes. +enum class TableByteOrder +{ + Little = 1, // the header's byte_order word, and the order every scalar is written in + Big = 2, +}; + +// One store, width as an argument. Every call site passes a literal width, so +// the loop folds to a store (and a byte swap on the foreign order); a name per +// width would claim four §11 names to save nothing. +inline void table_cook_put( uint8_t * at, uint64_t value, int32_t width, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * i ) ); } + } + else + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * ( width - 1 - i ) ) ); } + } +} + +// A 128-bit store as two lanes: sixteen bytes, the low lane first in the +// little order and the high lane first — each lane big-endian — in the big +// order, exactly as a u64 is one lane of eight (docs/SPEC-TABLES.md §7.2). +inline void table_cook_put128( uint8_t * at, uint64_t lo, uint64_t hi, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) { table_cook_put( at, lo, 8, order ); table_cook_put( at + 8, hi, 8, order ); } + else { table_cook_put( at, hi, 8, order ); table_cook_put( at + 8, lo, 8, order ); } +} + +// A buffer piece: the USED bytes and nothing else. The tail is already zero — +// the whole extent was zeroed before any field was written — so this copies the +// used prefix and leaves the rest, which is what makes a string's unused tail a +// consequence of one memset rather than a rule per buffer. A used length past +// the buffer, or below zero, is a value no reader could have produced and it is +// clamped rather than trusted: this writes inside the caller's buffer on every +// input. +inline void table_cook_bytes( uint8_t * at, const void * source, int64_t used, int64_t capacity ) +{ + if ( used <= 0 ) { return; } + const int64_t n = used < capacity ? used : capacity; + memcpy( at, source, (size_t) n ); +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE +#define LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// ---- the cooked form's WRITE side for a POINTERED root (docs/SPEC-TABLES.md §7.6) ---- +// +// A pointered root's cook is the region of §7.2: every node the numbering +// reached (§3.1), once, at its own type's alignment, in index order, the root +// at offset zero. This is that region while it is being laid out and written — +// the tool's own Layout and Write, in one struct. +// +// The OFFSETS are one per node, the root's zero at position 0 and node index k +// at position k - 1, which is the directory's own order (§6.3); they are the +// one allocation the write makes beyond the numbering, and they go through the +// same pair. A measure needs no offsets and leaves the pointer NULL. +struct TableCookRegion +{ + const TableNumbering * numbering = NULL; // node -> index, from the walk that placed it + int64_t * offsets = NULL; // index - 1 -> the node's region offset; NULL while measuring + int64_t count = 0; // nodes, the root included + int64_t bytes = 0; // the data part's length, rounded to align + int64_t align = 0; // the region's alignment: the nodes' greatest, never below eight + uint8_t * base = NULL; // where the data part is being written; NULL while measuring +}; + +// A reference slot: the SELF-RELATIVE delta from the slot's own address to the +// node's start (§6.3), and zero for null. The node is found by the address the +// numbering keyed it under, which is the same address the walk resolved through +// the same context — so a reference the numbering does not carry is a slot the +// walk never reached (a counted array's slot past its count, an absent +// optional's value) holding a node the region will not hold, and it is refused +// rather than written as a delta to nowhere. +inline bool table_cook_ref( const TableCookRegion & region, uint8_t * at, const void * pointee, TableByteOrder order ) +{ + if ( pointee == NULL ) { table_cook_put( at, 0, 8, order ); return true; } + uint64_t index = 0; + if ( !TableNumberingIndex( *region.numbering, pointee, index ) ) { return false; } + if ( index == 0 || index > (uint64_t) region.count ) { return false; } + const int64_t delta = region.offsets[index - 1] - (int64_t) ( at - region.base ); + table_cook_put( at, (uint64_t) delta, 8, order ); + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// table Bytes — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Bytes { + TableList data; // uint8: the element array, empty until an Add + int32_t after = 0; +}; + +// table Ints — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Ints { + TableList values; // int32: the element array, empty until an Add + int32_t after = 0; +}; + +// table Floats — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Floats { + TableList values; // float32: the element array, empty until an Add + int32_t after = 0; +}; + +// ---- prefill: the declared defaults, in place (docs/SPEC-TABLES.md) ---- + +inline void BytesReset( Bytes & value ); +inline void IntsReset( Ints & value ); +inline void FloatsReset( Floats & value ); + +inline void BytesReset( Bytes & value ) +{ + value.data.elements.value = 0; // uint8: empty + value.data.count = 0; + value.data.padding = 0; + value.after = 0; +} + +inline void IntsReset( Ints & value ) +{ + value.values.elements.value = 0; // int32: empty + value.values.count = 0; + value.values.padding = 0; + value.after = 0; +} + +inline void FloatsReset( Floats & value ) +{ + value.values.elements.value = 0; // float32: empty + value.values.count = 0; + value.values.padding = 0; + value.after = 0; +} + +// ---- the arena's reset hook (docs/SPEC-TABLES.md §6) ---- +// +// TableWorker::Alloc is a template and cannot name a member's Reset, so +// the arena reaches it through this overload set by argument-dependent +// lookup. It is how a node born in raw arena storage comes to hold the +// declared defaults without value-initialising the whole aggregate. + +inline void TableReset( Bytes & value ) { BytesReset( value ); } +inline void TableReset( Ints & value ) { IntsReset( value ); } +inline void TableReset( Floats & value ) { FloatsReset( value ); } + +// ---- pointer targets: allocation and resolution (docs/SPEC-TABLES.md §2) ---- +// +// A reference resolves differently in the two forms, and the CONTEXT says +// which: in the arena it is an offset; in a region it is a self-relative +// delta, so the const deref below is one add and needs no base pointer. + +// ---- codecs: measure/save/load per closure member ---- + +template inline int64_t BytesMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Bytes & value ); +template inline bool BytesSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Bytes & value ); +template inline bool BytesSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Bytes & value ); +inline bool BytesLoadBody( TableReader & r, const TableNodeMap & nodes, Bytes & value ); +template inline int64_t IntsMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Ints & value ); +template inline bool IntsSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Ints & value ); +template inline bool IntsSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Ints & value ); +inline bool IntsLoadBody( TableReader & r, const TableNodeMap & nodes, Ints & value ); +template inline int64_t FloatsMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Floats & value ); +template inline bool FloatsSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Floats & value ); +template inline bool FloatsSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Floats & value ); +inline bool FloatsLoadBody( TableReader & r, const TableNodeMap & nodes, Floats & value ); + +// ---- pointer-graph walkers: number (measure/save), pack (Lock) ---- + +template inline bool BytesNumber( const Ctx & ctx, TableNumbering & numbering, const Bytes & value ); +template inline int64_t BytesPackMeasure( const Ctx & ctx, TablePackMap & seen, const Bytes & value ); +template inline bool BytesPack( const Ctx & ctx, TablePackMap & seen, const Bytes & src, Bytes & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool IntsNumber( const Ctx & ctx, TableNumbering & numbering, const Ints & value ); +template inline int64_t IntsPackMeasure( const Ctx & ctx, TablePackMap & seen, const Ints & value ); +template inline bool IntsPack( const Ctx & ctx, TablePackMap & seen, const Ints & src, Ints & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool FloatsNumber( const Ctx & ctx, TableNumbering & numbering, const Floats & value ); +template inline int64_t FloatsPackMeasure( const Ctx & ctx, TablePackMap & seen, const Floats & value ); +template inline bool FloatsPack( const Ctx & ctx, TablePackMap & seen, const Floats & src, Floats & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +// ---- the numbering's bridge to each member's codec (docs/SPEC-TABLES.md §3.1) ---- + +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Bytes & value ) { return BytesMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Bytes & value ) { return BytesSaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Ints & value ) { return IntsMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Ints & value ) { return IntsSaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Floats & value ) { return FloatsMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Floats & value ) { return FloatsSaveBody( ctx, numbering, w, ids, value ); } + +template +inline int64_t BytesMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Bytes & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // data: a kind 14 array of kind 6 elements, INDEX order (§2.9) + TableListCursor cursor_data = TableListElements( ctx, value.data ); + if ( !cursor_data.ok ) { return -1; } // the slot and the head disagree + if ( cursor_data.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_data = ids.ref( 0x855b556730a34a05ull, 8 ); + int64_t body_data = 0; + body_data += 1 + TableLebBytes( (uint64_t) ( cursor_data.count ) ); // the element kind byte and the count + body_data += (int64_t) ( cursor_data.count ) * 1; + bytes += TableLebBytes( ref_data ) + 1 + TableLebBytes( (uint64_t) ( body_data ) ) + ( body_data ); + } + } + if ( value.after != 0 ) { bytes += TableLebBytes( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ) + 1 + 4; } // after + return bytes; +} + +template +inline bool BytesSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Bytes & value ) +{ + (void) ctx; (void) numbering; + { + TableListCursor cursor_data = TableListElements( ctx, value.data ); // data + if ( !cursor_data.ok ) { return false; } + if ( cursor_data.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_data = ids.ref( 0x855b556730a34a05ull, 8 ); + int64_t body_data = 0; + body_data += 1 + TableLebBytes( (uint64_t) ( cursor_data.count ) ); // the element kind byte and the count + body_data += (int64_t) ( cursor_data.count ) * 1; + w.putleb( ref_data ); w.put8( 14 ); w.putleb( (uint64_t) body_data ); // data + w.put8( 6 ); w.putleb( (uint64_t) ( cursor_data.count ) ); + for ( int32_t elem_i_data = 0; elem_i_data < cursor_data.count; elem_i_data++ ) + { + w.put8( uint8_t( cursor_data[elem_i_data] ) ); + } + } + } + if ( value.after != 0 ) + { + w.putleb( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ); w.put8( 4 ); // after + w.put32( uint32_t( value.after ) ); + } + return !w.overflow; +} + +template +inline bool BytesSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Bytes & value ) +{ + if ( !BytesSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool BytesLoadBody( TableReader & r, const TableNodeMap & nodes, Bytes & value ) +{ + (void) nodes; + BytesReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x855b556730a34a05ull: // data + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 6 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.data, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + uint8_t * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + if ( !sub.has( 1 ) ) { r.report->malformed = true; break; } + uint8_t decoded_v = uint8_t( sub.get8( ) ); + ( *slot ) = decoded_v; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xbf82010f6f71eae9ull: // after + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.after = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +template +inline int64_t IntsMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Ints & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // values: a kind 14 array of kind 4 elements, INDEX order (§2.9) + TableListCursor cursor_values = TableListElements( ctx, value.values ); + if ( !cursor_values.ok ) { return -1; } // the slot and the head disagree + if ( cursor_values.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_values = ids.ref( 0x21277bcf1a4d67fbull, 10 ); + int64_t body_values = 0; + body_values += 1 + TableLebBytes( (uint64_t) ( cursor_values.count ) ); // the element kind byte and the count + body_values += (int64_t) ( cursor_values.count ) * 4; + bytes += TableLebBytes( ref_values ) + 1 + TableLebBytes( (uint64_t) ( body_values ) ) + ( body_values ); + } + } + if ( value.after != 0 ) { bytes += TableLebBytes( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ) + 1 + 4; } // after + return bytes; +} + +template +inline bool IntsSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Ints & value ) +{ + (void) ctx; (void) numbering; + { + TableListCursor cursor_values = TableListElements( ctx, value.values ); // values + if ( !cursor_values.ok ) { return false; } + if ( cursor_values.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_values = ids.ref( 0x21277bcf1a4d67fbull, 10 ); + int64_t body_values = 0; + body_values += 1 + TableLebBytes( (uint64_t) ( cursor_values.count ) ); // the element kind byte and the count + body_values += (int64_t) ( cursor_values.count ) * 4; + w.putleb( ref_values ); w.put8( 14 ); w.putleb( (uint64_t) body_values ); // values + w.put8( 4 ); w.putleb( (uint64_t) ( cursor_values.count ) ); + for ( int32_t elem_i_values = 0; elem_i_values < cursor_values.count; elem_i_values++ ) + { + w.put32( uint32_t( cursor_values[elem_i_values] ) ); + } + } + } + if ( value.after != 0 ) + { + w.putleb( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ); w.put8( 4 ); // after + w.put32( uint32_t( value.after ) ); + } + return !w.overflow; +} + +template +inline bool IntsSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Ints & value ) +{ + if ( !IntsSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool IntsLoadBody( TableReader & r, const TableNodeMap & nodes, Ints & value ) +{ + (void) nodes; + IntsReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x21277bcf1a4d67fbull: // values + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 4 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.values, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + int32_t * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + if ( !sub.has( 4 ) ) { r.report->malformed = true; break; } + int32_t decoded_v = int32_t( sub.get32( ) ); + ( *slot ) = decoded_v; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xbf82010f6f71eae9ull: // after + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.after = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +template +inline int64_t FloatsMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Floats & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // values: a kind 14 array of kind 10 elements, INDEX order (§2.9) + TableListCursor cursor_values = TableListElements( ctx, value.values ); + if ( !cursor_values.ok ) { return -1; } // the slot and the head disagree + if ( cursor_values.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_values = ids.ref( 0x21277bcf1a4d67fbull, 10 ); + int64_t body_values = 0; + body_values += 1 + TableLebBytes( (uint64_t) ( cursor_values.count ) ); // the element kind byte and the count + body_values += (int64_t) ( cursor_values.count ) * 4; + bytes += TableLebBytes( ref_values ) + 1 + TableLebBytes( (uint64_t) ( body_values ) ) + ( body_values ); + } + } + if ( value.after != 0 ) { bytes += TableLebBytes( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ) + 1 + 4; } // after + return bytes; +} + +template +inline bool FloatsSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Floats & value ) +{ + (void) ctx; (void) numbering; + { + TableListCursor cursor_values = TableListElements( ctx, value.values ); // values + if ( !cursor_values.ok ) { return false; } + if ( cursor_values.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_values = ids.ref( 0x21277bcf1a4d67fbull, 10 ); + int64_t body_values = 0; + body_values += 1 + TableLebBytes( (uint64_t) ( cursor_values.count ) ); // the element kind byte and the count + body_values += (int64_t) ( cursor_values.count ) * 4; + w.putleb( ref_values ); w.put8( 14 ); w.putleb( (uint64_t) body_values ); // values + w.put8( 10 ); w.putleb( (uint64_t) ( cursor_values.count ) ); + for ( int32_t elem_i_values = 0; elem_i_values < cursor_values.count; elem_i_values++ ) + { + w.put32( table_float_to_bits( cursor_values[elem_i_values] ) ); + } + } + } + if ( value.after != 0 ) + { + w.putleb( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ); w.put8( 4 ); // after + w.put32( uint32_t( value.after ) ); + } + return !w.overflow; +} + +template +inline bool FloatsSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Floats & value ) +{ + if ( !FloatsSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool FloatsLoadBody( TableReader & r, const TableNodeMap & nodes, Floats & value ) +{ + (void) nodes; + FloatsReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x21277bcf1a4d67fbull: // values + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 10 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.values, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + float * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + if ( !sub.has( 4 ) ) { r.report->malformed = true; break; } + ( *slot ) = table_bits_to_float( sub.get32() ); + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xbf82010f6f71eae9ull: // after + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.after = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// BytesWireExtent: the extent Bytes's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool BytesWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x855b556730a34a05ull && field_kind == 14 ) // data: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( uint8_t ), (int64_t) alignof( uint8_t ), 6, 1, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// BytesExtentAt: the node extent Bytes's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as BytesExtentPack advances it (§2.8, §2.9). +template +inline bool BytesExtentAt( const Ctx & ctx, const Bytes & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.data ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( uint8_t ) - 1 ) & ~( (int64_t) alignof( uint8_t ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( uint8_t ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t BytesExtent( const Ctx & ctx, const Bytes & value ) +{ + int64_t at = 0; + if ( !BytesExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// BytesExtentPack: carve Bytes's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset BytesExtentAt advances (§2.8, §2.9). +template +inline bool BytesExtentPack( const Ctx & ctx, const Bytes & src, Bytes & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.data ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( uint8_t ) - 1 ) & ~( (int64_t) alignof( uint8_t ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( uint8_t ); + if ( at + bytes > capacity ) { return false; } + uint8_t * placed = (uint8_t *) ( extent + at ); + at += bytes; + dst.data.count = cursor.count; + dst.data.padding = 0; + dst.data.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.data.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( uint8_t ) ); // trivially copyable, by construction + } + } + return true; +} + +// IntsWireExtent: the extent Ints's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool IntsWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x21277bcf1a4d67fbull && field_kind == 14 ) // values: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( int32_t ), (int64_t) alignof( int32_t ), 4, 4, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// IntsExtentAt: the node extent Ints's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as IntsExtentPack advances it (§2.8, §2.9). +template +inline bool IntsExtentAt( const Ctx & ctx, const Ints & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.values ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( int32_t ) - 1 ) & ~( (int64_t) alignof( int32_t ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( int32_t ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t IntsExtent( const Ctx & ctx, const Ints & value ) +{ + int64_t at = 0; + if ( !IntsExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// IntsExtentPack: carve Ints's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset IntsExtentAt advances (§2.8, §2.9). +template +inline bool IntsExtentPack( const Ctx & ctx, const Ints & src, Ints & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.values ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( int32_t ) - 1 ) & ~( (int64_t) alignof( int32_t ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( int32_t ); + if ( at + bytes > capacity ) { return false; } + int32_t * placed = (int32_t *) ( extent + at ); + at += bytes; + dst.values.count = cursor.count; + dst.values.padding = 0; + dst.values.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.values.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( int32_t ) ); // trivially copyable, by construction + } + } + return true; +} + +// FloatsWireExtent: the extent Floats's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool FloatsWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x21277bcf1a4d67fbull && field_kind == 14 ) // values: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( float ), (int64_t) alignof( float ), 10, 4, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// FloatsExtentAt: the node extent Floats's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as FloatsExtentPack advances it (§2.8, §2.9). +template +inline bool FloatsExtentAt( const Ctx & ctx, const Floats & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.values ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( float ) - 1 ) & ~( (int64_t) alignof( float ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( float ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t FloatsExtent( const Ctx & ctx, const Floats & value ) +{ + int64_t at = 0; + if ( !FloatsExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// FloatsExtentPack: carve Floats's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset FloatsExtentAt advances (§2.8, §2.9). +template +inline bool FloatsExtentPack( const Ctx & ctx, const Floats & src, Floats & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.values ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( float ) - 1 ) & ~( (int64_t) alignof( float ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( float ); + if ( at + bytes > capacity ) { return false; } + float * placed = (float *) ( extent + at ); + at += bytes; + dst.values.count = cursor.count; + dst.values.padding = 0; + dst.values.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.values.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( float ) ); // trivially copyable, by construction + } + } + return true; +} + +// ---- Bytes.data: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline uint8_t * BytesDataAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool BytesDataErase( TableArena & arena, TableList & list, const uint8_t * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach BytesDataEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Ints.values: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline int32_t * IntsValuesAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool IntsValuesErase( TableArena & arena, TableList & list, const int32_t * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach IntsValuesEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Floats.values: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline float * FloatsValuesAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool FloatsValuesErase( TableArena & arena, TableList & list, const float * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach FloatsValuesEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// BytesNumber: number everything Bytes POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool BytesNumber( const Ctx & ctx, TableNumbering & numbering, const Bytes & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// BytesPackMeasure: the packed region bytes of everything Bytes POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t BytesPackMeasure( const Ctx & ctx, TablePackMap & seen, const Bytes & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// BytesPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool BytesPackEdges( const Ctx & ctx, TablePackMap & seen, const Bytes & src, Bytes & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool BytesPack( const Ctx & ctx, TablePackMap & seen, const Bytes & src, Bytes & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Bytes ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Bytes ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !BytesExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return BytesPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool BytesPackEdges( const Ctx & ctx, TablePackMap & seen, const Bytes & src, Bytes & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// IntsNumber: number everything Ints POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool IntsNumber( const Ctx & ctx, TableNumbering & numbering, const Ints & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// IntsPackMeasure: the packed region bytes of everything Ints POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t IntsPackMeasure( const Ctx & ctx, TablePackMap & seen, const Ints & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// IntsPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool IntsPackEdges( const Ctx & ctx, TablePackMap & seen, const Ints & src, Ints & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool IntsPack( const Ctx & ctx, TablePackMap & seen, const Ints & src, Ints & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Ints ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Ints ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !IntsExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return IntsPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool IntsPackEdges( const Ctx & ctx, TablePackMap & seen, const Ints & src, Ints & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// FloatsNumber: number everything Floats POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool FloatsNumber( const Ctx & ctx, TableNumbering & numbering, const Floats & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// FloatsPackMeasure: the packed region bytes of everything Floats POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t FloatsPackMeasure( const Ctx & ctx, TablePackMap & seen, const Floats & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// FloatsPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool FloatsPackEdges( const Ctx & ctx, TablePackMap & seen, const Floats & src, Floats & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool FloatsPack( const Ctx & ctx, TablePackMap & seen, const Floats & src, Floats & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Floats ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Floats ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !FloatsExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return FloatsPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool FloatsPackEdges( const Ctx & ctx, TablePackMap & seen, const Floats & src, Floats & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// ---- Bytes: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: BytesBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Bytes is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct BytesBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + BytesBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~BytesBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + BytesBuilder( const BytesBuilder & ) = delete; + BytesBuilder & operator=( const BytesBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Bytes * GetRoot() { return arena.locked ? NULL : (Bytes *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Bytes * AsConst() const { return (const Bytes *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool BytesBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Bytes & root = *(const Bytes *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = BytesPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = BytesExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + Bytes * destination = new ( packed ) Bytes; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !BytesPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Bytes on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// BytesNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t BytesNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// BytesNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void BytesNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// BytesNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t BytesNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// BytesNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t BytesNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// BytesNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void BytesNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = BytesNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? BytesNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool BytesNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Bytes & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return BytesNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t BytesMeasureWire( const Ctx & ctx, const Bytes & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( BytesNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = BytesMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t BytesSaveWire( const Ctx & ctx, const Bytes & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !BytesNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = BytesSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == BytesMeasure( root ) +} + +inline int64_t BytesMeasure( const Bytes * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return BytesMeasureWire( ctx, *root, allocator ); +} + +inline int64_t BytesSave( const Bytes * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return BytesSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t BytesMeasure( const BytesBuilder & builder ) +{ + if ( builder.region != NULL ) { return BytesMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return BytesMeasureWire( ctx, *(const Bytes *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t BytesSave( const BytesBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return BytesSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return BytesSaveWire( ctx, *(const Bytes *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t BytesMeasureMessage( const Bytes * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return BytesMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t BytesSaveMessage( const Bytes * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return BytesSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t BytesMeasureMessage( const BytesBuilder & builder ) +{ + if ( builder.region != NULL ) { return BytesMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return BytesMeasureWire( ctx, *(const Bytes *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t BytesSaveMessage( const BytesBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return BytesSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return BytesSaveWire( ctx, *(const Bytes *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// BytesLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t BytesLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !BytesWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = BytesNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// BytesLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Bytes * BytesLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Bytes ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !BytesWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = BytesNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xeeeea7adc131a244ull; + Bytes * root = new ( region ) Bytes; // lifetime only: LoadBody's first act is BytesReset + BytesReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = BytesNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + BytesNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + BytesNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Bytes ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + BytesLoadBody( r, nodes, *root ); + return root; +} + +// BytesLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t BytesLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !BytesWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = BytesNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// BytesLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Bytes * BytesLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Bytes ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !BytesWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = BytesNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xeeeea7adc131a244ull; + Bytes * root = new ( region ) Bytes; // lifetime only: LoadBody's first act is BytesReset + BytesReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = BytesNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + BytesNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + BytesNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Bytes ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + BytesLoadBody( r, nodes, *root ); + return root; +} + +// BytesLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool BytesLoadBuilder( BytesBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Bytes * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0xeeeea7adc131a244ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = BytesNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + BytesNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = BytesLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Ints: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: IntsBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Ints is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct IntsBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + IntsBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~IntsBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + IntsBuilder( const IntsBuilder & ) = delete; + IntsBuilder & operator=( const IntsBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Ints * GetRoot() { return arena.locked ? NULL : (Ints *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Ints * AsConst() const { return (const Ints *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool IntsBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Ints & root = *(const Ints *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = IntsPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = IntsExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + Ints * destination = new ( packed ) Ints; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !IntsPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Ints on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// IntsNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t IntsNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// IntsNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void IntsNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// IntsNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t IntsNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// IntsNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t IntsNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// IntsNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void IntsNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = IntsNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? IntsNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool IntsNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Ints & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return IntsNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t IntsMeasureWire( const Ctx & ctx, const Ints & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( IntsNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = IntsMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t IntsSaveWire( const Ctx & ctx, const Ints & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !IntsNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = IntsSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == IntsMeasure( root ) +} + +inline int64_t IntsMeasure( const Ints * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return IntsMeasureWire( ctx, *root, allocator ); +} + +inline int64_t IntsSave( const Ints * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return IntsSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t IntsMeasure( const IntsBuilder & builder ) +{ + if ( builder.region != NULL ) { return IntsMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return IntsMeasureWire( ctx, *(const Ints *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t IntsSave( const IntsBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return IntsSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return IntsSaveWire( ctx, *(const Ints *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t IntsMeasureMessage( const Ints * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return IntsMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t IntsSaveMessage( const Ints * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return IntsSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t IntsMeasureMessage( const IntsBuilder & builder ) +{ + if ( builder.region != NULL ) { return IntsMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return IntsMeasureWire( ctx, *(const Ints *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t IntsSaveMessage( const IntsBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return IntsSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return IntsSaveWire( ctx, *(const Ints *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// IntsLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t IntsLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !IntsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = IntsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// IntsLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Ints * IntsLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Ints ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !IntsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = IntsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x2034c5d17c00ceb7ull; + Ints * root = new ( region ) Ints; // lifetime only: LoadBody's first act is IntsReset + IntsReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = IntsNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + IntsNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + IntsNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Ints ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + IntsLoadBody( r, nodes, *root ); + return root; +} + +// IntsLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t IntsLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !IntsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = IntsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// IntsLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Ints * IntsLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Ints ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !IntsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = IntsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x2034c5d17c00ceb7ull; + Ints * root = new ( region ) Ints; // lifetime only: LoadBody's first act is IntsReset + IntsReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = IntsNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + IntsNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + IntsNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Ints ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + IntsLoadBody( r, nodes, *root ); + return root; +} + +// IntsLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool IntsLoadBuilder( IntsBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Ints * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0x2034c5d17c00ceb7ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = IntsNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + IntsNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = IntsLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Floats: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: FloatsBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Floats is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct FloatsBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + FloatsBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~FloatsBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + FloatsBuilder( const FloatsBuilder & ) = delete; + FloatsBuilder & operator=( const FloatsBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Floats * GetRoot() { return arena.locked ? NULL : (Floats *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Floats * AsConst() const { return (const Floats *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool FloatsBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Floats & root = *(const Floats *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = FloatsPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = FloatsExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + Floats * destination = new ( packed ) Floats; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !FloatsPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Floats on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// FloatsNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t FloatsNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// FloatsNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void FloatsNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// FloatsNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t FloatsNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// FloatsNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t FloatsNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// FloatsNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void FloatsNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = FloatsNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? FloatsNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool FloatsNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Floats & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return FloatsNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t FloatsMeasureWire( const Ctx & ctx, const Floats & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( FloatsNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = FloatsMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t FloatsSaveWire( const Ctx & ctx, const Floats & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !FloatsNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = FloatsSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == FloatsMeasure( root ) +} + +inline int64_t FloatsMeasure( const Floats * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return FloatsMeasureWire( ctx, *root, allocator ); +} + +inline int64_t FloatsSave( const Floats * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return FloatsSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t FloatsMeasure( const FloatsBuilder & builder ) +{ + if ( builder.region != NULL ) { return FloatsMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return FloatsMeasureWire( ctx, *(const Floats *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t FloatsSave( const FloatsBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return FloatsSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return FloatsSaveWire( ctx, *(const Floats *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t FloatsMeasureMessage( const Floats * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return FloatsMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t FloatsSaveMessage( const Floats * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return FloatsSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t FloatsMeasureMessage( const FloatsBuilder & builder ) +{ + if ( builder.region != NULL ) { return FloatsMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return FloatsMeasureWire( ctx, *(const Floats *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t FloatsSaveMessage( const FloatsBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return FloatsSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return FloatsSaveWire( ctx, *(const Floats *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// FloatsLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t FloatsLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !FloatsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = FloatsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// FloatsLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Floats * FloatsLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Floats ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !FloatsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = FloatsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x91638659f8f6ad42ull; + Floats * root = new ( region ) Floats; // lifetime only: LoadBody's first act is FloatsReset + FloatsReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = FloatsNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + FloatsNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + FloatsNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Floats ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + FloatsLoadBody( r, nodes, *root ); + return root; +} + +// FloatsLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t FloatsLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !FloatsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = FloatsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// FloatsLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Floats * FloatsLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Floats ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !FloatsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = FloatsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x91638659f8f6ad42ull; + Floats * root = new ( region ) Floats; // lifetime only: LoadBody's first act is FloatsReset + FloatsReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = FloatsNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + FloatsNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + FloatsNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Floats ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + FloatsLoadBody( r, nodes, *root ); + return root; +} + +// FloatsLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool FloatsLoadBuilder( FloatsBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Floats * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0x91638659f8f6ad42ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = FloatsNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + FloatsNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = FloatsLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- the cooked form: point at a cook (docs/SPEC-TABLES.md §7) ---- + +// BytesOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH BytesAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Bytes * BytesOpen( const void * bytes, uint64_t length ) +{ + return (const Bytes *) TableCookOpen( bytes, length, (uint64_t) sizeof( Bytes ), (uint64_t) alignof( Bytes ) ); +} + +// IntsOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH IntsAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Ints * IntsOpen( const void * bytes, uint64_t length ) +{ + return (const Ints *) TableCookOpen( bytes, length, (uint64_t) sizeof( Ints ), (uint64_t) alignof( Ints ) ); +} + +// FloatsOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH FloatsAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Floats * FloatsOpen( const void * bytes, uint64_t length ) +{ + return (const Floats *) TableCookOpen( bytes, length, (uint64_t) sizeof( Floats ), (uint64_t) alignof( Floats ) ); +} + +// ---- the cooked form: WRITE a cook (docs/SPEC-TABLES.md §7.6) ---- +// +// The bytes are `schema cook`'s, and the tool stays the reference: the two +// writers are held to one file, byte for byte, in both byte orders. A cook is +// content-addressed by (asset hash, build version), so two writers of one +// instance produce ONE artifact or the pair means nothing. + +template inline bool BytesCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Bytes & value, TableByteOrder order ); +template inline bool IntsCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Ints & value, TableByteOrder order ); +template inline bool FloatsCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Floats & value, TableByteOrder order ); + +template inline bool BytesCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Bytes & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // data: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.after, 4, order ); + return true; +} + +template inline bool IntsCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Ints & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // values: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.after, 4, order ); + return true; +} + +template inline bool FloatsCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Floats & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // values: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.after, 4, order ); + return true; +} + +template inline bool BytesCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Bytes & value, TableByteOrder order ); +template inline bool IntsCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Ints & value, TableByteOrder order ); +template inline bool FloatsCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Floats & value, TableByteOrder order ); + +// BytesCookExtent: Bytes's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool BytesCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Bytes & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // data: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.data ); + if ( !cursor.ok ) { return false; } + at = ( at + 0 ) & ~(int64_t) 0; // at alignof( uint8_t ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 1; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + table_cook_put( array + i * 1, (uint64_t) cursor[i], 1, order ); + } + } + return true; +} + +// IntsCookExtent: Ints's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool IntsCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Ints & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // values: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.values ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( int32_t ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 4; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + table_cook_put( array + i * 4, (uint64_t) cursor[i], 4, order ); + } + } + return true; +} + +// FloatsCookExtent: Floats's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FloatsCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Floats & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // values: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.values ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( float ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 4; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + { uint32_t bits = 0; memcpy( &bits, &cursor[i], 4 ); table_cook_put( array + i * 4, (uint64_t) bits, 4, order ); } + } + } + return true; +} + +// BytesCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool BytesCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Bytes & value, TableByteOrder order ) +{ + if ( !BytesCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return BytesCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// IntsCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool IntsCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Ints & value, TableByteOrder order ) +{ + if ( !IntsCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return IntsCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// FloatsCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool FloatsCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Floats & value, TableByteOrder order ) +{ + if ( !FloatsCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return FloatsCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// BytesCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool BytesCookLayout( const Ctx & ctx, const Bytes & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = BytesExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// BytesCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t BytesCookMeasureFrom( const Ctx & ctx, const Bytes & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( BytesNumberFrom( ctx, numbering, root ) && BytesCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// BytesCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool BytesCookFrom( const Ctx & ctx, const Bytes & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = BytesNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && BytesCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = BytesCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0xeeeea7adc131a244ull, 8, order ); // the root: fnv1a64( "Bytes" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// BytesCookMeasure / BytesCook over a REGION root — a locked builder's AsConst, a +// region BytesLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t BytesCookMeasure( const Bytes * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return BytesCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool BytesCook( const Bytes * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return BytesCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t BytesCookMeasure( const BytesBuilder & builder ) +{ + if ( builder.region != NULL ) { return BytesCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return BytesCookMeasureFrom( ctx, *(const Bytes *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool BytesCook( const BytesBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return BytesCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return BytesCookFrom( ctx, *(const Bytes *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// IntsCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool IntsCookLayout( const Ctx & ctx, const Ints & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = IntsExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// IntsCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t IntsCookMeasureFrom( const Ctx & ctx, const Ints & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( IntsNumberFrom( ctx, numbering, root ) && IntsCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// IntsCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool IntsCookFrom( const Ctx & ctx, const Ints & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = IntsNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && IntsCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = IntsCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0x2034c5d17c00ceb7ull, 8, order ); // the root: fnv1a64( "Ints" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// IntsCookMeasure / IntsCook over a REGION root — a locked builder's AsConst, a +// region IntsLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t IntsCookMeasure( const Ints * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return IntsCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool IntsCook( const Ints * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return IntsCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t IntsCookMeasure( const IntsBuilder & builder ) +{ + if ( builder.region != NULL ) { return IntsCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return IntsCookMeasureFrom( ctx, *(const Ints *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool IntsCook( const IntsBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return IntsCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return IntsCookFrom( ctx, *(const Ints *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// FloatsCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool FloatsCookLayout( const Ctx & ctx, const Floats & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = FloatsExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// FloatsCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t FloatsCookMeasureFrom( const Ctx & ctx, const Floats & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( FloatsNumberFrom( ctx, numbering, root ) && FloatsCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// FloatsCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool FloatsCookFrom( const Ctx & ctx, const Floats & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = FloatsNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && FloatsCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = FloatsCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0x91638659f8f6ad42ull, 8, order ); // the root: fnv1a64( "Floats" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// FloatsCookMeasure / FloatsCook over a REGION root — a locked builder's AsConst, a +// region FloatsLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t FloatsCookMeasure( const Floats * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return FloatsCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool FloatsCook( const Floats * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return FloatsCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t FloatsCookMeasure( const FloatsBuilder & builder ) +{ + if ( builder.region != NULL ) { return FloatsCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return FloatsCookMeasureFrom( ctx, *(const Floats *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool FloatsCook( const FloatsBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return FloatsCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return FloatsCookFrom( ctx, *(const Floats *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ---- relocatability, enforced: the wire is a pure length-prefixed +// stream AND the decoded storage is pointer-free — every closure type +// must stay trivially copyable and standard-layout, so instances can be +// memcpy'd, mmap'd, shared across processes, and walked through +// descriptor offsets. A failure here means a pointer, virtual or +// non-trivial member crept into generated storage. +// +// They ask the COMPILER ITSELF, which is what every C++ standard library +// answers the same two questions with — and it costs this header no +// include at all. +// A pointer FIELD is a TableRef — eight bytes and no address — so the +// property holds in BOTH forms: a fixed-size table is one relocatable +// struct, and a packed region is one relocatable block whose references +// are self-relative and therefore survive a plain memcpy. +static_assert( __is_trivially_copyable( Bytes ), "Bytes must stay relocatable" ); +static_assert( __is_standard_layout( Bytes ), "Bytes must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Ints ), "Ints must stay relocatable" ); +static_assert( __is_standard_layout( Ints ), "Ints must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Floats ), "Floats must stay relocatable" ); +static_assert( __is_standard_layout( Floats ), "Floats must stay standard-layout for offsetof" ); + +// ---- the cook's layout contract (docs/SPEC-TABLES.md §20.3) ---- +// +// The compiler derived every number below from the declaration and folded it +// into the BUILD VERSION; these asserts are this compiler saying whether it +// agrees. The model is not self-evidently right — on 32-bit System V +// alignof(uint64_t) is 4, not 8 — which is precisely why it is asserted +// rather than assumed. +static_assert( sizeof( Bytes ) == 24, "Bytes's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Bytes ) == 8, "Bytes's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Bytes, data ) == 0, "Bytes's field data moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Bytes, after ) == 16, "Bytes's field after moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Ints ) == 24, "Ints's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Ints ) == 8, "Ints's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Ints, values ) == 0, "Ints's field values moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Ints, after ) == 16, "Ints's field after moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Floats ) == 24, "Floats's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Floats ) == 8, "Floats's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Floats, values ) == 0, "Floats's field values moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Floats, after ) == 16, "Floats's field after moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); + +static_assert( alignof( uint8_t ) <= kTableAlign, "Bytes.data: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( int32_t ) <= kTableAlign, "Ints.values: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( float ) <= kTableAlign, "Floats.values: an unbounded array's element alignment must fit the arena's" ); + +// ---- reflection descriptors (tables only, docs/SPEC-TABLES.md) ---- + +inline const TableTypeInfo * BytesTableType(); +inline const TableTypeInfo * IntsTableType(); +inline const TableTypeInfo * FloatsTableType(); +// The descriptors are CONSTANT-INITIALISED data, and a field's target is +// the ADDRESS of another descriptor. These declarations are what let a +// self- or mutually-referential graph — Node naming itself through *Node — +// be expressed as constant data instead of a lazy link, which could not +// have been written race-free OR recursion-safe. The whole reflection +// surface is therefore immutable: read it from any thread, any time. +extern const TableTypeInfo BytesTableInfo; +extern const TableTypeInfo IntsTableInfo; +extern const TableTypeInfo FloatsTableInfo; + +inline const TableFieldInfo BytesTableFields[] = { + { "data", "data", "uint8", 0x855b556730a34a05ull, 6, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Bytes, data ), (uint32_t) sizeof( uint8_t ), (uint32_t) offsetof( Bytes, data.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Bytes, after ), (uint32_t) sizeof( Bytes::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo BytesTableInfo = { "Bytes", (uint32_t) sizeof( Bytes ), 2, BytesTableFields, +[]( void * p ) { BytesReset( *(Bytes *) p ); }, true }; +inline const TableTypeInfo * BytesTableType() { return &BytesTableInfo; } + +inline const TableFieldInfo IntsTableFields[] = { + { "values", "values", "int32", 0x21277bcf1a4d67fbull, 4, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Ints, values ), (uint32_t) sizeof( int32_t ), (uint32_t) offsetof( Ints, values.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Ints, after ), (uint32_t) sizeof( Ints::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo IntsTableInfo = { "Ints", (uint32_t) sizeof( Ints ), 2, IntsTableFields, +[]( void * p ) { IntsReset( *(Ints *) p ); }, true }; +inline const TableTypeInfo * IntsTableType() { return &IntsTableInfo; } + +inline const TableFieldInfo FloatsTableFields[] = { + { "values", "values", "float32", 0x21277bcf1a4d67fbull, 10, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Floats, values ), (uint32_t) sizeof( float ), (uint32_t) offsetof( Floats, values.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Floats, after ), (uint32_t) sizeof( Floats::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo FloatsTableInfo = { "Floats", (uint32_t) sizeof( Floats ), 2, FloatsTableFields, +[]( void * p ) { FloatsReset( *(Floats *) p ); }, true }; +inline const TableTypeInfo * FloatsTableType() { return &FloatsTableInfo; } + +// ---- the text form (docs/SPEC-TABLES.md §16) ---- + +// Bytes in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in ReportTable.cpp; link it to use them. +bool BytesFromJson( BytesBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t BytesToJsonMeasure( const Bytes * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t BytesToJson( const Bytes * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Ints in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in ReportTable.cpp; link it to use them. +bool IntsFromJson( IntsBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t IntsToJsonMeasure( const Ints * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t IntsToJson( const Ints * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Floats in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in ReportTable.cpp; link it to use them. +bool FloatsFromJson( FloatsBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t FloatsToJsonMeasure( const Floats * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t FloatsToJson( const Floats * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/SaveTable.cpp b/testdata/golden/tables/lists/SaveTable.cpp new file mode 100644 index 000000000..70e6c40f4 --- /dev/null +++ b/testdata/golden/tables/lists/SaveTable.cpp @@ -0,0 +1,3181 @@ +// Code generated by the schema compiler from Save.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — the TABLE wire's text form (docs/SPEC-TABLES.md §16). +// Compile this file to use FromJson / ToJson; a project that +// never reads or writes a text does not compile it and pays nothing. + +#include "SaveTable.h" + +#include // the text form: number formatting +#include // the text form: exact number conversion +#include // the text form: the runtime's decimal point + +// The guard is not vestigial. Several listdemo Table.cpp files may be +// concatenated into ONE translation unit — a unity build — and without it +// each would redefine the walk. It is also why the walk's functions may be +// weak (vague linkage) across separate objects: ODR requires their +// definitions to be token-identical, and the generic-walk gate is what +// proves that, byte for byte, across every generated .cpp. +#ifndef LISTDEMO_SCHEMA_TABLE_JSON +#define LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +// ---- the pointer adapters (docs/SPEC-TABLES.md §16.7) ---- +// +// The walk below is ONE walk, byte-identical in every generated .cpp, and a +// pointer is the one kind it cannot walk alone: reading one needs the +// builder's arena and writing one needs a region's deref, and neither exists +// in a unit that declares no pointer. So the walk calls these three and does +// not define them. A unit with no pointer defines them as stubs no field ever +// reaches; a pointered unit defines them in the graph half that follows the +// walk. + +struct TableJsonIn; +struct TableJsonOut; + +// a pointer field's object, or the `&node` reference standing in for it, into +// the slot; the cursor is on the opening brace +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// the node a pointer slot names, in place — or as `&node` when it is shared +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// the FIRST key of an object the walk is skipping begins with `&`: the cursor is +// on its value. A dropped definition still takes its label (§16.7); a fixed reader +// skips the value whole, as it skips everything else it does not place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); + +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- +// +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map +inline bool TableJsonIsMap( const TableFieldInfo * f ); +// the map as a plain JSON object keyed by the KEY, in ASCENDING key order +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that object back into the slot, in whatever order the text gives it +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); + +// ---- json walk: begin ---- +// +// The TEXT form (docs/SPEC-TABLES.md §16): one table, one text, one walk over the +// reflection descriptors (§8). Reading fills ONE caller-owned instance and +// allocates nothing beyond it; writing targets a caller buffer with the +// wire's measure/write symmetry. Everything AROUND this — which file goes +// with which instance, what key an instance is filed under, how instances +// link into a root table's collections — is a packer's opinion and stays +// with the tool that holds it. +// +// The dialect: trailing commas are accepted on read (the authoring files +// this exists for carry them) and never written; comments are not JSON and +// are refused; unknown keys are skipped and counted; a duplicate key is +// last-wins and counted; a key present with the wrong JSON type is skipped +// and counted, never coerced. + +static const int32_t kTableJsonMaxDepth = 128; + +// A key longer than this cannot name a field, so it is skipped as unknown. +static const int32_t kTableJsonMaxKey = 256; + +// The longest numeric token the walk will convert. Anything longer is a +// value no field can hold and counts as a kind mismatch. +static const int32_t kTableJsonMaxNumber = 512; + +// The decimal point the C runtime is CURRENTLY using. Number conversion is +// the one locale-sensitive corner of the grammar — JSON's point is always +// '.', the runtime's is whatever the program set — so every number crosses +// this one character on the way out and on the way back in. Nothing else in +// the walk consults the locale. +inline char TableJsonDecimalPoint() +{ + const struct lconv * conv = localeconv(); + if ( conv != NULL && conv->decimal_point != NULL && conv->decimal_point[0] != 0 ) + { + return conv->decimal_point[0]; + } + return '.'; +} + +// ---- storage access: the descriptors give an offset and a width, and the +// ---- storage is the HOST's, so every load and store goes through a width +// ---- switch rather than a memcpy into the low bytes of a wider word + +// finite: not a NaN, not an infinity. Written without — the walk's +// runtime surface stays the handful of functions it already names. +// A vocabulary entry the descriptor could not spell. The generated name +// functions answer "???" for a value outside the declared set, and that is +// not a name — writing it would put a spelling in the text that the reader +// then counts as unknown, turning a refusal into a silent loss. +inline bool TableJsonNamed( const char * name ) +{ + return name != NULL && strcmp( name, "???" ) != 0; +} + +inline bool TableJsonFinite( double v ) +{ + return v == v && v <= 1.7976931348623157e308 && v >= -1.7976931348623157e308; +} + +inline uint64_t TableJsonGetRaw( const void * storage, uint32_t width ) +{ + switch ( width ) + { + case 1: { uint8_t v = 0; memcpy( &v, storage, 1 ); return v; } + case 2: { uint16_t v = 0; memcpy( &v, storage, 2 ); return v; } + case 4: { uint32_t v = 0; memcpy( &v, storage, 4 ); return v; } + case 8: { uint64_t v = 0; memcpy( &v, storage, 8 ); return v; } + } + return 0; +} + +inline void TableJsonSetRaw( void * storage, uint32_t width, uint64_t value ) +{ + switch ( width ) + { + case 1: { uint8_t v = (uint8_t) value; memcpy( storage, &v, 1 ); break; } + case 2: { uint16_t v = (uint16_t) value; memcpy( storage, &v, 2 ); break; } + case 4: { uint32_t v = (uint32_t) value; memcpy( storage, &v, 4 ); break; } + case 8: { uint64_t v = value; memcpy( storage, &v, 8 ); break; } + } +} + +inline int64_t TableJsonGetSigned( const void * storage, uint32_t width ) +{ + uint64_t raw = TableJsonGetRaw( storage, width ); + if ( width < 8 ) + { + uint64_t sign = uint64_t( 1 ) << ( width * 8 - 1 ); + if ( ( raw & sign ) != 0 ) + { + raw |= ~( ( sign << 1 ) - 1 ); + } + } + return (int64_t) raw; +} + +// ---- the WIDE kinds (docs/SPEC-TABLES.md §3, §16.2) ---- +// +// The 128-bit integers and the fixed-point family convert EXACTLY, over two +// 64-bit lanes: a 128-bit integer is a decimal integer, a fixed value a +// decimal in WHOLE UNITS (1.0, -0.25, 3.0000152587890625) and nothing +// on either path passes through a double. Nothing here needs a 128-bit type +// either, which is what keeps this walk one text for every unit. +struct TableJsonWide +{ + uint64_t lo; + uint64_t hi; +}; + +inline bool TableJsonKindWide( uint8_t kind ) { return kind >= 18 && kind <= 29; } +inline bool TableJsonKindWideSigned( uint8_t kind ) { return kind == 18 || ( kind >= 20 && kind <= 24 ); } +inline bool TableJsonKindFixed( uint8_t kind ) { return kind >= 20 && kind <= 29; } + +inline bool TableJsonWideZero( TableJsonWide v ) { return v.lo == 0 && v.hi == 0; } +inline bool TableJsonWideNegative( TableJsonWide v ) { return ( v.hi >> 63 ) != 0; } + +inline int TableJsonWideCompare( TableJsonWide a, TableJsonWide b, bool is_signed ) +{ + if ( is_signed && TableJsonWideNegative( a ) != TableJsonWideNegative( b ) ) { return TableJsonWideNegative( a ) ? -1 : 1; } + if ( a.hi != b.hi ) { return a.hi < b.hi ? -1 : 1; } + if ( a.lo != b.lo ) { return a.lo < b.lo ? -1 : 1; } + return 0; +} + +inline TableJsonWide TableJsonWideShl( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.hi = v.lo << ( n - 64 ); return r; } + r.hi = ( v.hi << n ) | ( v.lo >> ( 64 - n ) ); + r.lo = v.lo << n; + return r; +} + +inline TableJsonWide TableJsonWideShr( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.lo = v.hi >> ( n - 64 ); return r; } + r.lo = ( v.lo >> n ) | ( v.hi << ( 64 - n ) ); + r.hi = v.hi >> n; + return r; +} + +inline TableJsonWide TableJsonWideNeg( TableJsonWide v ) +{ + TableJsonWide r; + r.lo = ~v.lo + 1; + r.hi = ~v.hi + ( r.lo == 0 ? 1 : 0 ); + return r; +} + +// v = v * m + a; the return is the carry out of 128 bits +inline uint32_t TableJsonWideMulAdd( TableJsonWide * v, uint32_t m, uint32_t a ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t carry = a; + for ( int i = 0; i < 4; i++ ) + { + uint64_t p = limb[i] * m + carry; + limb[i] = p & 0xffffffffull; + carry = p >> 32; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) carry; +} + +// v = v / d; the return is the remainder +inline uint32_t TableJsonWideDiv( TableJsonWide * v, uint32_t d ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t rem = 0; + for ( int i = 3; i >= 0; i-- ) + { + uint64_t cur = ( rem << 32 ) | limb[i]; + limb[i] = cur / d; + rem = cur % d; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) rem; +} + +// The storage of a wide kind, as lanes. A sixteen-byte storage is serialize's +// pair — native __int128 in the host's byte order, or the emulated struct with +// its low lane first — so the lanes are read in the host's order; a narrower +// storage is one lane, sign-extended for a signed kind. +inline TableJsonWide TableJsonWideLoad( const void * storage, uint32_t width, bool is_signed ) +{ + TableJsonWide v = { 0, 0 }; + if ( width == 16 ) + { + uint64_t half[2]; + memcpy( half, storage, 16 ); + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + v.lo = little ? half[0] : half[1]; + v.hi = little ? half[1] : half[0]; + return v; + } + v.lo = is_signed ? (uint64_t) TableJsonGetSigned( storage, width ) : TableJsonGetRaw( storage, width ); + v.hi = ( is_signed && ( v.lo >> 63 ) != 0 ) ? ~uint64_t( 0 ) : 0; + return v; +} + +inline void TableJsonWideStore( void * storage, uint32_t width, TableJsonWide v ) +{ + if ( width == 16 ) + { + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + uint64_t half[2]; + half[0] = little ? v.lo : v.hi; + half[1] = little ? v.hi : v.lo; + memcpy( storage, half, 16 ); + return; + } + TableJsonSetRaw( storage, width, v.lo ); +} + +// a counted field's companion: a string's length, a bytes' length, a counted +// array's count. Bounded by the declared extent on the way out, so a storage +// invariant a caller broke cannot walk off the end of the array. +inline int32_t TableJsonCount( const void * base, const TableFieldInfo * f ) +{ + if ( !f->counted ) + { + return f->array_bound; + } + int32_t count = 0; + memcpy( &count, (const uint8_t *) base + f->count_offset, sizeof( count ) ); + if ( count < 0 ) { count = 0; } + if ( count > f->array_bound ) { count = f->array_bound; } + return count; +} + +inline void TableJsonSetCount( void * base, const TableFieldInfo * f, int32_t count ) +{ + if ( f->counted ) + { + memcpy( (uint8_t *) base + f->count_offset, &count, sizeof( count ) ); + } +} + +// ---- what a field's kind expects to see in the text ---- +// +// One classifier, consulted by both directions, so a reader and a writer can +// never disagree about a kind's JSON form. 'o' object, 'a' array, 's' +// string, 'n' number, 'b' boolean. +// +// A vocabulary field is spelled by NAME: an enum is one name, a flags mask +// is the array of the names of its set bits. The two are told apart by the +// id column — an enum variant rides under a wire id, a flags BIT never does +// (docs/SPEC-TABLES.md §4), so a name function with no id function is flags. +// +// bytes(N) is the one kind whose element kind does not decide its form: it +// shares u8 with a plain array of u8, and rides as base64. The schema type +// name settles it, and "bytes" is a keyword no declaration can claim. +inline bool TableJsonIsBytes( const TableFieldInfo * f ) +{ + return f->is_array && f->kind == 6 && strcmp( f->type_name, "bytes" ) == 0; +} + +// An ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): its JSON form is an OBJECT +// keyed by variant name, not a positional array, because that is what the +// storage is — one slot per variant, addressed by the variant. +inline bool TableJsonIsKeyed( const TableFieldInfo * f ) +{ + return f->key_name != NULL; +} + +// THE KEY A STORAGE SLOT HOLDS (§2.4, §8): the storage shifts left, so slot i +// holds the key i + 1 and nothing is stored for None. This is the ONE place +// the walker spells the shift. +inline uint64_t TableJsonKeyedSlotKey( int64_t slot ) +{ + return (uint64_t) ( slot + 1 ); +} + +// A slot whose key names a variant of the keying enum. Every slot in +// [0, array_bound) does, unless the enum carries max-headroom variants outside +// a table closure, where a reserved value names nothing and its key id is 0 — +// the reserved id no declared name can fold to (§5). +inline bool TableJsonKeyedSlotValid( const TableFieldInfo * f, int64_t slot ) +{ + return f->key_id( TableJsonKeyedSlotKey( slot ) ) != 0; +} + +inline bool TableJsonIsFlags( const TableFieldInfo * f ) +{ + return f->enum_name != NULL && f->variant_id == NULL; +} + +inline bool TableJsonIsEnum( const TableFieldInfo * f ) +{ + return f->variant_id != NULL && f->arms == NULL; +} + +inline char TableJsonShape( const TableFieldInfo * f ) +{ + if ( TableJsonIsMap( f ) ) return 'o'; // a MAP: an object keyed by the KEY (§2.8) + if ( f->kind == 12 ) return 's'; // string + if ( TableJsonIsBytes( f ) ) return 's'; // bytes: base64 + if ( TableJsonIsKeyed( f ) ) return 'o'; // an object keyed by variant NAME + if ( f->is_array ) return 'a'; + if ( f->arms != NULL ) return 'o'; // union: an object with ONE key + if ( f->kind == 13 ) return 'o'; // nested table or type + if ( f->kind == 17 ) return f->table != NULL ? 'o' : 's'; // a pointer: the pointee's object in place, or null (§16.7); a byte buffer's string (§2.5) + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// the ELEMENT shape of an array field — the same classifier one level down +inline char TableJsonElementShape( const TableFieldInfo * f ) +{ + if ( f->arms != NULL ) return 'o'; // an element of an array of unions: one key, the arm (§2.6) + if ( f->kind == 13 ) return 'o'; + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// A guarded group rides only when its guard reads true — the wire's own +// elision (§4), carried into the text so a text and a wire written from one +// instance say the same thing. The guard is spelled as its branch condition +// over bool fields of the SAME type ("at_rest", "!at_rest", +// "active && has_target"), so evaluating it is a walk of the same +// descriptor. Nothing is inferred in the other direction: reading places +// every key it can name, and the guard is a plain bool key (§16.2). +inline bool TableJsonGuardHolds( const void * base, const TableTypeInfo * info, const char * guard ) +{ + const char * p = guard; + for ( ;; ) + { + while ( *p == ' ' || *p == '&' ) { p++; } + if ( *p == 0 ) { return true; } + bool want = true; + if ( *p == '!' ) { want = false; p++; } + const char * start = p; + while ( *p != 0 && *p != ' ' && *p != '&' ) { p++; } + size_t length = (size_t) ( p - start ); + bool value = false; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( strlen( f->name ) == length && strncmp( f->name, start, length ) == 0 ) + { + value = TableJsonGetRaw( (const uint8_t *) base + f->offset, f->elem_size ) != 0; + break; + } + } + if ( value != want ) { return false; } + } +} + +// ---- writing ---- + +// The writer sink MEASURES when the buffer is NULL and WRITES when it is +// not, over one code path — so measure and write agree byte for byte, the +// wire's invariant (§9) carried across. +struct TableJsonOut +{ + char * buffer; + int64_t capacity; + int64_t offset; + bool overflow; + void * graph; // the pointered write's identity map (§16.7); NULL for a fixed table + + void raw( const char * data, int64_t count ) + { + if ( buffer != NULL ) + { + if ( offset + count > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) count ); + } + offset += count; + } + void put( char c ) { raw( &c, 1 ); } + void text( const char * s ) { raw( s, (int64_t) strlen( s ) ); } + void line( int32_t depth ) + { + put( '\n' ); + for ( int32_t i = 0; i < depth; i++ ) { raw( " ", 2 ); } + } +}; + +inline const char * TableJsonBase64Alphabet() +{ + return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; +} + +inline void TableJsonWriteBase64( TableJsonOut & out, const uint8_t * data, int32_t length ) +{ + const char * alphabet = TableJsonBase64Alphabet(); + out.put( '"' ); + int32_t i = 0; + for ( ; i + 3 <= length; i += 3 ) + { + uint32_t triple = ( uint32_t( data[i] ) << 16 ) | ( uint32_t( data[i+1] ) << 8 ) | uint32_t( data[i+2] ); + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], + alphabet[ ( triple >> 6 ) & 0x3f ], alphabet[ triple & 0x3f ] }; + out.raw( quad, 4 ); + } + if ( i < length ) + { + int32_t left = length - i; + uint32_t triple = uint32_t( data[i] ) << 16; + if ( left == 2 ) { triple |= uint32_t( data[i+1] ) << 8; } + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], '=', '=' }; + if ( left == 2 ) { quad[2] = alphabet[ ( triple >> 6 ) & 0x3f ]; } + out.raw( quad, 4 ); + } + out.put( '"' ); +} + +// One UTF-8 sequence at s, or -1 when the bytes there are not one. Rejects +// the lot: a stray continuation, an overlong form, a surrogate half, and +// anything past U+10FFFF. +inline int32_t TableJsonUtf8( const char * s, int32_t remaining, int32_t * width ) +{ + unsigned char lead = (unsigned char) s[0]; + int32_t want = 0; + int32_t code = 0; + if ( lead < 0x80 ) { *width = 1; return lead; } + else if ( lead >= 0xc2 && lead <= 0xdf ) { want = 2; code = lead & 0x1f; } + else if ( lead >= 0xe0 && lead <= 0xef ) { want = 3; code = lead & 0x0f; } + else if ( lead >= 0xf0 && lead <= 0xf4 ) { want = 4; code = lead & 0x07; } + else { return -1; } + if ( remaining < want ) { return -1; } + for ( int32_t i = 1; i < want; i++ ) + { + unsigned char next = (unsigned char) s[i]; + if ( ( next & 0xc0 ) != 0x80 ) { return -1; } + code = ( code << 6 ) | ( next & 0x3f ); + } + if ( want == 3 && code < 0x800 ) { return -1; } // overlong + if ( want == 4 && code < 0x10000 ) { return -1; } // overlong + if ( code >= 0xd800 && code <= 0xdfff ) { return -1; } // a surrogate half + if ( code > 0x10ffff ) { return -1; } + *width = want; + return code; +} + +// A JSON text MUST be valid UTF-8 (RFC 8259 §8.1). The read path is +// byte-transparent — the wire imposes no encoding (§3) and a string may hold +// anything — so the WRITER is where that obligation is met: a byte that is +// not part of a well-formed sequence is written as U+FFFD, one per bad byte, +// and never raw. A text this walk writes is therefore readable by any +// conforming parser, which a raw byte would not be. The cost is stated +// plainly: for a string holding invalid UTF-8, the round trip is NOT +// byte-identical, because the alternative is emitting a text that is not +// JSON. +inline void TableJsonWriteString( TableJsonOut & out, const char * s, int32_t length ) +{ + static const char hex[] = "0123456789abcdef"; + out.put( '"' ); + for ( int32_t i = 0; i < length; i++ ) + { + unsigned char c = (unsigned char) s[i]; + switch ( c ) + { + case '"': out.raw( "\\\"", 2 ); break; + case '\\': out.raw( "\\\\", 2 ); break; + case '\b': out.raw( "\\b", 2 ); break; + case '\f': out.raw( "\\f", 2 ); break; + case '\n': out.raw( "\\n", 2 ); break; + case '\r': out.raw( "\\r", 2 ); break; + case '\t': out.raw( "\\t", 2 ); break; + default: + if ( c < 0x20 ) + { + char escape[6] = { '\\', 'u', '0', '0', hex[ c >> 4 ], hex[ c & 0xf ] }; + out.raw( escape, 6 ); + } + else if ( c < 0x80 ) + { + out.put( (char) c ); + } + else + { + int32_t width = 0; + if ( TableJsonUtf8( s + i, length - i, &width ) < 0 ) + { + out.raw( "\xef\xbf\xbd", 3 ); // U+FFFD, one per bad byte + } + else + { + out.raw( s + i, width ); + i += width - 1; + } + } + break; + } + } + out.put( '"' ); +} + +inline void TableJsonWriteUnsigned( TableJsonOut & out, uint64_t value ) +{ + char digits[24]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) ( value % 10 ) ); + value /= 10; + } while ( value != 0 ); + char text[24]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); +} + +inline void TableJsonWriteSigned( TableJsonOut & out, int64_t value ) +{ + if ( value < 0 ) + { + out.put( '-' ); + TableJsonWriteUnsigned( out, uint64_t( 0 ) - (uint64_t) value ); + return; + } + TableJsonWriteUnsigned( out, (uint64_t) value ); +} + +// A wide kind writes its raw storage as §16.2's text: a 128-bit integer as a +// decimal integer; a fixed value in WHOLE UNITS as the shortest exact decimal +// with at least one fractional digit (1.0, -0.25), the spelling the schema text +// gives a fixed default. The fraction terminates because a dyadic fraction has +// a finite decimal expansion — at most F digits. +inline void TableJsonWriteWide( TableJsonOut & out, const void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + TableJsonWide v = TableJsonWideLoad( storage, f->elem_size, is_signed ); + if ( is_signed && TableJsonWideNegative( v ) ) + { + out.put( '-' ); + v = TableJsonWideNeg( v ); + } + int frac = f->frac_bits; + TableJsonWide whole = TableJsonWideShr( v, frac ); + char digits[40]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) TableJsonWideDiv( &whole, 10 ) ); + } while ( !TableJsonWideZero( whole ) ); + char text[40]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); + if ( !TableJsonKindFixed( f->kind ) ) { return; } + out.put( '.' ); + // the fraction bits alone: v with everything at and above bit F cleared + TableJsonWide fraction = v; + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + if ( frac == 0 ) { fraction.lo = 0; } + if ( TableJsonWideZero( fraction ) ) + { + out.put( '0' ); + return; + } + while ( !TableJsonWideZero( fraction ) ) + { + // ×10: the digit is what lands at and above bit F, including the + // carry out of 128 bits when F leaves no room for it below + uint32_t carry = TableJsonWideMulAdd( &fraction, 10, 0 ); + uint64_t digit = TableJsonWideShr( fraction, frac ).lo; + if ( frac > 64 ) { digit |= uint64_t( carry ) << ( 128 - frac ); } + out.put( (char) ( '0' + (int) digit ) ); + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + } +} + +// A float writes at the SHORTEST precision that reads back as the same value +// at the field's own width, so a round trip is exact and a text stays +// readable. Non-finite values have no JSON spelling at all, and the writer +// REFUSES rather than losing one silently — the same rule measure and save +// already apply to an enum value no variant names (§5). +inline bool TableJsonWriteFloat( TableJsonOut & out, double value, bool single ) +{ + if ( !TableJsonFinite( value ) ) { return false; } + char text[64]; + int low = single ? 6 : 15; + int high = single ? 9 : 17; + int length = 0; + for ( int digits = low; ; digits++ ) + { + length = snprintf( text, sizeof( text ), "%.*g", digits, value ); + if ( length <= 0 || length >= (int) sizeof( text ) ) { return false; } + if ( digits >= high ) { break; } + // the round-trip check runs BEFORE the decimal point is normalised: + // the token still carries whatever point snprintf just produced + if ( single ) + { + if ( (double) strtof( text, NULL ) == value ) { break; } + } + else + { + if ( strtod( text, NULL ) == value ) { break; } + } + } + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int i = 0; i < length; i++ ) + { + if ( text[i] == point ) { text[i] = '.'; } + } + } + out.raw( text, length ); + return true; +} + +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration writes through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ); + +// one scalar, at one storage address: a nested object, a union, a +// vocabulary, or a number +inline bool TableJsonWriteScalar( TableJsonOut & out, const void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; None is {} + const TableUnionInfo * arms = f->arms(); + uint64_t tag = TableJsonGetRaw( (const uint8_t *) storage + arms->tag_offset, arms->tag_size ); + if ( tag == 0 ) + { + out.raw( "{}", 2 ); + return true; + } + if ( (int64_t) tag > f->enum_max ) + { + return false; // a tag no arm names, exactly as measure refuses it + } + const char * arm = f->enum_name( tag ); + // and refuse on the NAME, not merely on the bound: §16.2 says a value + // no variant NAMES is refused, so the check is the name. Writing + // whatever came back would emit "???", a spelling the reader counts + // as unknown — a silent round-trip loss in place of a refusal. + if ( !TableJsonNamed( arm ) ) { return false; } + out.put( '{' ); + out.line( depth + 1 ); + TableJsonWriteString( out, arm, (int32_t) strlen( arm ) ); + out.raw( ": ", 2 ); + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2): an arm that names + // no declaration carries the FIELD descriptor a field of its type + // would carry, offsets taken inside the union storage (§2.6), so the + // value walks through the field writer one key down. + if ( arms->arms[tag].field != NULL ) + { + if ( !TableJsonWriteField( out, storage, arms->arms[tag].field, depth + 1 ) ) + { + return false; + } + } + else if ( arms->arms[tag].table == NULL ) + { + out.raw( "null", 4 ); // a payload-free arm: the name selects it (§2.6) + } + else if ( !TableJsonWriteValue( out, (const uint8_t *) storage + arms->arms[tag].offset, arms->arms[tag].table, depth + 1 ) ) + { + return false; + } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->kind == 13 ) + { + return TableJsonWriteValue( out, storage, f->table, depth ); + } + if ( TableJsonIsEnum( f ) ) + { + uint64_t value = TableJsonGetRaw( storage, f->elem_size ); + // a value no variant names has no text spelling, exactly as it has no + // wire identity: the writer REFUSES rather than writing None over it, + // the rule measure and save already apply (docs/SPEC-TABLES.md §5) + if ( (int64_t) value > f->enum_max ) { return false; } + if ( value != 0 && f->variant_id( value ) == 0 ) { return false; } + const char * name = f->enum_name( value ); + if ( !TableJsonNamed( name ) ) { return false; } + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + return true; + } + if ( TableJsonIsFlags( f ) ) + { + uint64_t bits = TableJsonGetRaw( storage, f->elem_size ); + if ( bits == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + bool first = true; + for ( int64_t bit = 0; bit < 64; bit++ ) + { + if ( ( bits & ( uint64_t( 1 ) << bit ) ) == 0 ) { continue; } + if ( bit > f->enum_max ) + { + return false; // a bit no variant names has no text spelling + } + const char * name = f->enum_name( (uint64_t) bit ); + if ( !TableJsonNamed( name ) ) { return false; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + } + out.line( depth ); + out.put( ']' ); + return true; + } + switch ( f->kind ) + { + case 1: + out.text( TableJsonGetRaw( storage, f->elem_size ) != 0 ? "true" : "false" ); + return true; + case 10: + { + float v = 0.0f; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, (double) v, true ); + } + case 11: + { + double v = 0.0; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, v, false ); + } + case 2: case 3: case 4: case 5: + TableJsonWriteSigned( out, TableJsonGetSigned( storage, f->elem_size ) ); + return true; + default: + if ( TableJsonKindWide( f->kind ) ) + { + TableJsonWriteWide( out, storage, f ); + return true; + } + TableJsonWriteUnsigned( out, TableJsonGetRaw( storage, f->elem_size ) ); + return true; + } +} + +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ) +{ + const uint8_t * storage = (const uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonWriteMap( out, (const void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } + if ( f->kind == 17 && !f->is_array ) + { + return TableJsonWritePointer( out, storage, f, depth ); + } + if ( f->kind == 17 ) + { + // an ARRAY OF POINTERS (§2.1): the pointer row per element — the + // pointee's object in place, null, or `&node` for a shared one (§16.7) + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWritePointer( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; + } + if ( f->kind == 12 ) + { + TableJsonWriteString( out, (const char *) storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + TableJsonWriteBase64( out, storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + // one entry per SLOT, keyed by the variant that owns it, so inserting + // a variant next season moves nothing in the text either. Slot i holds + // the key i + 1: nothing is stored for None, so nothing is written for it. + out.put( '{' ); + bool first = true; + for ( int64_t slot = 0; slot < f->array_bound; slot++ ) + { + if ( !TableJsonKeyedSlotValid( f, slot ) ) { continue; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + const char * key = f->key_name( TableJsonKeyedSlotKey( slot ) ); + TableJsonWriteString( out, key, (int32_t) strlen( key ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteScalar( out, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + if ( first ) { out.raw( "}", 1 ); return true; } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->is_array ) + { + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWriteScalar( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + out.line( depth ); + out.put( ']' ); + return true; + } + return TableJsonWriteScalar( out, storage, f, depth ); +} + +// One instance's fields, in DECLARATION ORDER, defaults included — a text is +// for people and tools, and a text that elides is a text a reader has to know +// the schema to complete. `any` says whether the object is already open on +// entry — a shared node's `&node` opens it before the fields (§16.7) — and +// whether it is open on return. +inline bool TableJsonWriteFields( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth, bool & any ) +{ + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( f->guard[0] != 0 && !TableJsonGuardHolds( base, info, f->guard ) ) { continue; } + // an ABSENT optional writes no key: presence of the key IS the + // presence (§16.2), so an absent field is an absent key and nothing + // else would read back as absent + if ( f->optional && + TableJsonGetRaw( (const uint8_t *) base + f->present_offset, 1 ) == 0 ) + { + continue; + } + if ( !any ) { out.put( '{' ); } + else { out.put( ',' ); } + any = true; + out.line( depth + 1 ); + TableJsonWriteString( out, f->json, (int32_t) strlen( f->json ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, base, f, depth + 1 ) ) { return false; } + } + return true; +} + +// One instance as one object. The writer carries the reader's depth cap +// (§16.2): a pointer chain nests as deep as it is long (§16.7), and a text the +// writer produced past the cap would be a text the reader refuses. +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { return false; } + bool any = false; + if ( !TableJsonWriteFields( out, base, info, depth, any ) ) { return false; } + if ( !any ) + { + out.raw( "{}", 2 ); + return true; + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- reading ---- + +struct TableJsonIn +{ + const char * text; + int64_t size; + int64_t pos; + TableReport * report; + bool bad; // the text is not JSON: the walk stops and keeps what it placed + void * graph; // the pointered read's builder and label map (§16.7); NULL for a fixed table +}; + +inline void TableJsonSpace( TableJsonIn & in ) +{ + while ( in.pos < in.size ) + { + char c = in.text[in.pos]; + if ( c == ' ' || c == '\t' || c == '\n' || c == '\r' ) { in.pos++; continue; } + // comments are not JSON, and a walk that guessed at one would be + // reading a dialect nobody wrote down + if ( c == '/' ) { in.bad = true; } + return; + } +} + +inline char TableJsonPeek( TableJsonIn & in ) +{ + TableJsonSpace( in ); + return in.pos < in.size ? in.text[in.pos] : 0; +} + +// the shape of the value sitting at the cursor, without consuming it +inline char TableJsonValueShape( TableJsonIn & in ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return 'o'; + case '[': return 'a'; + case '"': return 's'; + case 't': case 'f': return 'b'; + case 'n': return 'z'; + case 0: return 0; + default: return 'n'; + } +} + +inline bool TableJsonLiteral( TableJsonIn & in, const char * word ) +{ + int64_t length = (int64_t) strlen( word ); + if ( in.pos + length > in.size || memcmp( in.text + in.pos, word, (size_t) length ) != 0 ) + { + in.bad = true; + return false; + } + in.pos += length; + return true; +} + +// one \uXXXX escape body; -1 when the four hex digits are not there +inline int TableJsonHex4( TableJsonIn & in ) +{ + if ( in.pos + 4 > in.size ) { return -1; } + int value = 0; + for ( int i = 0; i < 4; i++ ) + { + char c = in.text[in.pos + i]; + int digit; + if ( c >= '0' && c <= '9' ) { digit = c - '0'; } + else if ( c >= 'a' && c <= 'f' ) { digit = c - 'a' + 10; } + else if ( c >= 'A' && c <= 'F' ) { digit = c - 'A' + 10; } + else { return -1; } + value = ( value << 4 ) | digit; + } + in.pos += 4; + return value; +} + +inline int32_t TableJsonEncodeUtf8( uint32_t code, char * unit ) +{ + if ( code < 0x80 ) { unit[0] = (char) code; return 1; } + if ( code < 0x800 ) + { + unit[0] = (char) ( 0xc0 | ( code >> 6 ) ); + unit[1] = (char) ( 0x80 | ( code & 0x3f ) ); + return 2; + } + if ( code < 0x10000 ) + { + unit[0] = (char) ( 0xe0 | ( code >> 12 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( code & 0x3f ) ); + return 3; + } + unit[0] = (char) ( 0xf0 | ( code >> 18 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 12 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[3] = (char) ( 0x80 | ( code & 0x3f ) ); + return 4; +} + +// Scan one JSON string into a caller buffer. Bytes are appended ONE CODE +// POINT AT A TIME — an escape's encoding, or a UTF-8 sequence read whole — +// so a string longer than the field is clamped AT A CODE POINT BOUNDARY and +// never cut through a multi-byte character. Clamping is counted, never +// fatal, exactly as it is on the wire (§4). A NULL destination scans past a +// string without keeping it. +inline bool TableJsonScanString( TableJsonIn & in, char * out, int32_t capacity, int32_t * length ) +{ + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + int32_t placed = 0; + bool clamped = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos]; + if ( c == '"' ) { in.pos++; break; } + char unit[4]; + int32_t unit_length = 0; + if ( c == '\\' ) + { + in.pos++; + if ( in.pos >= in.size ) { in.bad = true; return false; } + char escape = in.text[in.pos++]; + switch ( escape ) + { + case '"': unit[0] = '"'; unit_length = 1; break; + case '\\': unit[0] = '\\'; unit_length = 1; break; + case '/': unit[0] = '/'; unit_length = 1; break; + case 'b': unit[0] = '\b'; unit_length = 1; break; + case 'f': unit[0] = '\f'; unit_length = 1; break; + case 'n': unit[0] = '\n'; unit_length = 1; break; + case 'r': unit[0] = '\r'; unit_length = 1; break; + case 't': unit[0] = '\t'; unit_length = 1; break; + case 'u': + { + int high = TableJsonHex4( in ); + if ( high < 0 ) { in.bad = true; return false; } + uint32_t code = (uint32_t) high; + if ( high >= 0xd800 && high <= 0xdbff && in.pos + 2 <= in.size && + in.text[in.pos] == '\\' && in.text[in.pos + 1] == 'u' ) + { + int64_t mark = in.pos; + in.pos += 2; + int low = TableJsonHex4( in ); + if ( low >= 0xdc00 && low <= 0xdfff ) + { + code = 0x10000 + ( ( (uint32_t) high - 0xd800 ) << 10 ) + ( (uint32_t) low - 0xdc00 ); + } + else + { + in.pos = mark; // a lone lead surrogate rides as itself + } + } + // a surrogate half that never found its partner has no + // UTF-8 encoding: encoding it anyway would manufacture + // CESU-8 — invalid UTF-8 — out of input that was valid + // JSON, so it reads as the replacement character + if ( code >= 0xd800 && code <= 0xdfff ) { code = 0xfffd; } + unit_length = TableJsonEncodeUtf8( code, unit ); + break; + } + default: in.bad = true; return false; + } + } + else if ( (unsigned char) c < 0x20 ) + { + in.bad = true; // a raw control character is not a JSON string body + return false; + } + else + { + // a UTF-8 sequence read WHOLE, so the clamp below can only land + // between code points. Only bytes that ACTUALLY look like + // continuations are taken: the wire imposes no encoding (§3), so + // a string may legitimately hold a stray lead byte, and one at + // the end of a text must not swallow the closing quote. + unsigned char lead = (unsigned char) c; + int32_t want = 1; + if ( ( lead & 0xe0 ) == 0xc0 ) { want = 2; } + else if ( ( lead & 0xf0 ) == 0xe0 ) { want = 3; } + else if ( ( lead & 0xf8 ) == 0xf0 ) { want = 4; } + unit[0] = c; + in.pos++; + unit_length = 1; + while ( unit_length < want && in.pos < in.size && + ( (unsigned char) in.text[in.pos] & 0xc0 ) == 0x80 ) + { + unit[unit_length++] = in.text[in.pos++]; + } + } + if ( out == NULL ) + { + placed += unit_length; // measured and not kept: a byte buffer's read sizes its node this way (§2.5) + } + else if ( placed + unit_length <= capacity ) + { + memcpy( out + placed, unit, (size_t) unit_length ); + placed += unit_length; + } + else + { + clamped = true; + } + } + if ( clamped ) { in.report->clamped++; } + if ( length != NULL ) { *length = placed; } + return true; +} + +// the numeric token at the cursor, copied out whole; false = not a number +// Scan one number, to JSON's OWN grammar (RFC 8259 §6) and not to a run of +// number-ish characters: +// +// number = [ "-" ] int [ frac ] [ exp ] +// int = "0" / ( digit1-9 *digit ) +// frac = "." 1*digit +// exp = ( "e" / "E" ) [ "-" / "+" ] 1*digit +// +// Scanning the production is what makes a typo in an authoring file a +// DIAGNOSTIC rather than a value: "1-2" scans as 1 and leaves "-2" where the +// object expects a comma, so the text is malformed — which is what §16.2 +// already promises. A permissive scan would hand "1-2" to a digit loop and +// report a clamp, and a config pipeline would never hear about it. Leading +// "+", leading zeros, ".5" and "3." are not JSON either. +inline bool TableJsonWalkNumber( TableJsonIn & in, bool * integral ) +{ + TableJsonSpace( in ); + bool whole = true; + if ( in.pos < in.size && in.text[in.pos] == '-' ) { in.pos++; } + // int: a lone zero, or a non-zero digit and any digits after it + if ( in.pos >= in.size ) { return false; } + if ( in.text[in.pos] == '0' ) + { + in.pos++; + } + else if ( in.text[in.pos] >= '1' && in.text[in.pos] <= '9' ) + { + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + } + else + { + return false; + } + // frac + if ( in.pos < in.size && in.text[in.pos] == '.' ) + { + in.pos++; + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + // exp + if ( in.pos < in.size && ( in.text[in.pos] == 'e' || in.text[in.pos] == 'E' ) ) + { + in.pos++; + if ( in.pos < in.size && ( in.text[in.pos] == '-' || in.text[in.pos] == '+' ) ) { in.pos++; } + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + *integral = whole; + return true; +} + +// the same production, with the token kept for conversion +inline bool TableJsonScanNumber( TableJsonIn & in, char * token, int32_t capacity, int32_t * length, bool * integral ) +{ + TableJsonSpace( in ); + int64_t start = in.pos; + if ( !TableJsonWalkNumber( in, integral ) ) { return false; } + int64_t count = in.pos - start; + if ( count <= 0 || count >= capacity ) { return false; } + memcpy( token, in.text + start, (size_t) count ); + token[count] = 0; + *length = (int32_t) count; + return true; +} + +// the token's exact double, through the runtime's own converter — which +// speaks the LOCALE's decimal point, so the token crosses back over that +// character on its way in +inline double TableJsonTokenDouble( const char * token, int32_t length, bool single ) +{ + char work[kTableJsonMaxNumber]; + memcpy( work, token, (size_t) length ); + work[length] = 0; + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int32_t i = 0; i < length; i++ ) + { + if ( work[i] == '.' ) { work[i] = point; } + } + } + if ( single ) { return (double) strtof( work, NULL ); } + return strtod( work, NULL ); +} + +// the token's exact integer, parsed digit by digit so no width and no +// locale can move it. Saturation is reported as a clamp, the wire's rule for +// a value outside what the reader can hold (§4). +inline int64_t TableJsonTokenInteger( const char * token, int32_t length, bool is_signed, bool * saturated ) +{ + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) + { + negative = token[i] == '-'; + i++; + } + uint64_t magnitude = 0; + bool over = false; + for ( ; i < length; i++ ) + { + uint64_t digit = (uint64_t) ( token[i] - '0' ); + if ( magnitude > ( UINT64_MAX - digit ) / 10 ) { over = true; break; } + magnitude = magnitude * 10 + digit; + } + if ( !is_signed ) + { + // -0 IS zero, and clamping it would report an event that did not + // happen; only a real negative magnitude is out of range here + if ( negative ) { *saturated = magnitude != 0; return 0; } + if ( over ) { *saturated = true; return (int64_t) UINT64_MAX; } + *saturated = false; + return (int64_t) magnitude; + } + if ( negative ) + { + if ( over || magnitude > ( uint64_t( 1 ) << 63 ) ) { *saturated = true; return INT64_MIN; } + *saturated = false; + if ( magnitude == ( uint64_t( 1 ) << 63 ) ) { return INT64_MIN; } + return -(int64_t) magnitude; + } + if ( over || magnitude > (uint64_t) INT64_MAX ) { *saturated = true; return INT64_MAX; } + *saturated = false; + return (int64_t) magnitude; +} + +// A number token into a wide kind's raw storage (docs/SPEC-TABLES.md §16.2). A +// 128-bit integer takes any token whose VALUE is integral; a fixed field any +// token whose value is EXACTLY representable in its Q I.F — a finer fraction +// is the wrong shape for the field, counted as a kind mismatch and never +// rounded, the rule SPEC.md §4.6 gives a fixed default. A magnitude past 128 +// bits saturates and counts as a clamp, as an int64 field saturates at +// INT64_MAX; the declared range clamps after it, on the RAW scale, as it does +// for every bounded scalar. +// +// The token is normalized to its digits with the decimal point after "point" +// of them. An integer part past 40 digits is above 2^128 whatever the digits +// are, and a value below 10^-40 is finer than 2^-127, the finest fraction any +// F can spell — so outside that band the answer is known without the +// arithmetic, and a token spelling 1e999999999 costs nothing to refuse. +inline bool TableJsonReadWide( TableJsonIn & in, const char * token, int32_t length, void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + int frac = f->frac_bits; + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { negative = token[i] == '-'; i++; } + const char * int_digits = token + i; + int32_t int_len = 0; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { int_len++; i++; } + const char * frac_digits = token + i; + int32_t frac_len = 0; + if ( i < length && token[i] == '.' ) + { + i++; + frac_digits = token + i; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { frac_len++; i++; } + } + int64_t exp = 0; + if ( i < length && ( token[i] == 'e' || token[i] == 'E' ) ) + { + i++; + bool exp_negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { exp_negative = token[i] == '-'; i++; } + while ( i < length && token[i] >= '0' && token[i] <= '9' ) + { + if ( exp < 100000 ) { exp = exp * 10 + ( token[i] - '0' ); } + i++; + } + if ( exp_negative ) { exp = -exp; } + } + // the digits, with the point after "point" of them; leading and trailing + // zeros stripped. digit( k ) reads the k-th of the int and frac runs. + int32_t start = 0, end = int_len + frac_len; + int64_t point = int_len + exp; + while ( start < end && ( start < int_len ? int_digits[start] : frac_digits[start - int_len] ) == '0' ) { start++; point--; } + while ( end > start && ( end - 1 < int_len ? int_digits[end - 1] : frac_digits[end - 1 - int_len] ) == '0' ) { end--; } + + TableJsonWide raw = { 0, 0 }; + bool saturated = false; + TableJsonWide signed_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) >> 1 }; + TableJsonWide signed_min = { 0, uint64_t( 1 ) << 63 }; + TableJsonWide unsigned_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) }; + if ( start == end ) + { + // zero, and -0 IS zero + } + else if ( point > 40 ) + { + saturated = true; + if ( !negative ) { raw = is_signed ? signed_max : unsigned_max; } + else if ( is_signed ) { raw = signed_min; } + } + else if ( point < -40 ) + { + in.report->kind_mismatch++; // finer than any F can spell + return true; + } + else + { + // the fraction FIRST, so an inexact value is the wrong shape whatever + // its magnitude: its digits, with the zeros a negative point puts in + // front, doubled F times; each doubling's carry is the next bit, and + // the value is exact iff nothing is left after the last one + char fd[kTableJsonMaxNumber + 48]; + int32_t fn = 0; + for ( int64_t z = point; z < 0; z++ ) { fd[fn++] = 0; } + for ( int32_t k = (int32_t) ( point > 0 ? point : 0 ) + start; k < end; k++ ) + { + fd[fn++] = (char) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ); + } + TableJsonWide fraction = { 0, 0 }; + for ( int b = 0; b < frac; b++ ) + { + int carry = 0; + for ( int32_t k = fn - 1; k >= 0; k-- ) + { + int d = fd[k] * 2 + carry; + fd[k] = (char) ( d % 10 ); + carry = d / 10; + } + fraction = TableJsonWideShl( fraction, 1 ); + fraction.lo |= (uint64_t) carry; + } + for ( int32_t k = 0; k < fn; k++ ) + { + if ( fd[k] != 0 ) + { + in.report->kind_mismatch++; + return true; + } + } + // then the whole part, saturating past 128 bits + TableJsonWide whole = { 0, 0 }; + for ( int64_t k = start; k < start + point && !saturated; k++ ) + { + uint32_t digit = k < end ? (uint32_t) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ) : 0; + if ( TableJsonWideMulAdd( &whole, 10, digit ) != 0 ) { saturated = true; } + } + if ( !saturated && frac > 0 && !TableJsonWideZero( TableJsonWideShr( whole, 128 - frac ) ) ) { saturated = true; } + if ( !saturated ) + { + raw = TableJsonWideShl( whole, frac ); + raw.lo |= fraction.lo; + raw.hi |= fraction.hi; + } + if ( is_signed ) + { + if ( !saturated && !negative && TableJsonWideNegative( raw ) ) { saturated = true; } + if ( !saturated && negative && TableJsonWideCompare( raw, signed_min, false ) > 0 ) { saturated = true; } + if ( saturated ) { raw = negative ? signed_min : signed_max; } + else if ( negative ) { raw = TableJsonWideNeg( raw ); } + } + else + { + if ( saturated ) { raw = unsigned_max; } + if ( negative && !TableJsonWideZero( raw ) ) { raw.lo = 0; raw.hi = 0; saturated = true; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->wide != NULL ) + { + TableJsonWide lo = { f->wide->lo[0], f->wide->lo[1] }; + TableJsonWide hi = { f->wide->hi[0], f->wide->hi[1] }; + if ( TableJsonWideCompare( raw, lo, is_signed ) < 0 ) { raw = lo; in.report->clamped++; } + else if ( TableJsonWideCompare( raw, hi, is_signed ) > 0 ) { raw = hi; in.report->clamped++; } + } + TableJsonWideStore( storage, f->elem_size, raw ); + return true; +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ); + +inline bool TableJsonSkipContainer( TableJsonIn & in, char close, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; // the opening bracket + bool first = true; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == close ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + if ( close == '}' ) + { + // the key is kept, because a skipped OBJECT may still be a + // pointer's: an `&node` opening it names a node the storage could + // not hold, and the numbering has to survive the drop (§16.7). + // Anywhere but first, the prefix is the reserved key out of place + // — in a pointered unit; a fixed unit skips the value whole. + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( key[0] == '&' && in.graph != NULL ) + { + if ( !first ) { in.report->malformed = true; in.bad = true; return false; } + if ( !TableJsonSkippedAmpersand( in, key, depth ) ) { return false; } + first = false; + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } + } + first = false; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return TableJsonSkipContainer( in, '}', depth ); + case '[': return TableJsonSkipContainer( in, ']', depth ); + case '"': return TableJsonScanString( in, NULL, 0, NULL ); + case 't': return TableJsonLiteral( in, "true" ); + case 'f': return TableJsonLiteral( in, "false" ); + case 'n': return TableJsonLiteral( in, "null" ); + case 0: in.bad = true; return false; + default: + { + // consumed, never converted: skipping needs no buffer, and this + // is the one walk a hostile text drives to the depth cap. It is + // the SAME production the value path scans, so an unknown key + // cannot smuggle past a number a named key would refuse. + bool integral = false; + if ( !TableJsonWalkNumber( in, &integral ) ) { in.bad = true; return false; } + return true; + } + } +} + +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration reads through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ); + +// place one scalar at one storage address +inline bool TableJsonReadScalar( TableJsonIn & in, void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; {} is None, and + // two keys is a text this walk will not guess at + const TableUnionInfo * arms = f->arms(); + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, 0 ); + if ( TableJsonPeek( in ) == '}' ) { in.pos++; return true; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t tag = 0; + for ( int64_t t = 1; t <= f->enum_max; t++ ) + { + if ( strcmp( f->enum_name( (uint64_t) t ), key ) == 0 ) { tag = t; break; } + } + if ( tag == 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + void * payload = (uint8_t *) storage + arms->arms[tag].offset; + const TableFieldInfo * arm = arms->arms[tag].field; + bool placed = true; + if ( arm != NULL ) + { + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2). A value of + // the wrong shape for that row is a KIND MISMATCH: the union + // reads None, the event is counted, and the enclosing object + // continues — the rule a FIELD's value lives under, one key + // down. A pointer arm's null is a null pointer, not a shape + // error, exactly as a pointer field's is (§16.7). + char got = TableJsonValueShape( in ); + if ( arm->kind == 17 && !arm->is_array && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + memset( payload, 0, (size_t) arms->arms[tag].size ); + } + else if ( got != TableJsonShape( arm ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( arm->kind == 17 && !arm->is_array ) + { + // A POINTER ARM'S VALUE IS THE POINTEE IN PLACE, or a + // node reference to one (§16.7) — the read a pointer + // FIELD takes, which is not the scalar walk + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadPointer( in, payload, arm, depth + 1 ) ) { return false; } + } + else + { + // SELECTION ZERO-ESTABLISHES THE ARM (SPEC §5): an arm + // takes no specified default, so zero is the establish + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadField( in, storage, arm, depth + 1 ) ) { return false; } + } + } + else if ( arms->arms[tag].table != NULL ) + { + if ( TableJsonValueShape( in ) != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else + { + arms->arms[tag].table->reset( payload ); + if ( !TableJsonReadTable( in, payload, arms->arms[tag].table, depth + 1 ) ) { return false; } + } + } + else + { + // A PAYLOAD-FREE ARM'S VALUE IS null (§2.6): the arm name + // selects it and there is nothing to place + if ( TableJsonValueShape( in ) != 'z' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( !TableJsonLiteral( in, "null" ) ) + { + return false; + } + } + if ( placed ) + { + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, (uint64_t) tag ); + } + } + char c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; // a second key: a one-of with two arms is not a value + return false; + } + if ( f->kind == 13 ) + { + f->table->reset( storage ); + return TableJsonReadTable( in, storage, f->table, depth + 1 ); + } + if ( TableJsonIsEnum( f ) ) + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + for ( int64_t v = 0; v <= f->enum_max; v++ ) + { + if ( strcmp( f->enum_name( (uint64_t) v ), name ) == 0 ) + { + TableJsonSetRaw( storage, f->elem_size, (uint64_t) v ); + return true; + } + } + // a name this build cannot name reads as None and counts as unknown, + // exactly as an unknown variant id does on the wire (§4) + TableJsonSetRaw( storage, f->elem_size, 0 ); + in.report->unknown++; + return true; + } + if ( TableJsonIsFlags( f ) ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + uint64_t bits = 0; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( c != '"' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + bool found = false; + for ( int64_t bit = 0; bit <= f->enum_max; bit++ ) + { + if ( strcmp( f->enum_name( (uint64_t) bit ), name ) == 0 ) + { + bits |= uint64_t( 1 ) << bit; + found = true; + break; + } + } + if ( !found ) { in.report->unknown++; } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + TableJsonSetRaw( storage, f->elem_size, bits ); + return true; + } + if ( f->kind == 1 ) + { + char c = TableJsonPeek( in ); + if ( c == 't' ) { if ( !TableJsonLiteral( in, "true" ) ) { return false; } TableJsonSetRaw( storage, f->elem_size, 1 ); return true; } + if ( !TableJsonLiteral( in, "false" ) ) { return false; } + TableJsonSetRaw( storage, f->elem_size, 0 ); + return true; + } + char token[kTableJsonMaxNumber]; + int32_t length = 0; + bool integral = false; + if ( !TableJsonScanNumber( in, token, kTableJsonMaxNumber, &length, &integral ) ) + { + in.bad = true; + return false; + } + if ( TableJsonKindWide( f->kind ) ) + { + return TableJsonReadWide( in, token, length, storage, f ); + } + if ( f->kind == 10 || f->kind == 11 ) + { + bool single = f->kind == 10; + double value = TableJsonTokenDouble( token, length, single ); + // A magnitude the field's format cannot hold is the WRONG SHAPE for + // the kind, and it never reaches storage: 1e400 is not a float64 and + // 1e300 is not a float32. Storing the infinity the conversion + // produced would leave an instance this walk called CLEAN that + // ToJsonMeasure then refuses forever (a non-finite float has no JSON + // spelling), and §16.1's one invariant is that a text which reads + // clean writes back. + if ( !TableJsonFinite( value ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( f->has_range ) + { + if ( value < f->range_min ) { value = f->range_min; in.report->clamped++; } + else if ( value > f->range_max ) { value = f->range_max; in.report->clamped++; } + } + if ( single ) + { + float narrow = (float) value; + if ( !TableJsonFinite( (double) narrow ) ) + { + in.report->kind_mismatch++; + return true; + } + memcpy( storage, &narrow, sizeof( narrow ) ); + } + else + { + memcpy( storage, &value, sizeof( value ) ); + } + return true; + } + // JSON HAS ONE NUMBER TYPE. 2.0 IS the integer 2 and 1e3 IS 1000, and a + // library that round-trips numbers through a double emits them that way — + // this walker's own float writer emits 1e+21. So an integer field takes + // any number whose VALUE is integral, however it was spelled; only a + // genuinely fractional value is the wrong shape for it. + bool is_signed = f->kind >= 2 && f->kind <= 5; + bool saturated = false; + int64_t value = 0; + if ( integral ) + { + value = TableJsonTokenInteger( token, length, is_signed, &saturated ); + } + else + { + double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( is_signed ) + { + if ( d >= 9223372036854775808.0 ) { value = INT64_MAX; saturated = true; } + else if ( d < -9223372036854775808.0 ) { value = INT64_MIN; saturated = true; } + else if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) d; } + } + else + { + if ( d < 0.0 ) + { + // a negative for an unsigned field clamps to zero, as the + // exact digit path already does + if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + value = 0; + saturated = true; + } + else if ( d >= 18446744073709551616.0 ) { value = (int64_t) UINT64_MAX; saturated = true; } + else if ( d != (double) (uint64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) (uint64_t) d; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->has_range ) + { + if ( (double) value < f->range_min ) { value = (int64_t) f->range_min; in.report->clamped++; } + else if ( (double) value > f->range_max ) { value = (int64_t) f->range_max; in.report->clamped++; } + } + // the field's own storage width is the last bound: a value past it + // clamps rather than wrapping, which is what the wire does too + if ( f->elem_size < 8 ) + { + if ( is_signed ) + { + int64_t high = ( int64_t( 1 ) << ( f->elem_size * 8 - 1 ) ) - 1; + int64_t low = -high - 1; + if ( value > high ) { value = high; in.report->clamped++; } + else if ( value < low ) { value = low; in.report->clamped++; } + } + else + { + uint64_t high = ( uint64_t( 1 ) << ( f->elem_size * 8 ) ) - 1; + if ( value < 0 ) { value = 0; in.report->clamped++; } + else if ( (uint64_t) value > high ) { value = (int64_t) high; in.report->clamped++; } + } + } + // at eight bytes the storage IS the parser's width, and an unsigned value + // past INT64_MAX rides here as a negative int64 by design — the token + // parser already turned a NEGATIVE token for an unsigned field into a + // clamped zero, so there is nothing left to bound. + TableJsonSetRaw( storage, f->elem_size, (uint64_t) value ); + return true; +} + +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ) +{ + uint8_t * storage = (uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonReadMap( in, (void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + + if ( f->kind == 12 ) + { + int32_t length = 0; + if ( !TableJsonScanString( in, (char *) storage, f->array_bound, &length ) ) { return false; } + storage[length] = 0; + TableJsonSetCount( base, f, length ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + // base64 decodes STRAIGHT INTO the field's storage, six bits at a + // time — no window, no temporary, so a bytes(N) of any declared + // extent reads the same way. A base64 body carries no escapes, so a + // backslash in one is simply not an alphabet character. + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + memset( storage, 0, (size_t) f->array_bound ); + TableJsonSetCount( base, f, 0 ); + const char * alphabet = TableJsonBase64Alphabet(); + int32_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + bool clamped = false; + bool malformed = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + const char * at = c != 0 ? strchr( alphabet, c ) : NULL; + if ( at == NULL ) { malformed = true; continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( at - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < f->array_bound ) + { + storage[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); + } + else + { + clamped = true; + } + } + } + if ( malformed ) + { + // a body that is not base64 is the wrong shape for the kind: the + // field keeps its default and the event is counted + in.report->kind_mismatch++; + return true; + } + if ( clamped ) { in.report->clamped++; } + TableJsonSetCount( base, f, placed ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + // every slot back to its declared defaults first, so a key the text + // omits keeps them and a repeated field key cannot leave an earlier + // occurrence's slots standing + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + void * slot = storage + (int64_t) i * f->elem_size; + if ( f->kind == 13 ) { f->table->reset( slot ); } + else { memset( slot, 0, (size_t) f->elem_size ); } + } + char shape = TableJsonElementShape( f ); + // A KEYED OBJECT'S KEYS ARE KEYS: a variant named twice is a duplicate + // key like any other, last-wins and counted (§16.2). Tracked the way + // a table's own field keys are — a bounded, allocation-free bitmask; + // a vocabulary wider than this still reads, its repeats simply stop + // being counted. + uint64_t seen[8] = {}; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t slot = -1; + for ( int64_t v = 0; v < f->array_bound; v++ ) + { + // nothing is stored for None, so "None" finds no slot and is + // an unknown key like any other name this reader cannot place + if ( !TableJsonKeyedSlotValid( f, v ) ) { continue; } + if ( strcmp( f->key_name( TableJsonKeyedSlotKey( v ) ), key ) == 0 ) { slot = v; break; } + } + if ( slot >= 0 && slot < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( slot & 63 ); + if ( ( seen[slot >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[slot >> 6] |= bit; + } + if ( slot < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; + } + if ( f->is_array ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + // LAST WINS has to be true of a repeated ARRAY key too, and it is + // wire-visible: a fixed array writes every slot, so a second, shorter + // occurrence overlaying a prefix would leave the first occurrence's + // tail standing. The field goes back to its declared defaults before + // this occurrence's elements are placed — the re-establishment a nested + // table and a union arm already get. A table element's defaults are + // its own (the reset hook); every other element kind's storage + // default is zero, which is what the generated array declares. + if ( f->kind == 13 ) + { + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + f->table->reset( storage + (int64_t) i * f->elem_size ); + } + } + else + { + memset( storage, 0, (size_t) f->array_bound * (size_t) f->elem_size ); + } + TableJsonSetCount( base, f, 0 ); + int32_t placed = 0; + char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( placed >= f->array_bound ) + { + // more elements than the reader's bound: the bounded prefix + // is kept and the excess counts, the wire's rule (§4) + in.report->clamped++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( f->kind == 17 ) + { + // an element of an ARRAY OF POINTERS (§2.1): null is a null slot, an + // object is the pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( storage + (int64_t) placed * f->elem_size, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed++; + } + else + { + if ( !TableJsonReadScalar( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + // a fixed array's tail keeps the defaults the prefill left there, + // exactly as a short wire count does + TableJsonSetCount( base, f, placed ); + return true; + } + return TableJsonReadScalar( in, storage, f, depth ); +} + +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ); + +// ONE table object: keys are field keys, unknown ones are skipped and +// counted, a repeated key is last-wins and counted. The instance is already +// at its declared defaults when this is entered, so a key the text never +// mentions keeps the default an absent field takes on the wire (§4). +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + return TableJsonReadTableKeys( in, base, info, depth, NULL ); +} + +// The keys of an object whose brace is already consumed. A pointer's object +// opens the same way a table's does, but its FIRST key may be `&node` (§16.7) +// and the adapter that reads it has to scan the key to know — so it hands the +// key it scanned in as `first_key`, with the colon consumed, and this places +// it before scanning the rest. +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ) +{ + // duplicate tracking, bounded and allocation-free: a table with more + // fields than this still reads, its repeats simply stop being counted + uint64_t seen[8] = {}; + for ( ;; ) + { + char key[kTableJsonMaxKey]; + char c = 0; + if ( first_key != NULL ) + { + memcpy( key, first_key, strlen( first_key ) + 1 ); // scanned into a buffer this size by the caller + first_key = NULL; + } + else + { + c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + } + int32_t index = -1; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + if ( strcmp( info->fields[i].json, key ) == 0 ) { index = i; break; } + } + if ( key[0] == '&' ) + { + // THE AMPERSAND PREFIX IS RESERVED TO THE FORM (docs/SPEC-TABLES.md + // §16.7). No declaration may take a key beginning with it, so this + // is never a field this build lacks — it is the sharing construct + // somewhere it cannot stand: `&node` is the FIRST key of a pointer's + // object and nothing else, and the adapter that reads a pointer + // has consumed it before these keys are read. MALFORMED, refused + // and counted; never counted as unknown, never skipped. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( index < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + const TableFieldInfo * f = &info->fields[index]; + if ( index < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( index & 63 ); + if ( ( seen[index >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[index >> 6] |= bit; + } + // PRESENCE OF THE KEY IS THE PRESENCE (§16.2): reaching this line + // is the key being present, so an optional is set present + // whatever its value — with one exception the page names: a JSON + // null, which reads as ABSENT rather than as a value. + char got = TableJsonValueShape( in ); + if ( f->kind == 17 && !f->is_array ) + { + // a pointer: null is a null pointer, an object is the pointee + // in place or an `&node` reference to one (§16.7), a string is + // a BYTE BUFFER's bytes (§2.5), and anything else is the wrong + // shape for the kind + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) base + f->offset, f->elem_size, 0 ); + } + else if ( got != TableJsonShape( f ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) base + f->offset, f, depth ) ) + { + return false; + } + } + else if ( f->optional && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + // absent, and back at its defaults: a repeated key whose last + // occurrence is null must not leave an earlier value standing + if ( f->table != NULL ) { f->table->reset( (uint8_t *) base + f->offset ); } + else { memset( (uint8_t *) base + f->offset, 0, (size_t) f->elem_size ); } + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 0 ); + } + else + { + if ( got != TableJsonShape( f ) ) + { + // the wrong JSON type for the kind: skipped, never coerced + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, base, f, depth ) ) + { + return false; + } + if ( f->optional ) + { + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 1 ); + } + } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +// ---- the two entry points the per-table wrappers name ---- + +inline bool TableJsonRead( void * value, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = NULL; + info->reset( value ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, value, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +inline int64_t TableJsonWrite( const void * value, const TableTypeInfo * info, char * buffer, int64_t capacity ) +{ + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = NULL; + if ( !TableJsonWriteValue( out, value, info, 0 ) ) { return -1; } + // THE CANONICAL TEXT ENDS WITH EXACTLY ONE NEWLINE (docs/SPEC-TABLES.md + // §16.1). Every writer emits it — this walk, the C# walk and + // "schema unpack" — and every reader accepts a text with or without one, + // because the trailing whitespace a read already skips is what makes the + // two the same text. It is a byte of the FORM rather than a file + // convention: a text that is written to a file, pasted into a diff and + // handed back through a pipe has to be one text in all three places, and a + // buffer whose last byte is a closing brace is the one shape that is not. + out.put( '\n' ); + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json walk: end ---- + +// ---- json graph walk: begin ---- +// +// THE VARIABLE CLASS's half of the text form (docs/SPEC-TABLES.md §16.7). The +// walk above places every kind but one; this defines the three adapters it +// calls for that one, and the two entry points a pointered table's wrappers +// name. The text is the fixed class's — a pointee is an object in place — and a +// node named more than once carries `&node`: defined once, with its fields, +// and referenced after by `{ "&node": N }` alone. + +// ---- the identity map ---- +// +// ONE map shape serves both directions. Writing keys it by a node's ADDRESS and +// counts the slots that name the node, so the second pass knows at a node's +// first occurrence whether it will be named again; reading keys it by the +// text's own label and answers the node it defined. Open addressing, a +// multiply-shift hash and quadrupling growth — TablePackMap's shape (§6.2), on +// the same terms: proportional to nodes, never to bytes, on the authoring +// side, and released before the call returns. + +struct TableJsonGraphEntry +{ + uint64_t key; // a node's address (write) or a label (read); 0 is an empty slot + int64_t count; // write: how many slots name this node + int64_t label; // write: the `&node` label assigned at its first write, 0 until then + uint8_t open; // the descent is still open: a reference here is a cycle (write), a self-reference (read) + uint32_t node; // read: the node's arena offset; 0 for a definition the reader dropped + const TableTypeInfo * type; // read: the node's table; NULL for a dropped one +}; + +struct TableJsonGraphMap +{ + TableJsonGraphEntry * entries; + int64_t capacity; // a power of two, or zero while empty + int64_t count; + TableAllocator allocator; // the caller's pair (§6.5): the builder's on read, the one handed to ToJson on write +}; + +inline void TableJsonGraphMapInit( TableJsonGraphMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TableJsonGraphMapShutdown( TableJsonGraphMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TableJsonGraphMapInit( map, map.allocator ); +} + +inline int64_t TableJsonGraphMapSlot( const TableJsonGraphMap & map, uint64_t key ) +{ + uint64_t hash = key * 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != 0 && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TableJsonGraphEntry * TableJsonGraphMapFind( TableJsonGraphMap & map, uint64_t key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +inline bool TableJsonGraphMapGrow( TableJsonGraphMap & map ) +{ + TableJsonGraphMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 64; + grown.count = 0; + grown.entries = (TableJsonGraphEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TableJsonGraphEntry ) ); // zeroed, by the pair's contract + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == 0 ) { continue; } + grown.entries[ TableJsonGraphMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// the entry for a key, made if it was not there; `taken` says which. NULL is the +// allocator refusing, and the walk refuses with it. +inline TableJsonGraphEntry * TableJsonGraphMapReach( TableJsonGraphMap & map, uint64_t key, bool & taken ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TableJsonGraphMapGrow( map ) ) { return NULL; } + } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + taken = entry->key != key; + if ( taken ) + { + entry->key = key; + map.count++; + } + return entry; +} + +// ---- reading: into a builder ---- + +struct TableJsonGraphIn +{ + TableWorker * worker; // where every node comes from + TableJsonGraphMap labels; // a label -> the node it defined +}; + +// `&node`'s value, the LABEL: a positive integer spelled as one — digits, no sign, no +// fraction, no exponent, no leading zero (§16.7). Anything else is malformed. +inline bool TableJsonScanLabel( TableJsonIn & in, uint64_t & label ) +{ + TableJsonSpace( in ); + if ( in.pos >= in.size || in.text[in.pos] < '1' || in.text[in.pos] > '9' ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + uint64_t value = 0; + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) + { + uint64_t digit = (uint64_t) ( in.text[in.pos] - '0' ); + if ( value > ( UINT64_MAX - digit ) / 10 ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + value = value * 10 + digit; + in.pos++; + } + label = value; + return true; +} + +// A BYTE BUFFER's text (docs/SPEC-TABLES.md §2.5, §16.2): a string. For a +// *string the string's bytes become the blob; for a *bytes the string is base64 +// and its decoded bytes do. The blob is allocated at EXACTLY the decoded +// length — the string is scanned once without keeping it to learn the length, +// and once into the node — so a blob of any size reads with no window and no +// bound to clamp against. A *bytes body that is not base64 is the wrong shape +// for the kind: the reference stays null and the event is counted. +inline bool TableJsonReadBlob( TableJsonIn & in, void * slot, const TableFieldInfo * f ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + TableRef * ref = (TableRef *) slot; + ref->value = 0; + if ( strcmp( f->type_name, "string" ) == 0 ) + { + const int64_t mark = in.pos; + int32_t length = 0; + if ( !TableJsonScanString( in, NULL, 0, &length ) ) { return false; } + in.pos = mark; + char * data = TableStringEmplace( *graph->worker, *ref, NULL, (int64_t) length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int32_t placed = 0; + return TableJsonScanString( in, data, length, &placed ); + } + // base64: the alphabet characters decide the length, six bits apiece + const char * alphabet = TableJsonBase64Alphabet(); + const int64_t mark = in.pos + 1; + int64_t symbols = 0; + bool malformed = false; + in.pos++; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + if ( c == 0 || strchr( alphabet, c ) == NULL ) { malformed = true; continue; } + symbols++; + } + if ( malformed ) + { + in.report->kind_mismatch++; + return true; + } + const int64_t length = ( symbols * 6 ) / 8; + uint8_t * data = TableBytesEmplace( *graph->worker, *ref, length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int64_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + for ( int64_t at = mark; ; at++ ) + { + char c = in.text[at]; + if ( c == '"' ) { break; } + const char * symbol = c != '=' ? strchr( alphabet, c ) : NULL; + if ( symbol == NULL ) { continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( symbol - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < length ) { data[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); } + } + } + return true; +} + +// A pointer's object. Its FIRST key decides what it is: `&node` naming a label not +// yet defined, with fields after it, is a DEFINITION; `&node` naming one already +// defined, alone, is a REFERENCE; any other key is a node named once, its +// object in place. The node comes from the +// builder's arena, and the slot holds its arena offset (§6.3). A pointer whose +// target is a BYTE BUFFER — no table — takes a string instead (§2.5). +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( f->table == NULL ) { return TableJsonReadBlob( in, slot, f ); } + // the pointee nests one level down, exactly as a by-value table does, and + // takes the same cap: a chain nests as deep as it is long (§16.7) + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + char c = TableJsonPeek( in ); + if ( c == '}' ) + { + // an empty object: a node at its defaults, named once + in.pos++; + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + return true; + } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( strcmp( key, "&node" ) != 0 ) + { + // a node named once: the pointee's object in place, and this key is + // its first field — unless it is the reserved prefix under a spelling + // this form does not have, which ReadTableKeys refuses + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return TableJsonReadTableKeys( in, node, f->table, depth + 1, key ); + } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->labels, label, taken ); + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + // ONE SPELLING, and what follows the label says which half it is: fields + // after a label the text has not defined DEFINE it, and a label alone that + // the text has defined REFERS to it. The other two are malformed — a label + // alone that the text never defined, which would otherwise read as a default + // node under a silent report, and a field after a label already defined, + // which would be a second definition. That is what keeps a typo loud. + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + bool bare = c == '}'; + if ( bare == taken ) { in.report->malformed = true; in.bad = true; return false; } + if ( bare ) + { + // A REFERENCE. A label is defined when its object CLOSES, so a + // reference met inside its own definition — at any depth of by-value + // nesting — names a node whose descent is still open: the cycle the + // wire refuses (§3.1), refused here where it is written. A definition + // the reader dropped names no node, so the slot stays null with + // nothing more counted — the drop was counted where it happened. A + // node of another table than the slot declares is a kind mismatch, as + // on the wire. + in.pos++; + if ( entry->open != 0 ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + if ( entry->type == NULL ) + { + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + if ( entry->type != f->table ) + { + memcpy( slot, &ref, sizeof( ref ) ); + in.report->kind_mismatch++; + return true; + } + ref.value = (int64_t) entry->node; + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + // A DEFINITION: the node is allocated, the label is its, and the keys after + // `&node` are its fields. The entry is OPEN until the object closes, so a + // reference to the label from inside the node's own fields is refused as + // the cycle it is; the node and its table are filled in at the close. + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + entry->open = 1; + if ( !TableJsonReadTableKeys( in, node, f->table, depth + 1, NULL ) ) { return false; } + entry = TableJsonGraphMapFind( graph->labels, label ); // the map may have grown under the descent + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + memcpy( &ref, slot, sizeof( ref ) ); + entry->node = (uint32_t) ref.value; + entry->type = f->table; + entry->open = 0; + return true; +} + +// An `&`-prefixed key opening an object the walk is SKIPPING — a value past an +// array's bound, an unknown key's value, a value of the wrong shape. A +// definition in there still takes its label, so the numbering survives whatever +// the storage could not hold (§16.7): the label is registered with no node, and a +// reference to it reads null. Any other prefixed key is the reserved prefix +// out of place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL || strcmp( key, "&node" ) != 0 ) { in.report->malformed = true; in.bad = true; return false; } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + if ( TableJsonGraphMapReach( graph->labels, label, taken ) == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return true; // a fresh entry is node 0, type NULL: a definition with no node +} + +// ---- writing: from a region's const root ---- + +struct TableJsonGraphOut +{ + TableJsonGraphMap nodes; // a node's address -> how many slots name it, and its `&node` once assigned + bool counting; // PASS ONE: count the references, refuse a cycle, emit nothing + int64_t next_label; +}; + +// The node a slot names: null as `null`, a node named once as its object in +// place, and a node named more than once under the construct. Which of the +// last two it is was learned in pass one; pass two spells it. +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphOut * graph = (TableJsonGraphOut *) out.graph; + if ( graph == NULL ) { return false; } + const void * node = f->resolve( slot ); + if ( node == NULL ) + { + out.raw( "null", 4 ); + return true; + } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->nodes, (uint64_t) (uintptr_t) node, taken ); + if ( entry == NULL ) { return false; } + if ( f->table == NULL ) + { + // A BYTE BUFFER (§2.5, §16.7): its text is a string, which has no + // first key to carry `&node`, so a blob named from more than one + // slot has no spelling this form can carry and the graph is refused — + // as a shared node with nothing to write is. A blob named once is its + // bytes in place: base64 for a *bytes, the string itself for a *string. + if ( graph->counting ) { entry->count++; return true; } + if ( entry->count > 1 ) { return false; } + const TableBlob * blob = (const TableBlob *) node; + if ( blob->length > (uint32_t) 0x7fffffff ) { return false; } + if ( strcmp( f->type_name, "string" ) == 0 ) { TableJsonWriteString( out, (const char *) ( blob + 1 ), (int32_t) blob->length ); } + else { TableJsonWriteBase64( out, (const uint8_t *) ( blob + 1 ), (int32_t) blob->length ); } + return true; + } + if ( graph->counting ) + { + // PASS ONE: one visit per node, every slot that names it counted, and + // a reference to a node whose descent is still open is a cycle — + // refused here as the wire refuses it (§3.1) + entry->count++; + if ( !taken ) { return entry->open == 0; } + entry->open = 1; + if ( !TableJsonWriteValue( out, node, f->table, depth ) ) { return false; } + entry = TableJsonGraphMapFind( graph->nodes, (uint64_t) (uintptr_t) node ); // the map may have grown under the descent + if ( entry == NULL ) { return false; } + entry->open = 0; + return true; + } + // PASS TWO: a node named once is its object in place; a node named more + // than once is DEFINED at its first occurrence — `&node` first, then its + // fields — and REFERENCED by `&node` alone after that, spelled the same way at + // every site. Labels run from 1 in first-write order and are the text's own, + // so a stray number in a hand-edited text is most often one never defined. + if ( entry->count <= 1 ) + { + return TableJsonWriteValue( out, node, f->table, depth ); + } + if ( depth > kTableJsonMaxDepth ) { return false; } + if ( entry->label != 0 ) + { + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + out.line( depth ); + out.put( '}' ); + return true; + } + entry->label = ++graph->next_label; + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + bool any = true; + int64_t before = out.offset; + if ( !TableJsonWriteFields( out, node, f->table, depth, any ) ) { return false; } + // a definition carries at least one field, because a label alone is a + // reference: a shared node with nothing to write has no definition this + // form can spell, and the writer refuses it as it refuses any value it + // cannot spell (§16.3) + if ( out.offset == before ) { return false; } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- the two entry points a pointered table's wrappers name ---- + +// The text into the builder's root. Every node the text names is allocated in +// the builder's arena through the field's own Emplace; the label map is the +// walk's, released before this returns. The root itself takes no label — nothing +// may name it (§16.7) — so an `&node` at the root is refused like any other key +// of the prefix. +inline bool TableJsonReadGraph( TableWorker & worker, void * root, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + if ( worker.arena == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } + TableJsonGraphIn graph; + graph.worker = &worker; + TableJsonGraphMapInit( graph.labels, worker.arena->allocator ); + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = &graph; + info->reset( root ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, root, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + TableJsonGraphMapShutdown( graph.labels ); + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +// The text of a region's const root: measured when the buffer is NULL, written +// when it is not, over one code path. Two passes over one walk — the first +// counts how many slots name each node and refuses a cycle, the second writes +// — so a node's first occurrence knows whether it will be named again. The +// ROOT's entry is open for the whole first pass, so a reference back at it is +// the cycle it is (§3.1), and it takes no label. +inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * info, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + if ( root == NULL ) { return -1; } + TableJsonGraphOut graph; + TableJsonGraphMapInit( graph.nodes, allocator ); + graph.counting = true; + graph.next_label = 0; + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph.nodes, (uint64_t) (uintptr_t) root, taken ); + if ( entry == NULL ) { TableJsonGraphMapShutdown( graph.nodes ); return -1; } + entry->open = 1; + TableJsonOut count; + count.buffer = NULL; + count.capacity = 0; + count.offset = 0; + count.overflow = false; + count.graph = &graph; + bool ok = TableJsonWriteValue( count, root, info, 0 ); + graph.counting = false; + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = &graph; + if ( ok ) { ok = TableJsonWriteValue( out, root, info, 0 ); } + TableJsonGraphMapShutdown( graph.nodes ); + if ( !ok ) { return -1; } + out.put( '\n' ); // the canonical text ends with exactly one newline (§16.1) + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json graph walk: end ---- + +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + +// ---- json map walk: begin ---- + +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} + +// the entry's two rows: fields[0] IS the key and fields[1] IS the value, which +// is what makes a user's own table of pairs the same bytes (§2.8) +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } + +inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } +inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } + +// AN INTEGER KEY IS THE INTEGER'S DECIMAL SPELLING, QUOTED, because a JSON +// object's keys are strings. Written digit by digit so no locale can move it. +inline void TableJsonWriteMapIntegerKey( TableJsonOut & out, const void * storage, const TableFieldInfo * key ) +{ + uint64_t magnitude = 0; + bool negative = false; + if ( TableJsonMapKeySigned( key ) ) + { + int64_t value = 0; + switch ( key->kind ) + { + case 2: value = (int64_t) *(const int8_t *) storage; break; + case 3: value = (int64_t) *(const int16_t *) storage; break; + case 4: value = (int64_t) *(const int32_t *) storage; break; + default: value = *(const int64_t *) storage; break; + } + negative = value < 0; + magnitude = negative ? ( ~(uint64_t) value ) + 1 : (uint64_t) value; + } + else + { + switch ( key->kind ) + { + case 6: magnitude = (uint64_t) *(const uint8_t *) storage; break; + case 7: magnitude = (uint64_t) *(const uint16_t *) storage; break; + case 8: magnitude = (uint64_t) *(const uint32_t *) storage; break; + default: magnitude = *(const uint64_t *) storage; break; + } + } + char digits[24]; + int32_t at = (int32_t) sizeof( digits ); + do { digits[--at] = (char) ( '0' + ( magnitude % 10 ) ); magnitude /= 10; } while ( magnitude != 0 ); + if ( negative ) { digits[--at] = '-'; } + TableJsonWriteString( out, digits + at, (int32_t) sizeof( digits ) - at ); +} + +inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const TableFieldInfo * key ) +{ + const uint8_t * storage = (const uint8_t *) entry + key->offset; + if ( TableJsonMapKeyIsString( key ) ) + { + // A STRING KEY IS THE STRING (§2.8): every JSON key of a map object is + // a KEY OF THE MAP and none is a field key, so the `&` prefix §16.7 + // reserves for field keys is ordinary data here. + TableJsonWriteString( out, (const char *) storage, *(const int32_t *) ( (const uint8_t *) entry + key->count_offset ) ); + return; + } + TableJsonWriteMapIntegerKey( out, (const void *) storage, key ); +} + +// ToJson WRITES ENTRIES IN ASCENDING KEY ORDER, so unpack then pack is +// byte-stable and a diff of two texts is a diff of two maps (§2.8, §17.2). +// A region holds them in that order already, so this is the array in place. +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "{}", 2 ); return true; } + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); + out.put( '{' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); + TableJsonWriteMapKey( out, entry, key ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// AN INTEGER KEY IS READ BY §16.2's INTEGER RULE AND BY NOTHING ELSE, so +// "2.0" and "1e3" are the integers 2 and 1000 and "-0" is zero. The token is +// walked as a JSON number over its own bytes; a token that rule calls +// malformed makes the KEY malformed, and a genuinely fractional value, or one +// outside the key kind's range, is kind_mismatch for that entry. +inline bool TableJsonMapKeyValue( const char * token, int32_t length, const TableFieldInfo * key, + int64_t & value, bool & fits ) +{ + fits = false; + TableReport scratch; + TableJsonIn probe = { token, (int64_t) length, 0, &scratch, false, NULL }; + bool integral = false; + if ( !TableJsonWalkNumber( probe, &integral ) ) { return false; } + if ( probe.pos != (int64_t) length ) { return false; } // trailing bytes: not a number + if ( !integral ) + { + const double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) { return true; } // a value no key kind holds + const double whole = d < 0 ? -d : d; + if ( whole != (double) (int64_t) whole ) { return true; } // genuinely fractional + } + bool saturated = false; + const bool is_signed = TableJsonMapKeySigned( key ); + value = integral ? TableJsonTokenInteger( token, length, is_signed, &saturated ) + : (int64_t) TableJsonTokenDouble( token, length, false ); + if ( saturated ) { return true; } // outside every width: kind_mismatch, never clamped + switch ( key->kind ) + { + case 2: fits = value >= -128 && value <= 127; break; + case 3: fits = value >= -32768 && value <= 32767; break; + case 4: fits = value >= -2147483647 - 1 && value <= 2147483647; break; + case 5: fits = true; break; + case 6: fits = value >= 0 && value <= 255; break; + case 7: fits = value >= 0 && value <= 65535; break; + case 8: fits = value >= 0 && (uint64_t) value <= 4294967295ull; break; + default: fits = integral; break; // uint64: the token's own magnitude + } + return true; +} + +// FromJson READS KEYS IN WHATEVER ORDER THE TEXT GIVES THEM. A repeated key is +// last-wins and counted duplicate, the object rule (§16.2) applied inside the +// map. An empty object is an empty map, and null is kind_mismatch. +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const char shape = TableJsonShape( value ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char token[kTableJsonMaxKey]; + int32_t token_length = 0; + if ( !TableJsonScanString( in, token, kTableJsonMaxKey - 1, &token_length ) ) { return false; } + token[token_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t key_value = 0; + bool place = true; + if ( !TableJsonMapKeyIsString( key ) ) + { + bool fits = false; + if ( !TableJsonMapKeyValue( token, token_length, key, key_value, fits ) ) + { + // A MALFORMED KEY STOPS THE READ where §16.1's rule stops it, + // with the instance holding what was placed before the stop. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( !fits ) { in.report->kind_mismatch++; place = false; } + } + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; + if ( place && entry == NULL ) + { + // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the + // wire's rule, because a clamped key is a merged entry (§2.8). + in.report->clamped++; + } + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) + { + in.report->duplicate++; // last-wins, the object rule inside the map + } + const char got = TableJsonValueShape( in ); + if ( entry == NULL ) + { + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( value->kind == 17 && !value->is_array ) + { + // A POINTER VALUE IS SHARED EXACTLY AS A POINTER FIELD IS (§2.8): + // null is a null slot, an object is the pointee in place or an + // &node reference to one (§16.7), anything else is the wrong shape — + // the same three the field-key loop gives a pointer field, because + // an entry's value IS a field line. + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) entry + value->offset, value->elem_size, 0 ); + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) entry + value->offset, value, depth + 1 ) ) + { + return false; + } + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, entry, value, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json map walk: end ---- + +// ---- json list walk: begin ---- + +// an unbounded array is the out-of-line array that is not a map (§8.1) +inline bool TableJsonIsList( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && !TableJsonIsMap( f ); +} + +// ToJson WRITES THE ELEMENTS IN INDEX ORDER, which is the only order there is, +// so unpack then pack is byte-stable without a rule of its own (§2.9, §17.2). +// A region holds the array in place, so this steps it at the descriptor's pitch. +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + const uint8_t * elements = TableJsonExtentElements( slot ); + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const uint8_t * element = elements + (int64_t) i * f->elem_size; + if ( f->kind == 17 ) + { + // a []*T's elements take the pointer row (§16.7): the pointee's + // object in place, null, or `&node` for a shared one + if ( !TableJsonWritePointer( out, element, f, depth + 1 ) ) { return false; } + } + else if ( !TableJsonWriteScalar( out, element, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; +} + +// FromJson READS EVERY ELEMENT THE TEXT CARRIES, appending each through the +// descriptor's place resolver: `[]` is an empty list, and null is +// kind_mismatch, the array row's own rule (§16.2). LAST WINS holds for a +// repeated key: the list goes back to EMPTY before this occurrence's elements +// land, the builder's storage being reclaimed at reset (§2.9). +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) slot, 8, 0 ); + TableJsonSetRaw( (uint8_t *) slot + 8, 4, 0 ); + const char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + void * element = f->place( *graph->worker, slot, NULL, 0, 0 ); + if ( element == NULL ) + { + // NOT ADDED: the arena could not carve another segment, or the + // count met the int32 cap. The text cannot be placed whole, and + // the read stops where §16.1's rule stops it. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( f->kind == 17 ) + { + // an element of a []*T (§2.9): null is a null slot, an object is the + // pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) element, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, element, f, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + // the wrong shape for the element kind: the slot keeps its + // defaults and the event counts, the array row's rule (§16.2) + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, element, f, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json list walk: end ---- + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +bool PlacementFromJson( Placement & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, PlacementTableType(), text, bytes, report ); +} + +int64_t PlacementToJsonMeasure( const Placement & value ) +{ + return TableJsonWrite( &value, PlacementTableType(), NULL, 0 ); +} + +int64_t PlacementToJson( const Placement & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, PlacementTableType(), buffer, capacity ); +} + +bool LogEntryFromJson( LogEntry & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, LogEntryTableType(), text, bytes, report ); +} + +int64_t LogEntryToJsonMeasure( const LogEntry & value ) +{ + return TableJsonWrite( &value, LogEntryTableType(), NULL, 0 ); +} + +int64_t LogEntryToJson( const LogEntry & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, LogEntryTableType(), buffer, capacity ); +} + +bool SaveFromJson( SaveBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Save * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, SaveTableType(), text, bytes, report ); +} + +int64_t SaveToJsonMeasure( const Save * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, SaveTableType(), NULL, 0, allocator ); +} + +int64_t SaveToJson( const Save * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, SaveTableType(), buffer, capacity, allocator ); +} + +bool PointFromJson( Point & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, PointTableType(), text, bytes, report ); +} + +int64_t PointToJsonMeasure( const Point & value ) +{ + return TableJsonWrite( &value, PointTableType(), NULL, 0 ); +} + +int64_t PointToJson( const Point & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, PointTableType(), buffer, capacity ); +} + +bool MixedFromJson( MixedBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Mixed * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, MixedTableType(), text, bytes, report ); +} + +int64_t MixedToJsonMeasure( const Mixed * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, MixedTableType(), NULL, 0, allocator ); +} + +int64_t MixedToJson( const Mixed * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, MixedTableType(), buffer, capacity, allocator ); +} + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/SaveTable.h b/testdata/golden/tables/lists/SaveTable.h new file mode 100644 index 000000000..a6400e50d --- /dev/null +++ b/testdata/golden/tables/lists/SaveTable.h @@ -0,0 +1,8531 @@ +// Code generated by the schema compiler from Save.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — protocol id 0xa5fbe602c119cdd9 (packets only: tables version by field id, not by protocol id) +// The TABLE wire (evolution-tolerant, docs/SPEC-TABLES.md): no serialize +// dependency — includable from any TU. + +#pragma once + +#include +#include // the prefill's scalar-array fills +#include // offsetof, for the reflection descriptors + +// ---- the hooks (docs/USAGE.md, "the C++ table runtime's hooks") ---- +// +// schema_assert — the runtime's own assert, and the refusal a debugger reads. +// NDEBUG removes it, exactly as it removes assert. A caller who already routes +// serialize's asserts writes `#define schema_assert serialize_assert` before +// including this header and both halves land in one handler. +#ifndef schema_assert +#include +#define schema_assert assert +#endif // #ifndef schema_assert + +// schema_fatal — what stands after the assert on a path that cannot continue. +// NDEBUG does not remove it. Supply it and is never included. +#ifndef schema_fatal +#include // abort +#define schema_fatal abort +#endif // #ifndef schema_fatal + +// schema_allocate / schema_release — what "no allocator handed in" means for +// this program. schema_allocate hands back ZEROED bytes and NULL on failure: +// an arena segment is copied whole, padding included, so anything left +// uninitialized here would reach a packed region. Supply both and +// is never included; hand a TableAllocator to a builder to route one +// structure's allocations somewhere else again. +#ifndef schema_allocate +#include // calloc, free +#define schema_allocate( bytes ) calloc( (size_t) 1, (size_t) ( bytes ) ) +#define schema_release( pointer ) free( pointer ) +#endif // #ifndef schema_allocate +#include // a node's lifetime starts in arena storage (placement new) +#include // one atomic per slab: the arena is lock-free by ownership + +#include "Save.h" + +#ifndef LISTDEMO_SCHEMA_TABLE_PRIMITIVES +#define LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +// THE CODEC DOES NOT DEPEND ON THE COMPILER'S INLINING BUDGET. A table of a +// realistic field count emits one large body per type, and the cursor a body +// writes through lives in the caller's `TableWriter`: across a call boundary +// that cursor round-trips through memory, and a `uint8_t *` store may alias the +// writer itself, so every put reloads it. When a budget runs out mid-body the +// codec silently degrades to that shape. Forcing the primitives and the +// fixed-class bodies inline is what keeps the cursor in registers and lets +// adjacent constant framing bytes merge into one store. +#if defined( _MSC_VER ) +#define LISTDEMO_TABLE_INLINE __forceinline +#elif defined( __GNUC__ ) || defined( __clang__ ) +#define LISTDEMO_TABLE_INLINE inline __attribute__(( always_inline )) +#else +#define LISTDEMO_TABLE_INLINE inline +#endif + +namespace listdemo { + +// WHY A READ WAS REFUSED, by name (docs/SPEC-TABLES.md §3.3, §11). A REFUSAL +// is not one of §4's events: nothing is decoded, no counter moves and no +// damage is reported, so five zero counters and a false flag are what a clean +// read prints too and only the verdict tells them apart. The reason says which +// refusal it was. +// +// This is the MESSAGE PATH's vocabulary and not the cooked form's (§7.4): a +// caller meeting one of these has been refused a MESSAGE on a connection, +// which is a different recovery with a different owner than a file a header +// match turned down. +enum TableMessageReason +{ + newer_form, // a FORM BYTE this reader does not carry (§3) + no_vocabulary, // no table for this connection: the message arrived before the announcement, or after a refused one + second_announcement, // a second announcement on a connection: it sets nothing, amends nothing, and the connection closes + vocabulary_too_large, // an announcement above the receiver's declared bound, refused before an entry is touched + message_form_as_file // a form 2 wire where a FILE was expected: its table is somewhere else +}; + +// The table-wire read report — the permissive contract's ledger. Silence +// (all zero) means the data matched this reader's schema exactly. +struct TableReport +{ + int32_t unknown = 0; // unknown field ids skipped (newer data) + int32_t kind_mismatch = 0; // known id, changed type — skipped, never misdecoded + int32_t clamped = 0; // out-of-range values clamped to declared bounds + // a key the TEXT form saw twice: last wins, and the repeat is counted + // (docs/SPEC-TABLES.md §16.2). The wire never raises it — a body carrying an + // id twice is legal input whose last occurrence wins, silently (§3). + int32_t duplicate = 0; + bool malformed = false; // framing damage; decode stopped, partial result kept + // THE REFUSAL VERDICT, which is not one of §4's events and moves no counter + // (docs/SPEC-TABLES.md §3): a FORM BYTE this reader does not carry. Five + // zero counters and a false flag are what a clean read prints too, so the + // verdict is what tells the two apart. + bool refused = false; + // WHICH refusal, and it is read only when refused is set: a read that + // was not refused has no reason, and this member is the one the caller + // must not look at then (docs/SPEC-TABLES.md §3.3). + TableMessageReason reason = newer_form; +}; + + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; +// ---- reflection (tables only, docs/SPEC-TABLES.md) ---- +// +// Static field descriptors for every type in the table closure: name, wire +// id/kind, storage offset, bounds, ranges, enum names and branch guards — +// enough to walk, print, diff, edit or bind any table value at runtime with +// no RTTI and no schema files. TableType() returns X's descriptor. + +struct TableTypeInfo; + +// One arm of a union field: where its payload sits inside the union's storage +// and what its payload looks like. The arm's NAME and its table-wire id come +// from the field's enum_name/variant_id functions at the same tag, so nothing +// is spelled twice (docs/SPEC-TABLES.md §8). +struct TableFieldInfo; + +struct TableUnionArmInfo +{ + uint32_t offset; // offsetof the arm's payload within the union storage + const TableTypeInfo * table; // the arm payload's descriptor, or NULL + // AN ARM IS A FIELD LINE (docs/SPEC-TABLES.md §2.6): an arm that names no + // declared type or table carries the FIELD descriptor a field of that + // type would carry instead — offsets taken within the union storage — so + // a generic walk meets an arm's kind, width, bounds and companions where + // it meets a field's. Exactly one of the two is non-NULL on a set arm. + const TableFieldInfo * field; + uint32_t size; // the arm's whole storage, which selection zero-establishes +}; + +// A union field's shape: the tag, and the arms indexed by it. Arms run +// [0, enum_max]; index 0 is the EMPTY arm and carries no payload. +struct TableUnionInfo +{ + uint32_t tag_offset; // offsetof the tag within the union storage + uint32_t tag_size; // sizeof the tag + const TableUnionArmInfo * arms; +}; + +// The exact raw range of a wide-kind field (docs/SPEC-TABLES.md §8.2): two 128-bit +// values as 64-bit lanes, low lane first, two's complement for the signed kinds. +struct TableWideRange +{ + uint64_t lo[2]; + uint64_t hi[2]; +}; + +// the arena's allocation front, defined with the variable-length runtime +// below; a descriptor names it only through a pointer parameter. +struct TableWorker; + +struct TableFieldInfo +{ + const char * name; // schema field name, e.g. "health" + const char * json; // the TEXT form's key: the json = "key" attribute, else name (§16.3) + const char * type_name; // schema type name, e.g. "float32", "Grade" + uint64_t id; // table-wire field id: fnv1a64 of the name, of the was alias after a rename (§5) + uint8_t kind; // table-wire kind; for arrays/strings/bytes, the ELEMENT kind + bool is_array; // fixed or counted array (bytes included) + bool is_pointer; // a *T pointer field: storage is an 8-byte TableRef; the target is a table + // THE TWO THE TEXT FORM NEEDS (docs/SPEC-TABLES.md §16.7), and they + // are here for the same reason is_pointer is: the walk is ONE walk + // over descriptors and cannot spell a target's own At or + // Emplace. `resolve` reads a slot in a REGION and answers the + // node it names, or NULL; `emplace` allocates one in a BUILDER's + // arena and points the slot at it. NULL on every field that is not + // a pointer, and emitted only in a unit that declares one. + const void * (*resolve)( const void * slot ); + void * (*emplace)( TableWorker & worker, void * slot ); + bool counted; // a _count/_length int32 companion exists (counted arrays, strings, bytes) + bool optional; // a ?T field: a _present bool companion decides whether it rides + int32_t array_bound; // array capacity / string max length; 0 for plain scalars + uint32_t offset; // offsetof the storage member + uint32_t elem_size; // sizeof the member (element size for arrays) + uint32_t count_offset; // offsetof the _count/_length companion, or 0xffffffff + uint32_t present_offset; // offsetof the _present companion, or 0xffffffff + const TableTypeInfo * table; // nested table's descriptor, or NULL + bool has_range; // a declared [min, max] (int or float) + double range_min; // NOTE: int64 ranges beyond 2^53 lose precision here + double range_max; + // the WIDE kinds (18-29, docs/SPEC-TABLES.md §3, §8.2): frac_bits is a fixed + // field's F — its storage holds units × 2^F — and wide is the declared + // range on that RAW scale, exact, as two 128-bit two's-complement values + // in 64-bit lanes (low lane first). NULL where the declaration bounds + // nothing (a bare uint128) and for every other kind; frac_bits is 0 for + // every kind that is not fixed-point. range_min/range_max still carry + // the declared bounds as doubles — whole units for a fixed field — for + // a walker that only shows them. + uint8_t frac_bits; + const TableWideRange * wide; + int64_t enum_max; // enums: highest valid value (None = 0 always valid); + // unions: the arm count (tag range [0, enum_max]); + // flags: the highest declared BIT INDEX; else -1 + // the vocabulary's names, indexed the same way enum_max bounds: an enum's + // value -> name, a union's tag -> arm name, a FLAGS field's bit index -> + // variant name. NULL for every other kind. + const char * (*enum_name)( uint64_t value ); + // the TABLE-WIRE id of one variant (docs/SPEC-TABLES.md §5): for an enum, the + // hash of the variant's name; for a union, the hash of the arm's name. + // 0 is the reserved id — an enum's None, a union's empty. NULL for every + // other kind — a FLAGS field's variants have no per-variant wire id (§4), + // so a NULL here beside a non-NULL enum_name is what says "flags". + // Walk [0, enum_max] to enumerate a vocabulary and its ids. + uint64_t (*variant_id)( uint64_t value ); + // an ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): the array has one slot per + // variant of key_type_name, indexed by the variant's value, and its slots + // ride under variant ids rather than positions. key_name and key_id are + // the key's vocabulary — walk [0, array_bound) to print slots by name. + // NULL on every other field. + const char * key_type_name; + const char * (*key_name)( uint64_t value ); + uint64_t (*key_id)( uint64_t value ); + // union fields: the tag and its arms, behind a function so the whole + // descriptor stays CONSTANT-INITIALISED (a captureless lambda converts to + // a function pointer at compile time; the arms themselves are a static + // inside it). NULL for every other kind. + const TableUnionInfo * (*arms)(); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded +}; + +struct TableTypeInfo +{ + const char * name; // schema type name + uint32_t size; // sizeof the storage struct + int32_t num_fields; + const TableFieldInfo * fields; + // put one instance back at its declared defaults, in place. A generic + // walker that fills a value has to be able to establish the defaults an + // absent field takes, and it holds no type to spell — this is the one + // thing the descriptors could not express without it. Placement-new + // value-init, exactly what the wire's read path does, and no temporary. + void (*reset)( void * storage ); + // the DERIVED mode (docs/SPEC-TABLES.md): false = fixed-size, a plain + // relocatable struct; true = variable-length, built through a Builder + // and read through a region root. Nobody declares it; the compiler + // works it out. + bool variable; +}; + +struct TableWriter +{ + uint8_t * buffer; + int64_t capacity; + int64_t offset = 0; + bool overflow = false; + + // the parameters do not repeat the member names: a parameter that hides a + // member is a warning the estate's compilers disagree about (gcc's + // -Wshadow and cl's C4458 refuse it, clang's -Wshadow does not), and this + // is a header a consumer compiles under its OWN flags + TableWriter( uint8_t * to_buffer, int64_t to_capacity ) : buffer( to_buffer ), capacity( to_capacity ) {} + + LISTDEMO_TABLE_INLINE void raw( const void * data, int64_t bytes ) + { + if ( offset + bytes > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) bytes ); + offset += bytes; + } + LISTDEMO_TABLE_INLINE void put8( uint8_t v ) { raw( &v, 1 ); } + LISTDEMO_TABLE_INLINE void put16( uint16_t v ) { uint8_t b[2] = { uint8_t( v ), uint8_t( v >> 8 ) }; raw( b, 2 ); } + LISTDEMO_TABLE_INLINE void put32( uint32_t v ) { uint8_t b[4] = { uint8_t( v ), uint8_t( v >> 8 ), uint8_t( v >> 16 ), uint8_t( v >> 24 ) }; raw( b, 4 ); } + LISTDEMO_TABLE_INLINE void put64( uint64_t v ) { put32( uint32_t( v ) ); put32( uint32_t( v >> 32 ) ); } + // a 128-bit value as two lanes, the low half first (docs/SPEC-TABLES.md §3) + LISTDEMO_TABLE_INLINE void put128( uint64_t lo, uint64_t hi ) { put64( lo ); put64( hi ); } + // EVERY LENGTH, COUNT, INDEX AND ID REFERENCE IS ONE CANONICAL UNSIGNED + // LEB128 (docs/SPEC-TABLES.md §3): seven value bits a byte, the lowest + // group first, the high bit set on every byte but the last. One value has + // one spelling, so two conforming writers agree byte for byte. + LISTDEMO_TABLE_INLINE void putleb( uint64_t v ) + { + while ( v >= 0x80 ) { put8( uint8_t( v ) | 0x80 ); v >>= 7; } + put8( uint8_t( v ) ); + } +}; + +// TableLebBytes is one value's spelling length, which a MEASURE needs before +// the bytes exist — the length of a body has to be known before it is written, +// because a length whose own width moves cannot be patched in place. +inline int64_t TableLebBytes( uint64_t v ) +{ + int64_t n = 1; + while ( v >= 0x80 ) { v >>= 7; n++; } + return n; +} + +// THE ID TABLE, WRITER SIDE (docs/SPEC-TABLES.md §3). It holds every id the +// body used, once each, in FIRST-USE order over the whole wire, and the body +// names them by position: reference k is the kth entry, counted from 1, and +// reference 0 names NO ID. +// +// Its capacity is a COMPILE-TIME fact of the unit — the distinct names its +// table closure can spell — so a save allocates nothing: the table is a local +// of Measure and of Save. The bucket chain makes ref constant time and makes +// truncate constant time too, which is what an ELIDED field needs: a field +// that turns out not to ride costs nothing in the id table either, so the walk +// interns its id, builds the payload that decides, and undoes the entry when +// nothing rides. +struct TableIds +{ + static const int32_t kCapacity = 60; + static const int32_t kBuckets = 128; + + uint64_t ids[ kCapacity ]; + int32_t chain[ kCapacity ]; + int32_t head[ kBuckets ]; + int32_t count; + bool overflow; + // THE MESSAGE FORM'S SLOTS (docs/SPEC-TABLES.md §3.3). A form 2 wire + // names ids through the CONNECTION's table, which is the unit's whole + // vocabulary in a compiler-settled order — so every reference is known at + // compile time and rides at the header as a literal beside the id. This + // flag is what selects it: false interns the id in first-use order and + // writes a trailer, true answers the slot and writes none, and the walk + // that decides is one walk. + bool vocabulary; + + TableIds() : count( 0 ), overflow( false ), vocabulary( false ) + { + for ( int32_t i = 0; i < kBuckets; i++ ) { head[i] = -1; } + } + + static LISTDEMO_TABLE_INLINE uint32_t bucket_of( uint64_t id ) + { + return uint32_t( ( id * 0x9E3779B97F4A7C15ull ) >> 57 ) & uint32_t( kBuckets - 1 ); + } + + // the reference an id takes: its message-form SLOT under the connection's + // table, or the file's own first-use entry + LISTDEMO_TABLE_INLINE uint64_t ref( uint64_t id, uint64_t slot ) + { + if ( vocabulary ) { return slot; } + return intern( id ); + } + + // the FILE form's half, appending the id on first use + uint64_t intern( uint64_t id ) + { + const uint32_t b = bucket_of( id ); + for ( int32_t i = head[b]; i >= 0; i = chain[i] ) + { + if ( ids[i] == id ) { return uint64_t( i ) + 1; } + } + if ( count >= kCapacity ) { overflow = true; return 1; } + ids[count] = id; chain[count] = head[b]; head[b] = count; count++; + return uint64_t( count ); + } + + // undo every entry appended since mark. An entry removed is the most + // recent one in its bucket, so it sits at that bucket's head. + void truncate( int32_t mark ) + { + // a SLOT costs no entry, so an elided field has nothing to undo + if ( vocabulary ) { return; } + while ( count > mark ) + { + count--; + head[ bucket_of( ids[count] ) ] = chain[count]; + } + } +}; + +// TableIdsBytes is the trailer's own size: the entries, each a fixed +// little-endian u64, and the ENTRY COUNT, the one fixed-width number on the +// wire (docs/SPEC-TABLES.md §3). +inline int64_t TableIdsBytes( const TableIds & ids ) { return int64_t( ids.count ) * 8 + 8; } + +// TableIdsWrite puts the trailer where the walk ended: a writer never patches, +// because first-use order is known only when the walk ends. +inline void TableIdsWrite( TableWriter & w, const TableIds & ids ) +{ + for ( int32_t i = 0; i < ids.count; i++ ) { w.put64( ids.ids[i] ); } + w.put64( uint64_t( ids.count ) ); +} + +// THE ID TABLE, READER SIDE (docs/SPEC-TABLES.md §3). A reader locates it from +// the END of the wire and resolves it ONCE, at open: the entries are eight +// bytes each and a body names them by position, so every field dispatches +// through an index rather than through a search over hashes. +struct TableIdTable +{ + const uint8_t * entries = NULL; + int64_t count = 0; + + // the id a reference names. ref is 1-based and bounds-checked by the + // caller: a reference ABOVE the entry count is framing damage on the body + // that carries it, and 0 names no id at all. + uint64_t at( uint64_t ref ) const + { + const uint8_t * e = entries + ( ref - 1 ) * 8; + uint64_t lo = uint64_t( e[0] ) | uint64_t( e[1] ) << 8 | uint64_t( e[2] ) << 16 | uint64_t( e[3] ) << 24; + uint64_t hi = uint64_t( e[4] ) | uint64_t( e[5] ) << 8 | uint64_t( e[6] ) << 16 | uint64_t( e[7] ) << 24; + return lo | ( hi << 32 ); + } +}; + +struct TableReader +{ + const uint8_t * buffer; + int64_t size; + int64_t offset = 0; + TableReport * report; + const TableIdTable * ids = NULL; + // ONLY THE ROOT BODY CARRIES THE NODE TABLE (docs/SPEC-TABLES.md §3.1), so + // a body has to know which it is: the reserved id inside a NESTED body is + // malformed, because a second numbering cannot exist. Every reader made + // for a payload is nested; the two the wire surfaces make for a root say so. + bool nested = true; + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report ) + : buffer( from_buffer ), size( from_size ), report( to_report ) {} + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report, const TableIdTable * to_ids ) + : buffer( from_buffer ), size( from_size ), report( to_report ), ids( to_ids ) {} + + LISTDEMO_TABLE_INLINE bool has( int64_t bytes ) const { return offset + bytes <= size; } + // A LENGTH IS A 64-BIT NUMBER AND A BUFFER IS NOT (docs/SPEC-TABLES.md + // §3): every length, count and index on this wire has sixty-four bits of + // capability, so one past what remains must be compared UNSIGNED. Casting + // it to int64 first turns 0xFFFFFFFFFFFFFFFF into -1, and a negative + // length looks like room. + LISTDEMO_TABLE_INLINE bool room( uint64_t bytes ) const { return bytes <= (uint64_t) ( size - offset ); } + LISTDEMO_TABLE_INLINE uint8_t get8() { return buffer[offset++]; } + LISTDEMO_TABLE_INLINE uint16_t get16() { uint16_t v = uint16_t( buffer[offset] ) | uint16_t( buffer[offset+1] ) << 8; offset += 2; return v; } + LISTDEMO_TABLE_INLINE uint32_t get32() { uint32_t v = uint32_t( buffer[offset] ) | uint32_t( buffer[offset+1] ) << 8 | uint32_t( buffer[offset+2] ) << 16 | uint32_t( buffer[offset+3] ) << 24; offset += 4; return v; } + LISTDEMO_TABLE_INLINE uint64_t get64() { uint64_t lo = get32(); uint64_t hi = get32(); return lo | ( hi << 32 ); } + LISTDEMO_TABLE_INLINE void get128( uint64_t & lo, uint64_t & hi ) { lo = get64(); hi = get64(); } + + // ONE CANONICAL UNSIGNED LEB128 (docs/SPEC-TABLES.md §3), and a + // non-minimal spelling is MALFORMED: 0x80 0x00 and 0x00 both spell zero, + // and only the second is legal input. An encoding past ten bytes, or a + // tenth byte with a bit above the 64th value bit, is malformed on the same + // rule. false = framing damage on the body carrying it. + bool getleb( uint64_t & value ) + { + // A NUMBER THIS READER REFUSES LEAVES THE CURSOR WHERE IT WAS. The + // caller's next question is often "did this body end exactly at its + // L", and a rejected number that had moved the cursor would answer + // that question with the damage already stepped over. + const int64_t at = offset; + value = 0; + uint32_t shift = 0; + for ( int32_t i = 0; i < 10; i++ ) + { + if ( !has( 1 ) ) { offset = at; return false; } + const uint8_t b = get8(); + if ( i == 9 && b > 1 ) { offset = at; return false; } + value |= uint64_t( b & 0x7F ) << shift; + if ( ( b & 0x80 ) == 0 ) + { + if ( i > 0 && b == 0 ) { offset = at; return false; } // a redundant continuation + return true; + } + shift += 7; + } + offset = at; + return false; + } + + // resolve one id reference against the file's table. false = a reference + // ABOVE the entry count, or a 0 where an id is required, both of which + // are framing damage on the body that carries it. + bool getid( uint64_t & id ) + { + uint64_t ref = 0; + if ( !getleb( ref ) ) { return false; } + if ( ref == 0 || ids == NULL || ref > (uint64_t) ids->count ) { return false; } + id = ids->at( ref ); + return true; + } + + // skip one payload by kind; false = framing damage. FOUR RULES COVER THE + // SET (docs/SPEC-TABLES.md §3), and a kind outside it is not skippable — + // which is why the set is closed and why kind 31 exists. + bool skip( uint8_t kind ) + { + switch ( kind ) + { + // the fixed-width kinds, each by its width: 18-29 are the 128-bit integers and + // the fixed-point family at every storage width (docs/SPEC-TABLES.md §3) + case 1: case 2: case 6: case 20: case 25: return has( 1 ) ? ( offset += 1, true ) : false; + case 3: case 7: case 21: case 26: return has( 2 ) ? ( offset += 2, true ) : false; + case 4: case 8: case 10: case 22: case 27: return has( 4 ) ? ( offset += 4, true ) : false; + case 5: case 9: case 11: case 23: case 28: return has( 8 ) ? ( offset += 8, true ) : false; + case 18: case 19: case 24: case 29: return has( 16 ) ? ( offset += 16, true ) : false; + case 17: case 30: // a NODE INDEX (§3.1) and an ENUM's variant reference: one LEB128 and stop + { + uint64_t ignored = 0; + return getleb( ignored ); + } + case 12: case 13: case 14: case 16: case 31: case 32: // 31 is the ESCAPE, 32 the payload-free kind + { + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + case 15: // union: the arm id reference, then its kind, its L and its payload (reference 0 = empty) + { + uint64_t arm = 0; + if ( !getleb( arm ) ) return false; + if ( arm == 0 ) return true; + if ( !has( 1 ) ) return false; + offset += 1; // the arm's kind byte + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + } + return false; + } +}; + +// The RESERVED node-table id, the one id the language holds back +// (docs/SPEC-TABLES.md §3.1, §5). It rides in every unit, pointered or not, +// because every body has to know that a NESTED body claiming one is damaged. +static const uint64_t kTableNodeTableFieldId = 0xFFFFFFFFFFFFFFFFull; + +// TableWireForm is the FORM BYTE, and it is the whole header +// (docs/SPEC-TABLES.md §3). A reader that meets a byte it does not know +// refuses the wire by name and never reports damage. +const uint8_t kTableWireForm = 1; + +// TableOpen reads the form byte and the trailer, in that order, and hands back +// the ROOT BODY. It answers one of three verdicts, because five zero counters +// and a false flag are what a clean read prints too: +// +// TableOpenOk the form is known and the table read whole +// TableOpenRefused a FORM BYTE this reader does not carry: nothing is +// decoded, nothing is counted, and no damage is reported +// TableOpenDamaged a table that cannot be read whole — fewer than eight +// bytes, a count whose entries run past the front of the +// file, a count that leaves no room for the form byte, or +// ONE ID IN TWO ENTRIES. The whole wire is malformed, +// nothing is decoded, and one event is counted. +// TableOpenBodyStopped the form and the table were good and the ROOT BODY +// could not be walked to its own terminator. What it +// decoded before that is kept, as everywhere on this wire. +enum TableOpenVerdict { TableOpenOk, TableOpenRefused, TableOpenDamaged, TableOpenBodyStopped }; + +inline TableOpenVerdict TableOpen( const uint8_t * buffer, int64_t bytes, TableIdTable & table, int64_t & body_bytes ) +{ + if ( bytes < 1 ) { return TableOpenDamaged; } + if ( buffer[0] != kTableWireForm ) { return TableOpenRefused; } + if ( bytes < 9 ) { return TableOpenDamaged; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + uint64_t count = lo | ( hi << 32 ); + if ( count > (uint64_t) ( bytes / 8 ) ) { return TableOpenDamaged; } + const int64_t span = (int64_t) count * 8 + 8; + if ( span + 1 > bytes ) { return TableOpenDamaged; } + table.entries = buffer + bytes - span; + table.count = (int64_t) count; + // THE ENTRIES ARE DISTINCT: a table that carries one id twice is malformed + // for the whole wire, because no wire this schema writes carries a repeat + // and it would leave one more shape of table for a hostile writer to aim + // at (docs/SPEC-TABLES.md §3). + for ( int64_t i = 1; i < table.count; i++ ) + { + const uint64_t id = table.at( uint64_t( i ) + 1 ); + for ( int64_t j = 0; j < i; j++ ) + { + if ( table.at( uint64_t( j ) + 1 ) == id ) { return TableOpenDamaged; } + } + } + body_bytes = bytes - span - 1; + return TableOpenOk; +} + +// TableBodyExtent walks a body's framing to the zero reference that ends it, +// so a reader can tell a body that ENDED EARLY — leaving bytes no field claims +// — from one that is merely damaged. ANY BYTE BETWEEN THE ROOT'S TERMINATOR +// AND THE TABLE'S FIRST ENTRY IS MALFORMED, because no field claims it and the +// two ends of the file have met (docs/SPEC-TABLES.md §3). +inline bool TableBodyEndsEarly( const uint8_t * body, int64_t bytes, const TableIdTable & table ) +{ + TableReport ignored; + TableReader r( body, bytes, &ignored, &table ); + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { return false; } + if ( ref == 0 ) { return r.offset != bytes; } + if ( ref > (uint64_t) table.count ) { return false; } + if ( !r.has( 1 ) ) { return false; } + if ( !r.skip( r.get8() ) ) { return false; } + } +} + +// THE MESSAGE FORM (docs/SPEC-TABLES.md §3.3): a FILE carries its own id +// table and a MESSAGE STREAM announces one and then carries none. +// +// A form 2 wire is TWO PARTS, the form byte and the root body: the body ends +// at its own zero reference as it does in a file, there is no trailer, and the +// message's last byte is the body's terminator. Its references resolve against +// the CONNECTION's table, which is the unit's whole vocabulary in the order +// the compiler settled. +const uint8_t kTableWireMessageForm = 2; + +// The RESERVED build-version id, the second id the language holds back (§5, +// §11), beside the node table's. It is the announcement's one required field, +// and a reserved id in any body but the one whose transport it is, is +// malformed (§3.1). +static const uint64_t kTableBuildVersionFieldId = 0xFFFFFFFFFFFFFFFEull; + +// The reserved NODE-TABLE id's own slot in this unit's vocabulary (§3.3). A +// pointered message names the node table through it, exactly as every other +// field header names its id through a slot. +static const uint64_t kTableNodeTableFieldSlot = 38; + +// THE UNIT'S ANNOUNCEMENT, byte for byte: 61 entries and 508 bytes. It is an +// ordinary form 1 FILE — the form byte, a body carrying the BUILD VERSION +// under the reserved id at kind 9, and the trailer that IS the connection's +// table, slot 1 the reserved id and slots 2 and up the vocabulary under one +// numbering. +// +// The vocabulary is the unit's whole closure in the COOK PROJECTION's order +// (§20.2) — each record in the order the projection renders it and each +// record's fields in the order the projection renders them, then each enum's +// variants and each union's arms — followed by the tail the projection does +// not name: the reserved node-table id, the three blob type ids as bytes, +// string and wstring, and every table's own name id in the projection's sorted +// record order. The tail is UNCONDITIONAL, so an ordinary edit only ever grows +// it at its end and never moves a slot a generated field header carries as a +// literal. +static const int64_t kTableAnnounceBytes = 508; +static const uint8_t kTableAnnounce[ kTableAnnounceBytes ] = { + 0x01, 0x01, 0x09, 0xc7, 0x71, 0x45, 0xca, 0xda, 0x0e, 0x7c, 0x8d, 0x00, + 0xfe, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x30, 0xb1, 0x3a, 0xff, + 0x4a, 0xd9, 0xb1, 0x40, 0x20, 0xea, 0x4d, 0x40, 0x8e, 0xa7, 0x19, 0xaa, + 0x26, 0xa9, 0x02, 0x0c, 0x9b, 0x01, 0x48, 0x78, 0xe9, 0xea, 0x71, 0x6f, + 0x0f, 0x01, 0x82, 0xbf, 0x6f, 0x2c, 0x41, 0x4f, 0xbf, 0x84, 0x78, 0x3e, + 0xf3, 0xa4, 0x48, 0x44, 0x19, 0xab, 0xd7, 0x56, 0x05, 0x4a, 0xa3, 0x30, + 0x67, 0x55, 0x5b, 0x85, 0xc9, 0xe2, 0x4e, 0x30, 0x69, 0x6a, 0xb4, 0x81, + 0xfb, 0x67, 0x4d, 0x1a, 0xcf, 0x7b, 0x27, 0x21, 0x74, 0xa2, 0x79, 0x44, + 0x8e, 0xe2, 0xe5, 0xb1, 0x84, 0x76, 0xbc, 0x2e, 0xef, 0x83, 0x76, 0x1e, + 0xc5, 0x99, 0xf7, 0x82, 0x76, 0x4e, 0x0a, 0xd9, 0xa8, 0x2e, 0x86, 0x70, + 0x84, 0xed, 0xf2, 0x4a, 0xbb, 0xf0, 0x0c, 0x9b, 0xcc, 0xfb, 0x2d, 0x73, + 0x68, 0xb7, 0xf0, 0xae, 0x4c, 0x0c, 0xf6, 0x52, 0xbf, 0xe9, 0xd1, 0x2f, + 0x93, 0xcd, 0xda, 0xdb, 0x22, 0x72, 0x34, 0x7d, 0xf6, 0x0b, 0x72, 0x17, + 0x07, 0x17, 0x02, 0x86, 0x4c, 0xf5, 0x63, 0xaf, 0x54, 0x15, 0x02, 0x86, + 0x4c, 0xf4, 0x63, 0xaf, 0x3a, 0x70, 0x6e, 0x3e, 0x93, 0x43, 0xe5, 0x9d, + 0x3d, 0x62, 0xcb, 0x8f, 0xec, 0xfc, 0xf7, 0x39, 0x09, 0x06, 0x02, 0x86, + 0x4c, 0xeb, 0x63, 0xaf, 0x09, 0x4b, 0x4d, 0x57, 0xaa, 0x33, 0x47, 0xd2, + 0x31, 0x54, 0xaf, 0x1d, 0x19, 0x73, 0x50, 0x12, 0xb2, 0x0f, 0x40, 0x27, + 0x0b, 0x6b, 0x98, 0x01, 0x38, 0x81, 0x0a, 0xf1, 0x1f, 0x06, 0xa7, 0xa3, + 0x0f, 0x62, 0xad, 0x07, 0x77, 0x47, 0x82, 0x5f, 0x42, 0x4f, 0x4f, 0x30, + 0x0d, 0x39, 0x84, 0x1c, 0x86, 0x1b, 0x63, 0x8e, 0xba, 0xad, 0xbc, 0xc4, + 0xec, 0x10, 0x5b, 0x36, 0x19, 0x4a, 0xc9, 0x3d, 0xea, 0x0c, 0xe8, 0x30, + 0x94, 0xfd, 0xe4, 0x7c, 0xec, 0x22, 0x02, 0x86, 0x4c, 0xfc, 0x63, 0xaf, + 0x05, 0x28, 0x02, 0x86, 0x4c, 0xff, 0x63, 0xaf, 0x52, 0x26, 0x02, 0x86, + 0x4c, 0xfe, 0x63, 0xaf, 0xb1, 0x45, 0xc3, 0x44, 0x35, 0xab, 0xfe, 0x73, + 0xc0, 0x7f, 0xb3, 0x8a, 0xbe, 0x08, 0x63, 0x7f, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xe4, 0x4f, 0x1c, 0x4f, 0x47, 0xc0, 0x2e, 0x2f, + 0x58, 0xfc, 0xaf, 0xfa, 0xd8, 0xe0, 0x4b, 0x70, 0xc7, 0xd4, 0x7b, 0x26, + 0xb0, 0x9d, 0x29, 0x5f, 0xcc, 0x14, 0x15, 0x7f, 0xcb, 0xc2, 0x58, 0xd8, + 0x84, 0x9e, 0x3a, 0x55, 0x8b, 0x37, 0xe2, 0x06, 0x2a, 0x89, 0xf5, 0x12, + 0x09, 0xc1, 0xca, 0x0a, 0x44, 0xa2, 0x31, 0xc1, 0xad, 0xa7, 0xee, 0xee, + 0xe8, 0xcf, 0xbf, 0x43, 0x73, 0x18, 0x43, 0xd0, 0x42, 0xad, 0xf6, 0xf8, + 0x59, 0x86, 0x63, 0x91, 0xb7, 0xce, 0x00, 0x7c, 0xd1, 0xc5, 0x34, 0x20, + 0x06, 0x68, 0x47, 0x98, 0xd1, 0xa1, 0xcf, 0x52, 0x5f, 0x82, 0x58, 0xac, + 0x36, 0x15, 0x78, 0x5e, 0xb8, 0x8b, 0x59, 0x6f, 0xc9, 0xc6, 0x86, 0xbb, + 0xc3, 0x64, 0x89, 0x50, 0xd2, 0x8d, 0xa7, 0xf1, 0x80, 0xea, 0x3a, 0xb9, + 0xf1, 0x21, 0xf7, 0x41, 0x11, 0xed, 0xd9, 0xce, 0x96, 0x92, 0x43, 0x8a, + 0xfb, 0x06, 0xc9, 0xfe, 0x19, 0xe1, 0x13, 0xa0, 0xa7, 0x0a, 0xc7, 0x54, + 0x12, 0xd6, 0x40, 0xdc, 0x08, 0xf0, 0xf5, 0xc0, 0x24, 0x5f, 0xf8, 0x33, + 0xc8, 0xfb, 0x85, 0x9a, 0xaf, 0xe0, 0xc9, 0x0c, 0x91, 0x0a, 0x55, 0x60, + 0xf7, 0xa2, 0x07, 0xec, 0x8b, 0x6d, 0x02, 0x86, 0x43, 0xf3, 0xc2, 0x2e, + 0x87, 0x27, 0xcc, 0x86, 0xf0, 0xe0, 0x26, 0x8f, 0x3d, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, +}; + +// AnnounceMeasure is the announcement's byte count, which is a constant of the +// unit and not a walk. +inline int64_t AnnounceMeasure() { return kTableAnnounceBytes; } + +// Announce writes the announcement into the caller's buffer and answers the +// bytes written — exactly AnnounceMeasure's answer — or -1 when the buffer is +// too small. It allocates nothing and walks nothing. +inline int64_t Announce( uint8_t * buffer, int64_t capacity ) +{ + if ( buffer == NULL || capacity < kTableAnnounceBytes ) { return -1; } + memcpy( buffer, kTableAnnounce, (size_t) kTableAnnounceBytes ); + return kTableAnnounceBytes; +} + +// TableVocabulary is ONE DIRECTION of ONE CONNECTION's id table (§3.3): the +// entries an announcement carried, whole, under one numbering with slot 1 the +// reserved build-version id. +// +// A peer holds TWO of these for a connection, the one it writes with and the +// one it reads with, and neither is the other's. A restart opens a fresh +// connection with empty tables and nothing is cached across connections, so +// its whole life is one connection's. It BORROWS the announcement's bytes rather than +// copying them, so a receiver holds one table a direction and its memory is +// the bound below and nothing else. +struct TableVocabulary +{ + // THE CONFORMING DEFAULT BOUND (§3.3): 32 KiB a direction, eight times the + // 500-id unit that is already a large one. A connection's table is bounded + // by nothing the wire carries, so the receiver declares the maximum and an + // announcement above it is refused by name before an entry is touched. + static const int64_t kDefaultMaxEntries = 4096; + + TableIdTable table; + uint64_t build_version = 0; + bool announced = false; + int64_t max_entries = kDefaultMaxEntries; +}; + +// AnnounceRead reads an announcement into one direction's table (§3.3). +// +// THE BOUND IS CHECKED BEFORE ANYTHING IS ALLOCATED: the entry count is a +// fixed little-endian u64 at the end, so a receiver reads it, compares it and +// refuses without touching an entry. After that it is §3's ordinary FILE read, +// because the announcement IS a file, with EXACTLY ONE STRICT CHECK over its +// body: the reserved build-version field present, exactly once, under kind 9, +// eight bytes wide. Everything else is an ordinary field under §4's tolerance, +// so an unknown one is skipped and counted and the announcement can GAIN a +// field in a later minor without a lockstep redeploy. +// +// The FIRST announcement sets the table and it is the only one that can. A +// SECOND is refused by name: it does not replace the table, it does not amend +// it and it changes nothing. A refused announcement sets NO TABLE. +inline bool AnnounceRead( TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + if ( vocabulary.announced ) + { + to->refused = true; + to->reason = second_announcement; + return false; + } + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireForm ) + { + to->refused = true; + to->reason = buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + return false; + } + if ( bytes < 9 ) { to->malformed = true; return false; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + if ( ( lo | ( hi << 32 ) ) > (uint64_t) vocabulary.max_entries ) + { + to->refused = true; + to->reason = vocabulary_too_large; + return false; + } + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else { to->refused = true; to->reason = newer_form; } + return false; + } + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) { to->malformed = true; return false; } + // the body, under §4's tolerance and this form's one strict check + TableReader r( buffer + 1, body_bytes, to, &table ); + uint64_t version = 0; + int32_t seen = 0; + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { to->malformed = true; return false; } + if ( ref == 0 ) { break; } + if ( ref > (uint64_t) table.count || !r.has( 1 ) ) { to->malformed = true; return false; } + const uint64_t id = table.at( ref ); + const uint8_t kind = r.get8(); + if ( id != kTableBuildVersionFieldId ) + { + to->unknown++; + if ( !r.skip( kind ) ) { to->malformed = true; return false; } + continue; + } + if ( kind != 9 || !r.has( 8 ) ) { to->refused = true; to->reason = no_vocabulary; return false; } + version = r.get64(); + seen++; + } + if ( seen != 1 ) { to->refused = true; to->reason = no_vocabulary; return false; } + vocabulary.table = table; + vocabulary.build_version = version; + vocabulary.announced = true; + return true; +} + +inline float table_bits_to_float( uint32_t bits ) { float f; memcpy( &f, &bits, 4 ); return f; } +inline uint32_t table_float_to_bits( float f ) { uint32_t b; memcpy( &b, &f, 4 ); return b; } +inline double table_bits_to_double( uint64_t bits ) { double d; memcpy( &d, &bits, 8 ); return d; } +inline uint64_t table_double_to_bits( double d ) { uint64_t b; memcpy( &b, &d, 8 ); return b; } + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +#ifndef LISTDEMO_SCHEMA_TABLE_ARENA +#define LISTDEMO_SCHEMA_TABLE_ARENA + +namespace listdemo { + +// ---- variable-length tables: tuning constants (docs/SPEC-TABLES.md) ---- +// +// The segment size and the count multiply to exactly 2^32: the u32 reference +// is the arena's hard ceiling, and these constants saturate it rather than +// leaving address space unreachable. Slab handout costs one atomic per slab, +// so per-node allocation costs no synchronization at all. + +static const uint32_t kTableSegmentBits = 22; // 4 MiB segments +static const uint32_t kTableSegmentSize = 1u << kTableSegmentBits; +static const uint32_t kTableSegmentMask = kTableSegmentSize - 1u; +static const uint32_t kTableMaxSegments = 1u << ( 32 - kTableSegmentBits ); // 1024 -> 4 GiB +static const uint32_t kTableSlabBytes = 64u * 1024u; // one atomic per slab +static const uint32_t kTableAlign = 8; // every node starts 8-aligned +static const uint32_t kTableAllocFailed = 0xFFFFFFFFu; + +// ---- THE CALLER'S ALLOCATOR (docs/SPEC-TABLES.md §6.5) ---- +// +// Every allocation the variable-length runtime makes goes through one of +// these — the arena's segments, the pack walk's identity map, the numbering's +// entry array, the packed region, and the tool path's node directory. There is +// no other call to the C library on this path, so a counting allocator sees +// every byte and a game's own heap can own all of it. +// +// It is the shape TableBlockAllocator already has (§19.1): two function +// pointers and a context the caller carries. What it adds is a CONTRACT ON +// alloc — the bytes come back ZEROED. Lock copies whole nodes, PADDING +// INCLUDED, so anything left uninitialized reaches a packed region; the default +// pair reaches that through calloc, which costs nothing measurable because a +// fresh segment is untouched pages either way. +struct TableAllocator +{ + void * ( *alloc )( void * context, int64_t bytes ); // ZEROED bytes, NULL on failure + void ( *free )( void * context, void * pointer ); + void * context; +}; + +// The default pair, and it is the one every entry point takes when the caller +// names none. It calls schema_allocate / schema_release, so a program with its +// own C-library replacement can move the floor without writing a struct at all. +inline void * table_default_alloc( void * context, int64_t bytes ) { (void) context; return schema_allocate( bytes ); } +inline void table_default_free( void * context, void * pointer ) { (void) context; schema_release( pointer ); } + +inline TableAllocator TableDefaultAllocator() +{ + TableAllocator allocator; + allocator.alloc = table_default_alloc; + allocator.free = table_default_free; + allocator.context = NULL; + return allocator; +} + +// ---- TableRef: a relocatable reference (never a machine pointer) ---- +// +// Two encodings, one slot, and the FORM says which is in force: +// +// in the arena — the node's arena offset (segment index in the high bits) +// in a region — the SELF-RELATIVE byte delta from this slot's own address, +// so a deref is one add, needs no base pointer, and a whole +// region relocates by memcpy with zero fix-up +// +// 0 is null in both, and a slot can never name the node that contains it, so +// zero names nothing real in either form. +// +// A REGION DELTA HAS NO REQUIRED SIGN (§6.3). A region is packed depth-first, +// so a node's FIRST reference points forward; every LATER reference to that +// same node points BACK at the one body it already has, which is exactly what +// makes one node one node in a region. Sharing and a back-reference are the +// same fact, and nothing validates a reference by its sign. +// +// IT IS EIGHT BYTES, SIGNED, so ONE REGION REACHES EVERYTHING (§6.3, §7): a +// four-byte slot bounded a region at 2 GiB, and the scale a cook exists for is +// *"100mbs or many gigabytes of data in Assets.bin"*. +struct TableRef +{ + int64_t value = 0; + bool null() const { return value == 0; } +}; + +// TableSlot is what Alloc hands back: usable as the node pointer (write +// fields through it) AND as the reference to store in a pointer field. +template struct TableSlot +{ + T * ptr = NULL; + TableRef ref; + T * operator->() const { return ptr; } + T & operator*() const { return *ptr; } + operator T *() const { return ptr; } + operator TableRef() const { return ref; } + bool null() const { return ptr == NULL; } +}; + +inline uint32_t TableAlignUp( uint32_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( kTableAlign - 1 ); } +inline int64_t TableAlignUp64( int64_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( int64_t( kTableAlign ) - 1 ); } + +// ---- a BYTE BUFFER's node (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// A *bytes or *string slot is a TableRef like every pointer slot, and it names +// a BLOB NODE: this eight-byte header and then the bytes, at offset eight so +// the data is eight-aligned. A *string blob carries one more zero byte after +// its data, so a region hands back a C string with no copy. The node's extent +// is the header plus its bytes, rounded to the arena's alignment like every +// node's; on the wire it is a record whose body is the bytes (§3.1). +struct TableBlob +{ + uint32_t length; + uint32_t zero; +}; + +static const int64_t kTableBlobHeader = 8; // length (u32), then four zero bytes +static const int64_t kTableBlobMaxLength = 0xFFFFFFFF; // a record's length is a u32 (§3.1) + +// the node's storage: the header, the bytes, a string's terminator, rounded +// to the arena's alignment like every node +inline int64_t TableBlobStorage( int64_t length, bool terminated ) +{ + return TableAlignUp64( kTableBlobHeader + length + ( terminated ? 1 : 0 ) ); +} + +// What a read answers: a pointer INTO the region and the length, NULL and +// zero for a null slot. Off a locked region, a loaded one or an opened cook +// the pointer is one add from the slot, and nothing is copied. +struct TableBytesView +{ + const uint8_t * data; + int64_t length; +}; + +struct TableStringView +{ + const char * data; // zero-terminated + int64_t length; +}; + +// What AllocBytes and AllocString hand back: the bytes to write through, the +// length asked for, and the reference to store in the slot — the three +// answers TableSlot gives for a table node. +struct TableBytesSlot +{ + uint8_t * data = NULL; + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +struct TableStringSlot +{ + char * data = NULL; // room for length bytes and the terminator, already zero + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +// ---- the arena: segmented, slab-handed, lock-free by ownership ---- +// +// Allocation is thread-local inside a worker's slab — no atomics on the node +// path. A worker takes its next slab with ONE compare-exchange, and a new +// segment is published with one more. Nothing ever moves: a segment, once +// allocated, lives untouched until the arena is torn down, so a T* obtained +// from Alloc stays valid while other workers allocate, and an offset stays +// correct while the arena grows. +// +// The model this DELIBERATELY refuses: one buffer under a lock, grown by +// realloc. A realloc moves the buffer under workers mid-write; offsets fix +// identity but not the raw references already resolved from them, and the +// resulting corruption is invisible until much later. Segments never move, so +// that bug class cannot be written here. +// +// Slack: at most one slab tail per worker plus one slab per segment (a slab +// that will not fit is skipped rather than split), i.e. under 2% of a segment +// plus threads x 64 KiB. That is the price of never synchronizing per node. +struct TableArena +{ + std::atomic segments[ kTableMaxSegments ]; + std::atomic cursor; // (segment << kTableSegmentBits) | bytes handed out + bool locked = false; // MONOTONIC: Lock() is one-way, there is no unlock + // THE ARENA CARRIES ITS OWN, so everything downstream of a builder — + // segments, pack map, numbering, region, node directory — allocates through + // the one pair the caller named, with nothing to thread by hand. + TableAllocator allocator; +}; + +inline void TableArenaInit( TableArena & arena, TableAllocator allocator ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + arena.segments[i].store( NULL, std::memory_order_relaxed ); + } + arena.cursor.store( 0, std::memory_order_relaxed ); + arena.locked = false; + arena.allocator = allocator; +} + +inline void TableArenaShutdown( TableArena & arena ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + uint8_t * segment = arena.segments[i].exchange( NULL, std::memory_order_acq_rel ); + if ( segment != NULL ) { arena.allocator.free( arena.allocator.context, segment ); } + } + arena.cursor.store( 0, std::memory_order_relaxed ); +} + +// one L1 load plus an add: the segment table is 8 KiB and stays hot +inline uint8_t * TableArenaAt( const TableArena & arena, uint32_t offset ) +{ + return arena.segments[ offset >> kTableSegmentBits ].load( std::memory_order_relaxed ) + ( offset & kTableSegmentMask ); +} + +// TableArenaGrabSlab hands one worker its next private slab. Returns +// kTableAllocFailed when the arena's address space or the allocator is +// exhausted — a loud refusal, never a silent smaller slab. +inline uint32_t TableArenaGrabSlab( TableArena & arena ) +{ + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t segment = cursor >> kTableSegmentBits; + uint32_t used = cursor & kTableSegmentMask; + // strictly less: a slab is never split across segments, and the tail + // is the documented slack + if ( used + kTableSlabBytes < kTableSegmentSize ) + { + if ( arena.segments[segment].load( std::memory_order_acquire ) == NULL ) + { + // THE SEGMENT COMES BACK ZEROED, which is the allocator's + // contract and not an extra pass here: Lock copies whole nodes, + // PADDING INCLUDED, so anything uninitialized reaches a packed + // region. Value-initializing a node with placement new zeroes + // its MEMBERS and not its padding, so the zeroing has to happen + // at the segment or not at all. It costs nothing measurable: a + // fresh segment is untouched pages either way, and the default + // pair's calloc has the kernel hand them over zeroed. + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, (int64_t) kTableSegmentSize ); + if ( memory == NULL ) { return kTableAllocFailed; } + uint8_t * expected = NULL; + if ( !arena.segments[segment].compare_exchange_strong( expected, memory, std::memory_order_acq_rel ) ) + { + // another worker published this segment first + arena.allocator.free( arena.allocator.context, memory ); + } + } + if ( arena.cursor.compare_exchange_weak( cursor, cursor + kTableSlabBytes, std::memory_order_acq_rel ) ) + { + return ( segment << kTableSegmentBits ) | used; + } + continue; + } + uint32_t next_segment = segment + 1; + if ( next_segment >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + arena.cursor.compare_exchange_weak( cursor, next_segment << kTableSegmentBits, std::memory_order_acq_rel ); + } +} + +// TableArenaGrabSpan reserves a SPAN of the arena's address space for one node +// larger than a slab — a BYTE BUFFER of any size (docs/SPEC-TABLES.md §2.5) — +// and allocates it as one contiguous block. It takes whole segment indices +// from the cursor, starting at the index after the cursor's so nothing else +// is ever handed out inside the span, and publishes the block under the first +// of them; the indices the span covers past that one stay NULL, which is +// enough, because only a node's START is ever resolved through the segment +// table and a blob's bytes follow its header inside the one allocation. The +// unused tail of the segment the cursor was in is slack, like a slab tail. +// Returns kTableAllocFailed when the address space or the allocator is +// exhausted — a loud refusal, never a smaller blob. +inline uint32_t TableArenaGrabSpan( TableArena & arena, int64_t bytes ) +{ + if ( bytes <= 0 || bytes > ( (int64_t) kTableMaxSegments - 2 ) * (int64_t) kTableSegmentSize ) { return kTableAllocFailed; } + const uint32_t spanned = (uint32_t) ( ( bytes + kTableSegmentSize - 1 ) >> kTableSegmentBits ); + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t start = ( cursor >> kTableSegmentBits ) + 1; + if ( start + spanned >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + uint32_t next = ( start + spanned ) << kTableSegmentBits; + if ( !arena.cursor.compare_exchange_weak( cursor, next, std::memory_order_acq_rel ) ) { continue; } + // the span is this worker's now: nothing else can publish under its + // first index, so a plain store suffices, and the block comes back + // ZEROED like every segment — the blob's bytes and its tail are zeros + // until written + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, bytes ); + if ( memory == NULL ) { return kTableAllocFailed; } + arena.segments[start].store( memory, std::memory_order_release ); + return start << kTableSegmentBits; + } +} + +// ---- TableWorker: one thread's allocation front ---- +// +// The threading contract, stated plainly: +// * Alloc on YOUR OWN worker is safe concurrently with any other worker's. +// No locks, no atomics per node. +// * Writing fields of a node ANOTHER worker allocated is your own +// synchronization problem — this runtime does not arbitrate it. +// * Lock and Save are single-threaded: call them after the workers have +// joined. +struct TableWorker +{ + TableArena * arena = NULL; + uint32_t next = 0; + uint32_t end = 0; + + template TableSlot Alloc() + { + static_assert( alignof( T ) <= kTableAlign, "a table node's alignment must fit the arena's" ); + TableSlot slot; + if ( arena == NULL || arena->locked ) { return slot; } + uint32_t bytes = TableAlignUp( (uint32_t) sizeof( T ) ); + if ( bytes > kTableSlabBytes ) { return slot; } // a node larger than a slab: refused, never split + if ( end == 0 || next + bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return slot; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + uint32_t at = next; + next += bytes; + // A NODE IS BORN IN TWO HALVES: start its lifetime in the raw + // storage, then write the declared defaults ONE MEMBER AT A TIME. + // + // It is "T", not "T{}". Value-initialising the whole aggregate says + // the same thing and costs cl O(BYTES) TO COMPILE — it expands element + // by element in its front end — while both halves here cost + // O(declarations). The slab cap below refuses a large node at RUN + // TIME and bounds nothing at compile time: the cost is paid by + // whatever T a caller instantiates this with. + // Padding is not the difference: value-initialisation zeroes MEMBERS + // and not padding either way, which is why the segment is calloc'd. + // + // TableReset is an OVERLOAD SET, one per closure member, reached from + // this template by argument-dependent lookup on T's own namespace — + // Alloc is a template and cannot spell Reset. + // + // The reset is here because ONE DEFINITION SAYS WHAT THE DECLARED + // DEFAULTS ARE, and it is Reset. Default-initialisation lands on + // the same values today, because a member with a non-zero default + // carries a member initializer that says so — but that is the class + // definition agreeing with Reset, not the arena reading it, and #320's + // fix was itself a pass that MOVED initialisation between the two. + // The arena reads the definition. + slot.ptr = new ( TableArenaAt( *arena, at ) ) T; + TableReset( *slot.ptr ); + slot.ref.value = at; + return slot; + } + + // Alloc a BYTE BUFFER's node of exactly length bytes (docs/SPEC-TABLES.md + // §2.5): the blob header and its bytes, zeroed, in this thread's slab when + // it fits and in a span of the arena's own when it does not. NULL is the + // arena locked, a length below zero or past a record's u32, or the + // allocator refusing. The offset comes back for the reference. + TableBlob * AllocBlob( int64_t length, bool terminated, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( length < 0 || length > kTableBlobMaxLength ) { return NULL; } + const int64_t bytes = TableBlobStorage( length, terminated ); + if ( bytes > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, bytes ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + } + else + { + if ( end == 0 || next + (uint32_t) bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) bytes; + } + TableBlob * blob = (TableBlob *) TableArenaAt( *arena, at ); + blob->length = (uint32_t) length; // the bytes after it are the segment's zeros + blob->zero = 0; + return blob; + } + + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries + // no type id, takes no index and has no Reset, so it goes through the same + // slab and span the blob path uses rather than through Alloc. + uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( bytes <= 0 || align > (int64_t) kTableAlign ) { return NULL; } + const int64_t rounded = TableAlignUp64( bytes ); + if ( rounded > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, rounded ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + return TableArenaAt( *arena, at ); + } + if ( end == 0 || next + (uint32_t) rounded > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) rounded; + return TableArenaAt( *arena, at ); // the segment came back zeroed + } + // a *bytes node: the bytes to write through, and the reference to store + TableBytesSlot AllocBytes( int64_t length ) + { + TableBytesSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, false, at ); + if ( blob == NULL ) { return slot; } + slot.data = (uint8_t *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } + + // a *string node: room for length bytes and the zero byte after them + TableStringSlot AllocString( int64_t length ) + { + TableStringSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, true, at ); + if ( blob == NULL ) { return slot; } + slot.data = (char *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } +}; + +// ---- TablePackMap: the pack walk's identity map (docs/SPEC-TABLES.md §3.1, §6.2) ---- +// +// ONE ENTRY PER REACHABLE NODE, and that map IS identity: a node must know +// where it landed to be named a second time, so Lock packs a shared node ONCE +// and every later reference resolves to the one body it already has. That is +// the same first-visit numbering the wire uses, so the pack order and the node +// order are one order. +// +// COLOURING AN ENTRY WHILE ITS DESCENT IS OPEN COSTS ONE BIT, and it is what +// makes a data cycle free to refuse: a reference to an entry still open is a +// cycle, and Lock returns failure rather than recursing away. The ROOT's entry +// is open for the whole walk. +// +// The map is proportional to NODES, never to bytes, and it lives on the +// AUTHORING side, where §6.5 licenses allocation. Nothing on the reading path +// ever builds one. +struct TablePackEntry +{ + const void * key; // the node's address in the graph being packed + int64_t offset; // where that node landed in the region + uint8_t open; // its descent is still open: a reference here is a cycle +}; + +struct TablePackMap +{ + TablePackEntry * entries = NULL; + int64_t capacity = 0; // a power of two, or zero while empty + int64_t count = 0; + TableAllocator allocator; // the caller's, carried from the walk that built it +}; + +inline void TablePackMapInit( TablePackMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TablePackMapShutdown( TablePackMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TablePackMapInit( map, map.allocator ); +} + +// The two walks behind Lock re-derive the SAME map from the same graph — the +// numbering is never carried between them (§3.1) — so the second starts from +// an empty map and keeps the capacity the first paid for. +inline void TablePackMapReset( TablePackMap & map ) +{ + if ( map.entries != NULL ) { memset( map.entries, 0, (size_t) map.capacity * sizeof( TablePackEntry ) ); } + map.count = 0; +} + +// open addressing, linear probing, a multiply-shift hash over the address: a +// node key is a pointer and its low bits are alignment, so the low bits alone +// would collide on every node of one type +inline int64_t TablePackMapSlot( const TablePackMap & map, const void * key ) +{ + uint64_t hash = (uint64_t) (uintptr_t) key; + hash *= 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != NULL && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TablePackEntry * TablePackMapFind( TablePackMap & map, const void * key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TablePackEntry * entry = &map.entries[ TablePackMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +// QUADRUPLING, not doubling, and the reason is measured: growth rehashes every +// entry, and on a graph of 131,071 nodes the doubling schedule spent 45% of +// Lock in rehashing alone. Quadrupling from 1024 buys 1.35x on that graph and +// keeps the map NODE-proportional (§6.2) — under 128 bytes a node at its +// worst, right after a grow, and about 64 on average. +inline bool TablePackMapGrow( TablePackMap & map ) +{ + TablePackMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 1024; + grown.entries = (TablePackEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TablePackEntry ) ); + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == NULL ) { continue; } + grown.entries[ TablePackMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// REACH a node: one probe answers both questions the walk has. A true "taken" +// says this is a FIRST visit, and the entry is now the node's, coloured open +// at "offset"; otherwise the entry is the one the node already has, and its +// open bit says cycle or sharing. NULL is an allocation failure, and it is a +// refusal like any other: Lock fails rather than packing a graph it cannot +// track. +// +// It is one call and not a find followed by an insert because the walk asks +// this question twice per node — once to measure, once to pack — and every +// probe is a miss into a table larger than L2. +inline TablePackEntry * TablePackMapReach( TablePackMap & map, const void * key, int64_t offset, bool & taken, int64_t & slot ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TablePackMapGrow( map ) ) { return NULL; } + } + slot = TablePackMapSlot( map, key ); + TablePackEntry * entry = &map.entries[slot]; + taken = entry->key != key; // an empty slot is a first visit; the key is never NULL + if ( taken ) + { + entry->key = key; + entry->offset = offset; + entry->open = 1; + map.count++; + } + return entry; +} + +// The descent finished: the node keeps its entry — identity outlives the +// descent — and stops being a cycle. The "hint" is the slot Reach returned, and it +// is checked against the key rather than trusted, so a rehash between the two +// costs a second probe instead of correctness. +inline void TablePackMapClose( TablePackMap & map, const void * key, int64_t hint ) +{ + if ( hint >= 0 && hint < map.capacity && map.entries[hint].key == key ) + { + map.entries[hint].open = 0; + return; + } + TablePackEntry * entry = TablePackMapFind( map, key ); + if ( entry != NULL ) { entry->open = 0; } +} + +// ---- resolution contexts: which encoding a walk is reading ---- + +struct TableArenaCtx { const TableArena * arena; }; +struct TableRegionCtx {}; + +// ---- a BYTE BUFFER's resolution (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// The same two encodings a table pointer has, resolved the same way: a +// self-relative delta in a region — one add, no base — and an arena offset +// while the builder is mutable. The blob is reached through its header, and a +// view is the header plus eight and the header's first word. Nothing here +// allocates and nothing copies: off a locked region, a loaded one or an +// opened cook the view points INTO the region. +inline const TableBlob * TableBlobAt( const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableRegionCtx &, const TableRef & ref ) { return TableBlobAt( ref ); } +inline const TableBlob * TableBlobAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} + +inline TableBytesView TableBytesViewOf( const TableBlob * blob ) +{ + TableBytesView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const uint8_t *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} +inline TableStringView TableStringViewOf( const TableBlob * blob ) +{ + TableStringView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const char *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} + +// the const form's hot path: one add, no base +inline TableBytesView TableBytesAt( const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ref ) ); } +inline TableStringView TableStringAt( const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ref ) ); } +// and the context forms a walk uses: a region context, an arena context, or +// the arena itself while the builder is mutable +template inline TableBytesView TableBytesAt( const Ctx & ctx, const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ctx, ref ) ); } +template inline TableStringView TableStringAt( const Ctx & ctx, const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ctx, ref ) ); } + +// allocate a blob in the arena and point the slot at it; the slot holds the +// arena offset, as every slot does while the builder is mutable +inline uint8_t * TableBytesEmplace( TableWorker & worker, TableRef & slot, int64_t length ) +{ + TableBytesSlot allocated = worker.AllocBytes( length ); + slot = allocated.ref; + return allocated.data; +} +// the text is copied in when one is given; a NULL text leaves the zeros for +// the caller to fill +inline char * TableStringEmplace( TableWorker & worker, TableRef & slot, const char * text, int64_t length ) +{ + TableStringSlot allocated = worker.AllocString( length ); + slot = allocated.ref; + if ( allocated.data != NULL && text != NULL && length > 0 ) { memcpy( allocated.data, text, (size_t) length ); } + return allocated.data; +} + +// ---- the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table, and a +// pointer field rides as an INDEX into it under kind 17. The encoding is +// flat: no pointer edge is a nesting level, so a chain's length is not a depth, +// and two references to one node are one node. +// +// THE FIELD RIDES ONCE: an L with sixty-four bits of capability frames a +// numbering of any size, so the whole numbering is one contiguous payload and a +// save's node bodies have no aggregate ceiling. + +static const uint64_t kTableNodeIndexNull = 0; // absence and null are one value +static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts the table + +// The not-materialized sentinel (§6.3): a record whose type id this build could +// not name. Distinct from every real offset including the root's 0, so an index +// resolving through it yields NULL and can never fabricate the root. +static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; + +// ---- the numbering, on the SAVE side ---- +// +// One entry per reachable node in FIRST-VISIT order, so entry k is node index +// k + 2. The two thunks are what let one loop write a table of mixed types: the +// numbering walk knows each target's type STATICALLY at the site it numbers it, +// so it stores the instantiation there and the loop never asks what a node is. +struct TableNumbering; + +struct TableNodeEntry +{ + const void * node; + uint64_t type_id; + // the type id's MESSAGE-FORM SLOT (docs/SPEC-TABLES.md §3.3), stored where + // the numbering walk stores the id itself and for the same reason: the + // target's type is known STATICALLY at the site that numbers it, so a + // form 2 save reads the slot out of the entry instead of looking an id up. + // Every pointer target's type id is an entry of the announcement, which is + // what makes the slot a compile-time fact of a POINTERED message too. + uint64_t type_slot; + int64_t ( * measure )( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ); + bool ( * save )( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ); +}; + +struct TableNumbering +{ + TablePackMap seen; // node -> index; the ROOT is index 1, open for the whole walk + TableNodeEntry * entries = NULL; + int64_t count = 0; + int64_t capacity = 0; +}; + +// The numbering allocates through the map's pair rather than carrying a second +// copy of it: one numbering is one walk, and a walk has one allocator. +inline void TableNumberingInit( TableNumbering & n, TableAllocator allocator ) +{ + TablePackMapInit( n.seen, allocator ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +inline void TableNumberingShutdown( TableNumbering & n ) +{ + TableAllocator allocator = n.seen.allocator; + TablePackMapShutdown( n.seen ); + allocator.free( allocator.context, n.entries ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +// The index a numbered node was given, for the save that writes it into a +// pointer slot. False means the two walks disagree about the graph, which is a +// refusal and never a guess. +inline bool TableNumberingIndex( const TableNumbering & n, const void * node, uint64_t & index ) +{ + if ( n.seen.capacity == 0 ) { return false; } + const TablePackEntry & entry = n.seen.entries[ TablePackMapSlot( n.seen, node ) ]; + if ( entry.key != node ) { return false; } + index = (uint64_t) entry.offset; + return true; +} + +inline bool TableNumberingAppend( TableNumbering & n, const TableNodeEntry & entry ) +{ + if ( n.count == n.capacity ) + { + // GROW BY COPY, never by realloc: the allocator hook is a PAIR, and a + // game's heap is not required to have a resize primitive at all. The + // schedule quadruples, so the copying is amortized to a constant per + // entry and the growth is the same growth it always was. + int64_t capacity = n.capacity != 0 ? n.capacity * 4 : 256; + TableAllocator allocator = n.seen.allocator; + TableNodeEntry * grown = (TableNodeEntry *) allocator.alloc( allocator.context, capacity * (int64_t) sizeof( TableNodeEntry ) ); + if ( grown == NULL ) { return false; } + if ( n.entries != NULL ) + { + memcpy( grown, n.entries, (size_t) n.count * sizeof( TableNodeEntry ) ); + allocator.free( allocator.context, n.entries ); + } + n.entries = grown; + n.capacity = capacity; + } + n.entries[n.count++] = entry; + return true; +} + +// The thunks the numbering stores. Each resolves to the closure member's own +// MeasureBody / SaveBodyFields through an overload set in the member's DECLARING +// file, reached by argument-dependent lookup at instantiation — the same bridge +// the arena's TableReset uses, and the reason a numbering may span the files of +// one unit without any file naming another's members. +template +inline int64_t TableNodeMeasureThunk( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ) +{ + return TableNodeMeasure( *(const Ctx *) ctx, numbering, ids, *(const T *) node ); +} + +template +inline bool TableNodeSaveThunk( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ) +{ + return TableNodeSave( *(const Ctx *) ctx, numbering, w, ids, *(const T *) node ); +} + +// ---- a BYTE BUFFER's record (docs/SPEC-TABLES.md §2.5, §3.1) ---- +// +// A blob rides as a node record under one of two RESERVED type ids — the fold +// a table's name takes, over the keywords "bytes" and "string", which no table +// can be named — with the bytes as its body and nothing framed inside. These +// two thunks are what the numbering stores for a blob, as it stores a +// member's codec for a table: the length, and the bytes verbatim. +static const uint64_t kTableBytesTypeId = 0x2f2ec0474f1c4fe4ull; // fnv1a64( "bytes" ) +static const uint64_t kTableStringTypeId = 0x704be0d8faaffc58ull; // fnv1a64( "string" ) + +template +inline int64_t TableBlobMeasureThunk( const void *, const TableNumbering &, TableIds &, const void * node ) +{ + return (int64_t) ( (const TableBlob *) node )->length; +} + +template +inline bool TableBlobSaveThunk( const void *, const TableNumbering &, TableWriter & w, TableIds &, const void * node ) +{ + const TableBlob * blob = (const TableBlob *) node; + w.raw( (const void *) ( blob + 1 ), (int64_t) blob->length ); + return true; +} + +// TableNodeTableMeasure and TableNodeTableSave are the framing, and they are +// ONE fill rule written twice — measure derives it from the graph and save +// derives the same one, which is what makes measure == save hold across a +// pointer graph (§3.1). +// +// The field rides ONCE, under the reserved id, kind 12: the payload opens with +// the count and then carries the records back to back, each a type id +// REFERENCE, a length and a body. The reserved id is interned BEFORE the +// records, and a record's type id before its body, which is the first-use order +// the trailer is written in (§3). +template +inline int64_t TableNodeTablePayload( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + int64_t payload = TableLebBytes( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + payload += TableLebBytes( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return -1; } + payload += TableLebBytes( (uint64_t) body ) + body; + } + return payload; +} + +template +inline int64_t TableNodeTableMeasure( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return 0; } // a root that reaches no nodes writes none of them + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return -1; } + return TableLebBytes( ref ) + 1 + TableLebBytes( (uint64_t) payload ) + payload; +} + +template +inline bool TableNodeTableSave( const Ctx & ctx, TableWriter & w, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return true; } + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return false; } + w.putleb( ref ); + w.put8( 12 ); // kind 12 is the opaque byte payload: a reader that cannot name the id skips by L + w.putleb( (uint64_t) payload ); + w.putleb( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + w.putleb( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return false; } + w.putleb( (uint64_t) body ); + if ( !n.entries[k].save( (const void *) &ctx, n, w, ids, n.entries[k].node ) ) { return false; } + } + return true; +} + +// ---- the numbering, on the LOAD side: a region's NODE DIRECTORY (§6.3) ---- +// +// The wire's numbering made resident: one entry per numbered node, in index +// order, position i describing node index i + 1 — so position 0 is the ROOT at +// offset 0. It is ATTRIBUTION, and attribution is separable: nothing that reads +// a structure touches it, a deref is one add on a self-relative offset, and a +// caller may release it once Load returns. +struct TableNodeDirEntry +{ + uint64_t offset; + uint64_t type_id; +}; + +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; + +// TableNodeMap is what a pointer slot resolves through while a body decodes. +struct TableNodeMap +{ + uint8_t * base = NULL; + const TableNodeDirEntry * entries = NULL; + int64_t count = 0; // the ROOT's entry included, so it is records + 1 + bool good = false; // the node table read whole; a numbering that failed resolves nothing + // WHERE THE NODES LIVE, and therefore what a resolved slot holds: a region + // takes the SELF-RELATIVE delta so a deref is one add, and the tool's + // builder path takes the node's ARENA OFFSET (§6.3). + bool arena = false; + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. + TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; +}; + +// TableNodeResolve places one node index in a pointer slot, and every failure +// is one of §4's events with the pointer left null. The declared TARGET type id +// is checked at every index, the root's included: the root carries no record +// and therefore no wire type id, so the READER'S OWN root type is what the +// claim is checked against. +inline void TableNodeResolve( const TableNodeMap & map, TableRef & slot, uint64_t index, uint64_t target, TableReport * report ) +{ + slot.value = 0; + if ( index == kTableNodeIndexNull || !map.good ) { return; } + if ( index - 1 >= (uint64_t) map.count ) + { + report->malformed = true; // an index above node_count + 1 + return; + } + const TableNodeDirEntry & entry = map.entries[index - 1]; + if ( entry.offset == kTableNodeAbsent ) + { + // a node whose type id this build could not name KEEPS ITS INDEX, and + // every pointer naming it reads null. The unknown was counted once, at + // the node, not once per pointer. + return; + } + if ( entry.type_id != target ) + { + report->kind_mismatch++; + return; + } + slot.value = map.arena ? (int64_t) entry.offset + : (int64_t) ( ( map.base + entry.offset ) - (const uint8_t *) &slot ); +} + +// ---- the record SCAN, and it is the whole of load's bound (§3.1) ---- +// +// Reading follows no reference. The scan walks the root body's top-level fields, +// finds the ONE under the reserved id, and reads records out of its payload in +// order — the field rides once, so nothing is copied to make a body contiguous +// and the generated body decoder never learns the transport exists. +struct TableNodeScan +{ + TableReader fields; // over the ROOT body, skipping past everything else + const uint8_t * payload; // the node-table field's payload + int64_t payload_size; + int64_t payload_offset; + bool opened; // the root body has been walked for the field + uint64_t declared; + int64_t records; + bool present; // the root body carries a node table at all + bool malformed; + const TableIdTable * ids; +}; + +inline TableNodeScan TableNodeScanBegin( const uint8_t * body, int64_t size, TableReport * report, const TableIdTable * ids ) +{ + TableNodeScan s = { TableReader( body, size, report, ids ), NULL, 0, 0, false, 0, 0, false, false, ids }; + return s; +} + +// find the node-table field, or answer false when the root body has none. A +// body carrying an id more than once is legal input and THE LAST OCCURRENCE +// WINS (docs/SPEC-TABLES.md §3), so the walk runs to the terminator and keeps +// the last rather than stopping at the first. +inline bool TableNodeScanOpen( TableNodeScan & s ) +{ + if ( s.opened ) { return false; } + s.opened = true; + for ( ;; ) + { + uint64_t ref = 0; + if ( !s.fields.getleb( ref ) ) { break; } + if ( ref == 0 ) { break; } // the terminator + if ( s.ids == NULL || ref > (uint64_t) s.ids->count ) { break; } + const uint64_t id = s.ids->at( ref ); + if ( !s.fields.has( 1 ) ) { break; } + const uint8_t kind = s.fields.get8(); + if ( id == kTableNodeTableFieldId ) + { + s.present = true; + if ( kind != 12 ) { s.malformed = true; return false; } + uint64_t length = 0; + if ( !s.fields.getleb( length ) || !s.fields.room( length ) ) { s.malformed = true; return false; } + s.payload = s.fields.buffer + s.fields.offset; + s.payload_size = (int64_t) length; + s.fields.offset += (int64_t) length; + continue; + } + if ( !s.fields.skip( kind ) ) { break; } + } + if ( s.payload == NULL ) { return false; } + TableReader head( s.payload, s.payload_size, s.fields.report, s.ids ); + if ( !head.getleb( s.declared ) ) { s.malformed = true; return false; } + s.payload_offset = head.offset; + return true; +} + +// the next record, or false at the end of the table — s.malformed says whether +// the end was the end or the framing giving out +inline bool TableNodeScanNext( TableNodeScan & s, uint64_t & type_id, const uint8_t * & body, int64_t & length ) +{ + if ( !s.opened && !TableNodeScanOpen( s ) ) { return false; } + if ( s.payload == NULL || s.payload_offset >= s.payload_size ) { return false; } + TableReader rec( s.payload, s.payload_size, s.fields.report, s.ids ); + rec.offset = s.payload_offset; + uint64_t ref = 0; + if ( !rec.getleb( ref ) || ref == 0 || s.ids == NULL || ref > (uint64_t) s.ids->count ) + { + s.malformed = true; // a type id reference of 0, or one past the table + return false; + } + type_id = s.ids->at( ref ); + uint64_t declared_length = 0; + if ( !rec.getleb( declared_length ) ) + { + s.malformed = true; // a record whose length is damaged + return false; + } + if ( declared_length > (uint64_t) ( s.payload_size - rec.offset ) ) + { + s.malformed = true; // a record whose length runs past its field + return false; + } + body = s.payload + rec.offset; + length = (int64_t) declared_length; + s.payload_offset = rec.offset + length; + s.records++; + return true; +} + +// The record scan is AUTHORITATIVE: node_count is data from the wire, and a +// count that disagrees with the scan is malformed. Nothing is sized from it +// before the scan has confirmed it. +inline bool TableNodeScanWhole( TableNodeScan & s ) +{ + if ( s.malformed ) { return false; } + if ( !s.present ) { return true; } // no node table at all is not a broken one + return s.declared == (uint64_t) s.records; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_ARENA + +#ifndef LISTDEMO_SCHEMA_TABLE_EXTENT +#define LISTDEMO_SCHEMA_TABLE_EXTENT + +namespace listdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_EXTENT + +#ifndef LISTDEMO_SCHEMA_TABLE_MAP +#define LISTDEMO_SCHEMA_TABLE_MAP + +namespace listdemo { + +// ---- a MAP: a sorted entry array, and the lookup over it (§2.8) ---- +// +// On the wire, in a region and in a cook a map is an array of one generated +// ENTRY table held in ascending key order. What this adds is Find — a binary +// search over that array where it lies — and a builder that inserts, replaces +// and erases by key. Nothing here is stored: a region and a cook carry the +// array and the count, and not one byte about a hash or a probe. + +// entries carved from ONE call to the allocator pair; a new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableMapSegmentEntries = 32; + +// TableDeclRef names a type in an unevaluated context and is never defined — +// what 's declval is for, without the include the generated corpus +// refuses to pay for (the iterator_traits note, §13.9). +template T & TableDeclRef(); + +// THE ORDER IS TOTAL, AND IT IS THE SAME IN NINE LANGUAGES (§2.8). Integers +// compare by VALUE, signed for the signed kinds and unsigned for the unsigned. +// Strings compare by BYTES, unsigned, a shorter string that is a prefix of a +// longer one first: memcmp over the common length, then the lengths. Never a +// locale, never a code point, never a case fold. +inline int TableKeyOrder( uint64_t a, uint64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( int64_t a, int64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( const char * a, int32_t a_length, const char * b, int32_t b_length ) +{ + const int32_t common = a_length < b_length ? a_length : b_length; + if ( common > 0 ) + { + const int order = memcmp( (const void *) a, (const void *) b, (size_t) common ); + if ( order != 0 ) { return order < 0 ? -1 : 1; } + } + return a_length < b_length ? -1 : ( a_length > b_length ? 1 : 0 ); +} + +// the length of a NUL-terminated key at a call site, bounded by the storage it +// has to fit: a key one byte longer than the bound is refused, never truncated +inline int32_t TableKeyLength( const char * key, int32_t bound ) +{ + if ( key == NULL ) { return 0; } + for ( int32_t i = 0; i <= bound; i++ ) { if ( key[i] == 0 ) { return i; } } + return bound + 1; // longer than the bound: the caller refuses it +} + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.8, §7.2) ---- +// +// An int64 self-relative reference to the entry array and an int32 count, then +// padding to eight. The reference is a TableRef like a pointer's: in the arena +// it names the builder's HEAD, in a region it is the delta from the slot to +// the first entry, and 0 is the empty map in both. +template struct TableMap +{ + TableRef entries; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Entry * Entries() const + { + return entries.value != 0 ? (const Entry *) ( (const uint8_t *) &entries + entries.value ) : NULL; + } + int32_t size() const { return count; } + + // FIND: floor( log2 n ) + 1 key compares, in place, no allocation. NULL + // when absent, and on a map[K]*T the RESOLVED pointer, which is what a + // pointer field's accessor answers. + template const Entry * FindEntry( Key key ) const + { + const Entry * base = Entries(); + int32_t low = 0, high = count; + while ( low < high ) + { + const int32_t mid = low + ( high - low ) / 2; + const int order = TableEntryOrder( base[mid], key ); + if ( order == 0 ) { return base + mid; } + if ( order < 0 ) { low = mid + 1; } else { high = mid; } + } + return NULL; + } + // the return type is DEDUCED, so it is worked out when a call site + // instantiates Find and not when the holder's record declares the slot — + // which is what lets the entry's own overloads be declared after it + template auto Find( Key key ) const + { + return TableEntryFound( FindEntry( key ) ); + } + + // ---- iteration: ASCENDING key order, the key beside the value ---- + // + // A proxy BY VALUE, the keyed array's shape (§2.4): for ( auto [ key, + // value ] : map ). It carries no iterator_traits, for the reason + // TableKeyed's does not (§13.9). + struct ConstEntry + { + decltype( TableEntryKey( TableDeclRef() ) ) key; + decltype( TableEntryFound( (const Entry *) NULL ) ) value; + }; + + struct ConstIterator + { + const Entry * at; + ConstEntry operator*() const { return ConstEntry{ TableEntryKey( *at ), TableEntryFound( at ) }; } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Entries() }; } + ConstIterator end() const { return ConstIterator{ Entries() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.8, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first entry is inserted. Each +// segment is a fixed number of entries carved from one call to the allocator +// pair. An entry's address is stable for the arena's life, so a value handed +// back by an insert stays valid while other entries arrive. +struct TableMapHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an insert appends into + int32_t live; + int32_t dead; +}; + +template struct TableMapSegment +{ + TableRef next; + int32_t used; // entries carved from this segment + int32_t padding; + uint32_t dead[ ( kTableMapSegmentEntries + 31 ) / 32 ]; // Erase marks one bit, never the entry + Entry entries[ kTableMapSegmentEntries ]; +}; + +inline bool TableMapSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// ---- the ORDERED CURSOR the four writing walks read (§2.8) ---- +// +// Measure, Save, Lock and Cook each write a map's entries in ascending key +// order with no key twice, deriving the order from the builder's entries as +// each walk derives the numbering (§3.1). Nothing passes between them, so +// measure == save over a map is a real check on two sorts agreeing. +// +// A REGION is already sorted, so its cursor is the array in place and +// allocates nothing. The BUILDER's is the sort: an array of entry pointers +// allocated through the pair and released before the walk returns, because +// sorting the segments themselves would move entries whose addresses a caller +// holds. +template struct TableMapCursor +{ + const Entry * const * order = NULL; // the builder's form: sorted pointers + const Entry * entries = NULL; // the region's form: the array in place + int32_t count = 0; + TableAllocator allocator; + bool ok = false; + const Entry * operator[]( int32_t index ) const + { + return order != NULL ? order[index] : entries + index; + } +}; + +// heapsort: O( n log n ) once per map, no recursion, no allocation past the +// pointer array the caller already paid for +template inline void TableMapSort( const Entry ** order, int32_t count ) +{ + for ( int32_t start = count / 2 - 1; start >= 0; start-- ) + { + int32_t root = start; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= count ) { break; } + if ( child + 1 < count && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * swap = order[root]; order[root] = order[child]; order[child] = swap; + root = child; + } + } + for ( int32_t end = count - 1; end > 0; end-- ) + { + const Entry * swap = order[0]; order[0] = order[end]; order[end] = swap; + int32_t root = 0; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= end ) { break; } + if ( child + 1 < end && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * hold = order[root]; order[root] = order[child]; order[child] = hold; + root = child; + } + } +} + +// the REGION form: the array is already sorted, so the cursor is the array +template +inline TableMapCursor TableMapOrder( const TableRegionCtx &, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.entries = map.Entries(); + cursor.count = map.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: gather the LIVE entries out of the segment chain in +// insertion order, then sort. A dead entry costs nothing on any wire (§2.8). +template +inline TableMapCursor TableMapOrder( const TableArena & arena, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.allocator = arena.allocator; + cursor.count = map.count; + if ( map.entries.value == 0 || map.count <= 0 ) { cursor.ok = map.count == 0; cursor.count = 0; return cursor; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + if ( head->live != map.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + const Entry ** order = (const Entry **) arena.allocator.alloc( arena.allocator.context, (int64_t) map.count * (int64_t) sizeof( const Entry * ) ); + if ( order == NULL ) { return cursor; } + int32_t at = 0; + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 && at < map.count ) + { + const TableMapSegment * segment = (const TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used && at < map.count; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + order[at++] = segment->entries + i; + } + segment_ref = segment->next; + } + if ( at != map.count ) + { + arena.allocator.free( arena.allocator.context, order ); + return cursor; + } + TableMapSort( order, map.count ); + cursor.order = order; + cursor.ok = true; + return cursor; +} + +template +inline TableMapCursor TableMapOrder( const TableArenaCtx & ctx, const TableMap & map ) +{ + return TableMapOrder( *ctx.arena, map ); +} + +template inline void TableMapRelease( TableMapCursor & cursor ) +{ + if ( cursor.order != NULL ) { cursor.allocator.free( cursor.allocator.context, (void *) cursor.order ); } + cursor.order = NULL; +} + +// ---- the builder's five (§2.8) ---- +// +// Insert APPENDS after one LINEAR SCAN of the live entries for the key it may +// replace, Find is that same scan, and Erase is the scan and one bit. The +// builder builds NO INDEX, and that is a rule: the sort happens once, at Lock, +// Save or Cook, and every lookup that matters runs over the sorted region. + +// the head, allocated when the first entry is inserted +template +inline TableMapHead * TableMapReach( TableWorker & worker, TableMap & map ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( map.entries.value != 0 ) { return (TableMapHead *) TableArenaAt( *worker.arena, (uint32_t) map.entries.value ); } + uint32_t at = 0; + TableMapHead * head = (TableMapHead *) worker.AllocRaw( (int64_t) sizeof( TableMapHead ), (int64_t) alignof( TableMapHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + map.entries.value = (int64_t) at; + return head; +} + +// one entry's storage, appended: the current segment when it has room, a new +// one carved from one call to the pair when it does not +template +inline Entry * TableMapAppend( TableWorker & worker, TableMapHead * head, TableMap & map ) +{ + TableMapSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableMapSegmentEntries ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableMapSegment *) worker.AllocRaw( (int64_t) sizeof( TableMapSegment ), (int64_t) alignof( TableMapSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableMapSegment * previous = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Entry * entry = segment->entries + segment->used; + segment->used++; + head->live++; + map.count++; + return entry; +} + +// the LINEAR SCAN: the live entries in insertion order, O( n ) key compares +template +inline Entry * TableMapScan( const TableArena & arena, const TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) == 0 ) { return segment->entries + i; } + } + segment_ref = segment->next; + } + return NULL; +} + +// ERASE marks the entry DEAD, one bit in the segment's slot and not in the +// entry table, and decrements the live count. Its storage is reclaimed at +// RESET and never reused mid-build, because reusing a slot would make "an +// entry's address is stable" false for exactly one case. +template +inline bool TableMapErase( TableArena & arena, TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return false; } + TableMapHead * head = (TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) != 0 ) { continue; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + map.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INSERTION order, live entries only (§2.8) ---- +template struct TableMapEach +{ + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableMapSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableMapSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + auto operator*() const { return TableEntryEach( segment->entries + index ); } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableMapEach TableMapEachOf( const TableArena & arena, const TableMap & map ) +{ + TableMapEach each = { &arena, TableRef() }; + if ( map.entries.value != 0 ) + { + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + each.first = head->first; + } + return each; +} + +// ---- the LOAD side: where a decoded entry lands (§2.8) ---- +// +// THE READER TRUSTS NOTHING and spends one compare per entry. Every load path +// applies the same rules and produces one report (§4), so the region load of +// §6.5 and LoadBuilder never disagree about a wire. These two shapes are what +// makes that true with one generated decoder: a REGION carves the entry array +// out of the holder node's own extent, and the TOOL's path appends into the +// builder's arena, and the decoder above them cannot tell which it has. + +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. + +// TableMapFill is one map field being decoded: where the next entry lands, and +// the entry that last LANDED, which is what the ascending check compares +// against. +template struct TableMapFill +{ + TableMap * map = NULL; + Entry * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; +}; + +template +inline TableMapFill TableMapFillBegin( const TableNodeMap & nodes, TableMap & map, uint32_t n ) +{ + TableMapFill fill; + fill.map = ↦ + map.entries.value = 0; + map.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Entry ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Entry ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Entry *) base; + fill.capacity = (int32_t) n; + map.entries.value = (int64_t) ( base - (const uint8_t *) &map.entries ); + fill.ok = true; + return fill; +} + +// the entry that last LANDED — NULL before the first +template inline Entry * TableMapFillLast( TableMapFill & fill ) +{ + if ( fill.map->count <= 0 ) { return NULL; } + if ( fill.array != NULL ) { return fill.array + ( fill.map->count - 1 ); } + return TableMapLive( *fill.worker->arena, *fill.map, fill.map->count - 1 ); +} + +// the next slot, at the entry type's declared defaults +template inline Entry * TableMapFillNext( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + if ( fill.map->count >= fill.capacity ) { return NULL; } + Entry * entry = fill.array + fill.map->count; + TableReset( *entry ); + fill.map->count++; + return entry; + } + TableMapHead * head = TableMapReach( *fill.worker, *fill.map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( *fill.worker, head, *fill.map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// A MAP WITH HALF ITS KEYS IS NOT A MAP (§2.8): at the first entry whose key +// kind disagrees with the reader's declaration the map resets to EMPTY, one +// kind_mismatch is counted for the map, and its remaining bytes are skipped. +template inline void TableMapFillReset( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + fill.map->entries.value = 0; + fill.map->count = 0; + return; + } + if ( fill.map->entries.value != 0 ) + { + TableMapHead * head = (TableMapHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.map->entries.value ); + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + } + fill.map->count = 0; +} + +// an EMPTY map's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableMapFillEnd( TableMapFill & fill ) +{ + if ( fill.array != NULL && fill.map->count == 0 ) { fill.map->entries.value = 0; } +} + +// the k-th LIVE entry of a builder map, in insertion order — what the tool +// path's ascending check compares against +template +inline Entry * TableMapLive( const TableArena & arena, const TableMap & map, int32_t index ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + int32_t at = 0; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( at == index ) { return segment->entries + i; } + at++; + } + segment_ref = segment->next; + } + return NULL; +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.8, §6.5) ---- +// +// LoadMeasure's term for a map is N x sizeof( Entry ) rounded to +// alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this +// reads no field: it walks the map's own header and, where an entry's value +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect, +// and a refusal carries its reason (§6.5). +// A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its +// own L and the body's terminator, and under this form's variable lengths that +// footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a +// map's L can carry, and therefore what a LoadMeasure may be asked for. +static const int64_t kTableMapEntryFloor = 2; + +inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry + at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); + at += (int64_t) n * entry_size; + if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// ---- the TEXT form's placement (docs/SPEC-TABLES.md §2.8, §16) ---- +// +// The text is a plain JSON object keyed by the KEY, and the generic walk fills +// it through the ENTRY'S OWN descriptor — so all it needs from here is one +// entry at one key, handed back at its defaults. It is the builder's Insert +// with the ENTRY returned rather than its value, because the walk writes the +// value through a field row and not through a typed pointer. +template +inline Entry * TableMapPlace( TableWorker & worker, TableMap & map, Key key ) +{ + if ( worker.arena == NULL ) { return NULL; } + Entry * found = TableMapScan( *worker.arena, map, key ); + if ( found != NULL ) + { + TableResetMapValue( *found ); // a repeated key is LAST-WINS, whole + return found; + } + TableMapHead * head = TableMapReach( worker, map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( worker, head, map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// ---- the OPTIONAL RUNTIME INDEX (§2.8) ---- +// +// Open addressing with LINEAR PROBING over the sorted array, built AT LOAD for +// a map large enough that log n compares over a cold array cost more than one +// hash and a probe. IT IS NEVER STORED: the caller measures it, owns its +// storage, builds it in one pass and releases it whenever. +// +// ITS HASH AND ITS LOAD FACTOR ARE NOT A CROSS-PORT CONTRACT, and that is a +// rule. What a port is held to is the CONTRACT of the lookup: the same value +// the sorted array's Find returns for the same key, and no allocation past the +// storage the caller handed in. +struct TableMapIndex +{ + int32_t * slots = NULL; // entry indices, +1; 0 is an empty slot + int32_t capacity = 0; + bool good = false; +}; + +// this runtime's own, and no port reproduces it: fnv1a64 over the key's bytes +inline uint64_t TableMapHash( const void * bytes, int32_t length ) +{ + uint64_t hash = 0xCBF29CE484222325ull; + const uint8_t * at = (const uint8_t *) bytes; + for ( int32_t i = 0; i < length; i++ ) { hash ^= (uint64_t) at[i]; hash *= 0x100000001B3ull; } + return hash; +} +inline uint64_t TableMapHash( uint64_t key ) { return TableMapHash( (const void *) &key, (int32_t) sizeof( key ) ); } + +// this runtime's own load factor, and no port reproduces it either: the next +// power of two at or above twice the count, so a probe run stays short +inline int32_t TableMapIndexSlots( int32_t count ) +{ + int32_t slots = 8; + while ( slots < count * 2 ) { slots *= 2; } + return slots; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_MAP + +#ifndef LISTDEMO_SCHEMA_TABLE_LIST +#define LISTDEMO_SCHEMA_TABLE_LIST + +namespace listdemo { + +// ---- an UNBOUNDED ARRAY: a counted array whose count the data decides (§2.9) ---- +// +// On the wire, in a region and in a cook a list is the kind 14 body a [..N]T +// writes, its elements by-value records inside the holder's node extent. What +// this adds is the slot, a builder that appends into segments that never +// move, and a const surface that indexes and iterates in place. There is no +// sort, no key and no lookup: the order is INSERTION order, and it is +// identity the way position is identity in a fixed array. + +// elements carved from ONE call to the allocator pair. A new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableListSegmentElements = 32; + +// THE ELEMENT STORAGE: T itself, and a TableRef slot for a []*T, whose +// elements are references exactly as a pointer field's slot is (§2.1) +template struct TableListStorage { typedef T Element; }; +template struct TableListStorage { typedef TableRef Element; }; + +// WHAT THE CONST FORM ANSWERS: the element by reference, and on a []*T the +// RESOLVED pointer, one add on the self-relative delta, NULL for a null slot, +// exactly as At answers it (§6.2, §6.3) +template struct TableListConst +{ + typedef const T & Result; + static Result At( const T * element ) { return *element; } +}; +template struct TableListConst +{ + typedef const T * Result; + static Result At( const TableRef * element ) + { + return element->value != 0 ? (const T *) ( (const uint8_t *) element + element->value ) : NULL; + } +}; + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.9, §7.2) ---- +// +// An int64 self-relative reference to the element array and an int32 count, +// then padding to eight. The reference is a TableRef like a pointer's: in the +// arena it names the builder's HEAD, in a region it is the delta from the slot +// to the first element, and 0 is the empty list in both. It is the map's slot +// exactly, because it is the same two facts. +template struct TableList +{ + typedef typename TableListStorage::Element Element; + + TableRef elements; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Element * Elements() const + { + return elements.value != 0 ? (const Element *) ( (const uint8_t *) &elements + elements.value ) : NULL; + } + int32_t size() const { return count; } + + // INDEXING IS BOUNDS-CHECKED IN EVERY BUILD (§2.4, §2.9): the extent is a + // number that CAME FROM A FILE, so an index past it is not a mistake a + // release build gets to make cheaply. There is no undefined-behavior path + // here in any configuration. The assert carries the message where a + // debugger can read it and NDEBUG removes that. The fatal is what stands + // after it. Both go through the hooks: define schema_assert and + // schema_fatal and this refusal lands in your own handler. + void RefuseIndex( int32_t index ) const + { + if ( (uint32_t) index >= (uint32_t) count ) + { + schema_assert( false && "an unbounded array is indexed inside its count, which came from a file" ); + schema_fatal(); + } + } + typename TableListConst::Result operator[]( int32_t index ) const + { + RefuseIndex( index ); + return TableListConst::At( Elements() + index ); + } + + // ---- iteration: INDEX order, the element and no key ---- + // + // It carries no iterator_traits, for the reason TableKeyed's does not + // (§13.9). + struct ConstIterator + { + const Element * at; + typename TableListConst::Result operator*() const { return TableListConst::At( at ); } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Elements() }; } + ConstIterator end() const { return ConstIterator{ Elements() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.9, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first element is added. Each +// segment is a fixed number of elements carved from one call to the allocator +// pair. An element's address is stable for the arena's life, so a T * handed +// back by Add stays valid while other elements arrive. +struct TableListHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an Add appends into + int32_t live; + int32_t dead; +}; + +template struct TableListSegment +{ + TableRef next; + int32_t used; // elements carved from this segment + int32_t padding; + uint32_t dead[ ( kTableListSegmentElements + 31 ) / 32 ]; // Erase marks one bit, never the element + Element elements[ kTableListSegmentElements ]; +}; + +inline bool TableListSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// the head, allocated when the first element is added +template +inline TableListHead * TableListReach( TableWorker & worker, TableList & list ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( list.elements.value != 0 ) { return (TableListHead *) TableArenaAt( *worker.arena, (uint32_t) list.elements.value ); } + uint32_t at = 0; + TableListHead * head = (TableListHead *) worker.AllocRaw( (int64_t) sizeof( TableListHead ), (int64_t) alignof( TableListHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + list.elements.value = (int64_t) at; + return head; +} + +// one element's storage, appended: the current segment when it has room, a +// new one carved from one call to the pair when it does not. NULL means NOT +// ADDED: an arena that cannot carve another segment, or a count at the int32 +// cap (§2.2, §2.9). +template +inline typename TableList::Element * TableListAppend( TableWorker & worker, TableListHead * head, TableList & list ) +{ + typedef typename TableList::Element Element; + if ( list.count >= INT32_MAX ) { return NULL; } // the int32 storage cap + TableListSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableListSegmentElements ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableListSegment *) worker.AllocRaw( (int64_t) sizeof( TableListSegment ), (int64_t) alignof( TableListSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableListSegment * previous = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Element * element = segment->elements + segment->used; + segment->used++; + head->live++; + list.count++; + return element; +} + +// ADD, whole: the head, the append, and the element at its declared defaults +// (§2.9). The text form's placement is this same call, because a list has no +// key to place under (§16). +template +inline typename TableList::Element * TableListPlace( TableWorker & worker, TableList & list ) +{ + typedef typename TableList::Element Element; + TableListHead * head = TableListReach( worker, list ); + if ( head == NULL ) { return NULL; } + Element * element = TableListAppend( worker, head, list ); + if ( element == NULL ) { return NULL; } + new ( element ) Element(); // value-init: the declared defaults, and null for a slot + return element; +} + +// ERASE, ADDRESSED BY THE POINTER (§2.9): the element Add handed back is the +// handle, because a list has no key and the address is the one thing the +// builder promises never moves (§6.4). It marks the element DEAD, one bit in +// the segment's slot and not in the element storage, and decrements the live +// count. False when the pointer is not this list's. Its storage is reclaimed +// at RESET and never reused mid-build, the map's rule for the map's reason. +template +inline bool TableListErase( TableArena & arena, TableList & list, const typename TableList::Element * element ) +{ + typedef typename TableList::Element Element; + if ( list.elements.value == 0 || element == NULL ) { return false; } + TableListHead * head = (TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableListSegment * segment = (TableListSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + if ( element >= segment->elements && element < segment->elements + segment->used ) + { + const int32_t i = (int32_t) ( element - segment->elements ); + if ( TableListSegmentDead( segment->dead, i ) ) { return false; } // already erased + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + list.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INDEX order, live elements only (§2.9) ---- +template struct TableListEach +{ + typedef typename TableList::Element Element; + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableListSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableListSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + Element * operator*() const { return segment->elements + index; } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableListEach TableListEachOf( const TableArena & arena, const TableList & list ) +{ + TableListEach each = { &arena, TableRef() }; + if ( list.elements.value != 0 ) + { + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + each.first = head->first; + } + return each; +} + +// ---- the INDEX-ORDER CURSOR the four writing walks read (§2.9) ---- +// +// Measure, Save, Lock and Cook each visit a list's live elements in the order +// they were added, and they allocate nothing to do it: a region's cursor is +// the array in place, and the builder's walks the segment chain. Indexing the +// builder's form is SEQUENTIAL by construction, every walk steps i, i + 1, +// i + 2, so the cursor remembers where the last access landed and moves one +// live slot per step. An access behind the memo restarts from the first +// segment, which no walk here does. +template struct TableListCursor +{ + const Element * elements = NULL; // the region's form: the array in place + const TableArena * arena = NULL; // the builder's form: the segments + TableRef first; + int32_t count = 0; + bool ok = false; + // the memo: the segment and slot the last access landed on, and the live + // index that slot holds + mutable const TableListSegment * segment = NULL; + mutable int32_t within = -1; + mutable int32_t logical = -1; + + const Element * At( int32_t index ) const + { + if ( elements != NULL ) { return elements + index; } + if ( segment == NULL || index < logical ) + { + segment = first.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL; + within = -1; + logical = -1; + } + while ( logical < index ) + { + for ( ;; ) + { + within++; + while ( segment != NULL && within >= segment->used ) + { + segment = segment->next.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + within = 0; + } + if ( segment == NULL ) { return NULL; } // the slot and the head disagree + if ( !TableListSegmentDead( segment->dead, within ) ) { break; } + } + logical++; + } + return segment->elements + within; + } + const Element & operator[]( int32_t index ) const { return *At( index ); } +}; + +// the REGION form: the array is the cursor +template +inline TableListCursor::Element> TableListElements( const TableRegionCtx &, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.elements = list.Elements(); + cursor.count = list.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: the live elements out of the segment chain, in the +// order they were added. A dead element costs nothing on any wire (§2.9). +template +inline TableListCursor::Element> TableListElements( const TableArena & arena, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.arena = &arena; + cursor.count = list.count; + if ( list.elements.value == 0 || list.count <= 0 ) { cursor.ok = list.count == 0; cursor.count = 0; return cursor; } + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + if ( head->live != list.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + cursor.first = head->first; + cursor.ok = true; + return cursor; +} + +template +inline TableListCursor::Element> TableListElements( const TableArenaCtx & ctx, const TableList & list ) +{ + return TableListElements( *ctx.arena, list ); +} + +// ---- the LOAD side: where a decoded element lands (§2.9) ---- +// +// The same two shapes the map's fill takes, because the decoder above them +// cannot tell which it has: a REGION carves the element array out of the +// holder node's own extent, PRE-ORDER, and the TOOL's path appends into the +// builder's arena. +template struct TableListFill +{ + typedef typename TableList::Element Element; + TableList * list = NULL; + Element * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; + bool refused = false; // a count above the int32 cap on the tool's path: LoadBuilder answers NULL +}; + +template +inline TableListFill TableListFillBegin( const TableNodeMap & nodes, TableList & list, uint64_t n ) +{ + typedef typename TableList::Element Element; + TableListFill fill; + fill.list = &list; + list.elements.value = 0; + list.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( n > (uint64_t) INT32_MAX ) + { + // A COUNT ABOVE THE int32 STORAGE CAP (§2.2, §2.9): into a region it was + // refused by LoadMeasure before this ran, and into a builder it is the + // refusal LoadBuilder answers NULL for, moving no counter + fill.refused = nodes.carve->worker != NULL; + return fill; + } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Element ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Element ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Element *) base; + fill.capacity = (int32_t) n; + list.elements.value = (int64_t) ( base - (const uint8_t *) &list.elements ); + fill.ok = true; + return fill; +} + +// the next slot, at the element's declared defaults. NULL when the arena +// could not carve, which the decoder reports as framing damage +template inline typename TableList::Element * TableListFillNext( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count >= fill.capacity ) { return NULL; } + Element * element = fill.array + fill.list->count; + new ( element ) Element(); + fill.list->count++; + return element; + } + return TableListPlace( *fill.worker, *fill.list ); +} + +// A SLOT WHOSE ELEMENT NEVER LANDED is given back (§2.9, §4): the array keeps +// what it decoded, and an element whose own framing gave out before one byte +// of it decoded was not decoded. The region's form uncounts it, and the builder's +// marks it dead, which is what the storage rule allows mid-build. +template inline void TableListFillDrop( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count > 0 ) { fill.list->count--; } + return; + } + if ( fill.list->elements.value == 0 ) { return; } + TableListHead * head = (TableListHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.list->elements.value ); + if ( head->last.value == 0 ) { return; } + TableListSegment * segment = (TableListSegment *) TableArenaAt( *fill.worker->arena, (uint32_t) head->last.value ); + if ( segment->used <= 0 ) { return; } + const int32_t i = segment->used - 1; + if ( TableListSegmentDead( segment->dead, i ) ) { return; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + fill.list->count--; +} + +// an EMPTY list's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableListFillEnd( TableListFill & fill ) +{ + if ( fill.array != NULL && fill.list->count == 0 ) { fill.list->elements.value = 0; } +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.9, §6.5) ---- +// +// N x sizeof( T ) rounded to alignof( T ), AT EVERY DEPTH. N is framing and +// not a value, so this reads no field: it walks the list's own header and, +// where a table element holds a list or a map of its own, the elements' +// headers under it. Every -1 carries its REASON (§6.5): the int32 cap first, +// because a count past it cannot fit any body, and then the body's own L. +inline bool TableListWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t elem_size, int64_t elem_align, uint8_t elem_kind, int64_t elem_floor, + TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != elem_kind ) { return true; } // another element kind: §4's ordinary kind mismatch, the field reads empty + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; } // an N the list's L cannot carry + at = ( at + elem_align - 1 ) & ~( elem_align - 1 ); + at += (int64_t) n * elem_size; + if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_LIST + +#ifndef LISTDEMO_SCHEMA_BUILD_VERSION +#define LISTDEMO_SCHEMA_BUILD_VERSION + +namespace listdemo { + +// THE BUILD VERSION (docs/SPEC-TABLES.md §20): one digest over every fact the bytes +// this build produces depend on — the type wire's protocol id, every record's +// layout as the compiler's own C ABI model computes it, and the facts that +// decide what a load PUTS in those slots. It is the number a cook's header +// carries and the number Open compares, and the number a block's prologue +// carries and BlockOpen compares: a build version answers "which build?" and +// not "which form?", and what separates the two forms is their MAGIC. +// +// There are TWO ids in the design and they are not interchangeable: the +// PROTOCOL ID is the type wire's and nothing else, and the BUILD VERSION is +// what everything cooked or blocked is keyed by. A table edit moves this and +// never the protocol id; a type edit moves both. +static const uint64_t BuildVersion = 0x8d7c0edaca4571c7ull; + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_BUILD_VERSION + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK +#define LISTDEMO_SCHEMA_TABLE_COOK + +namespace listdemo { + +// ---- the cooked form (docs/SPEC-TABLES.md §7) ---- +// +// A cooked file is a HEADER, a DATA part and an ATTRIBUTION part, in that +// order. Every word of the header is a u64 written in the byte order the cook +// was produced in, and the header is 64 bytes: +// +// 0 magic 0x4b4f4f434d484353, read BYTEWISE before anything else +// 8 build_version the unit's id (docs/SPEC-TABLES.md §20) +// 16 byte_order 1 little, 2 big — the order that WROTE the file +// 24 data_length the region's bytes, rounded up to alignment +// 32 attribution_length the directory's bytes, or 0 +// 40 alignment the region's alignment, never below eight +// 48 reserved zero +// 56 reserved zero +// +// The DATA part is Lock's region written verbatim (§7.2) — the root at its +// base — and it is what a runtime points at. The ATTRIBUTION part is the node +// directory (§6.3), and NOTHING THAT READS THE STRUCTURE TOUCHES IT: it is +// written beside the data for schema cook-check, so a build that ships no +// tooling need not carry it at all. +static const int64_t kTableCookHeaderBytes = 64; + +// THE MAGIC'S VALUE, and a consumer written from the page needs the constant +// rather than a description of one. It is "SCHMCOOK" read as ASCII in the byte +// order a little-endian store produces — the same shape the block form's +// SCHMABLK takes, so a hex dump of a little-endian cook is legible and the two +// accelerators sit in one vocabulary. +// +// IT IS STORED IN THE PRODUCER'S ORDER, which is what makes it the byte-order +// check as well as the form check: a consumer reads back this build's +// constant, or that constant byte-reversed — which identifies a cook of the +// OTHER order — or something that is not a cook. All three answers but the +// first refuse, and a cook and a BLOCK are separated here too, because a +// form's identity belongs in its magic rather than in a second digest. +static const uint64_t TableCookMagic = 0x4b4f4f434d484353ull; + +// THIS BUILD's byte order, as the header's own word carries it. The magic is +// what REFUSES a foreign order; this word is what RECORDS which order wrote +// the file, so a refusal names the order rather than inferring it and a tool +// dumping a cook reads the fact. A file whose magic matched and whose order +// word did not is corrupt, and there is no reading that recovers it. +// +// The BUILD VERSION cannot do either job: §20.1 digests byteorder as a +// GENERATION input, little for every target schema generates for today, so +// two builds of one schema for two orders emit the same id. +#if defined( __BYTE_ORDER__ ) && defined( __ORDER_BIG_ENDIAN__ ) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__ +static const uint64_t TableCookByteOrder = 2; // big +#else +static const uint64_t TableCookByteOrder = 1; // little +#endif + +// The greatest region alignment a cooked file may name. The DATA part begins +// at align_up( 64, alignment ), which is 64 for every unit this language can +// declare — the largest alignment it has is sixteen — so a word past this cap +// describes a file no build of this schema wrote (docs/SPEC-TABLES.md §7.1). +static const uint64_t TableCookMaxAlign = 64; + +// The header read, BYTEWISE. memcpy is the portable spelling of "these eight +// bytes, in this machine's order"; every compiler this repo builds under folds +// it to one load, and it is the only read in the whole of Open that is not a +// comparison. +inline uint64_t table_cook_read64( const uint8_t * p ) +{ + uint64_t v; + memcpy( &v, p, sizeof( v ) ); + return v; +} + +// TableCookOpen: THE WHOLE CHECK, in one place, because §7 states the +// enumeration once and every generated Open is that one enumeration plus +// its own root's two layout facts. +// +// THE CHECK, in order: the magic read bytewise, the byte order it establishes, +// the build version against this build's own, both RESERVED words zero, the +// region alignment the header names, the two part lengths against the length +// the caller passed — a truncated file and a file with trailing bytes are the +// same refusal — the root's own storage inside the data part, and the +// alignment of the base. +// +// AND THAT IS ALL OF IT. On a match the bytes ARE what this build wrote, in +// this build's layout and this build's byte order, so there is nothing to +// validate and nothing to fix up: the caller gets the root. Nothing per node +// happens here, which is what makes open O(1) in the file's size; a walk of +// any shape would forfeit that, and validating an untrusted file is schema +// cook-check's job and a person's decision (§7.4). +// +// EVERY NUMBER BELOW COMES OUT OF THE FILE, so the arithmetic is unsigned and +// each term is BOUNDED BEFORE IT IS ADDED: a forged length near 2^64 must +// refuse, and an addition that wrapped would be the defect the comparison +// after it was supposed to catch. Nothing past length is read on any path, +// including every refusing one. +inline const uint8_t * TableCookOpen( const void * bytes, uint64_t length, uint64_t root_size, uint64_t root_align ) +{ + if ( bytes == NULL ) { return NULL; } + if ( length < (uint64_t) kTableCookHeaderBytes ) { return NULL; } + const uint8_t * raw = (const uint8_t *) bytes; + // the MAGIC, bytewise and first: it is what establishes the byte order + // every other header word is read in, so nothing else may be read before + // it. A byte-reversed constant is a cook of the other order and refuses + // here, which is why the order never reaches a fix-up pass. + if ( table_cook_read64( raw ) != TableCookMagic ) { return NULL; } + if ( table_cook_read64( raw + 16 ) != TableCookByteOrder ) { return NULL; } + if ( table_cook_read64( raw + 8 ) != BuildVersion ) { return NULL; } + // the RESERVED words: a non-zero one means a writer used a form this build + // does not understand, and Open refuses rather than ignoring it. + if ( table_cook_read64( raw + 48 ) != 0 ) { return NULL; } + if ( table_cook_read64( raw + 56 ) != 0 ) { return NULL; } + const uint64_t data_length = table_cook_read64( raw + 24 ); + const uint64_t attribution_length = table_cook_read64( raw + 32 ); + const uint64_t alignment = table_cook_read64( raw + 40 ); + // THE ALIGNMENT WORD IS DATA, and it is the one header field the rest of + // the check does arithmetic WITH rather than only comparison against. A + // region's alignment is a power of two, never below eight (the floor that + // puts the attribution part on an eight-byte boundary without a second + // padding rule) and never past the cap above; a word that is none of those + // rounds nothing and aligns nothing, so it is refused before it is used. + if ( alignment < 8 || alignment > TableCookMaxAlign ) { return NULL; } + if ( ( alignment & ( alignment - 1 ) ) != 0 ) { return NULL; } + // and it must be an alignment THE ROOT CAN SIT AT, since the root is at + // the region's base: both are powers of two, so "at least the root's" + // is one division. + if ( ( alignment % root_align ) != 0 ) { return NULL; } + // The DATA part begins at align_up( 64, alignment ). It is DERIVED and not + // a header field, because a fact a reader computes is a fact two writers + // cannot disagree about. + const uint64_t data_offset = ( (uint64_t) kTableCookHeaderBytes + alignment - 1 ) & ~( alignment - 1 ); + if ( length < data_offset ) { return NULL; } + // the two part lengths against the length the caller passed. The whole + // file is data_offset + data_length + attribution_length, and a length + // that is not EXACTLY that refuses — truncation and trailing bytes are one + // refusal, and both terms are subtracted rather than added so no sum can + // carry. + if ( data_length > length - data_offset ) { return NULL; } + if ( attribution_length != length - data_offset - data_length ) { return NULL; } + // the ROOT sits at the region's base, so the region has to hold it: a + // shorter data part describes a root partly outside the file, which is the + // one way a match-and-point reader could hand back storage it never + // received. + if ( data_length < root_size ) { return NULL; } + const uint8_t * base = raw + data_offset; + // the alignment of the BASE. The header pads the data part to the region's + // alignment, so a base an allocator or mmap gave you is already aligned — + // mmap gives page alignment for free — and a base that is not is a caller's + // buffer this form cannot be read out of. + if ( ( (uintptr_t) base % (uintptr_t) alignment ) != 0 ) { return NULL; } + return base; +} + +// ---- the cooked form, the WRITE side (docs/SPEC-TABLES.md §7.6) ---- +// +// THE BYTE ORDER IS THE TARGET'S, NOT THE HOST'S. A cook is produced in the +// byte order of the build that will read it (§7), so the fixing happens here — +// offline, once, on the writing side — and never at Open. Passing +// TableByteOrder::Big on a little-endian machine produces a big-endian build's +// file, and nothing about the writing host reaches the bytes. +enum class TableByteOrder +{ + Little = 1, // the header's byte_order word, and the order every scalar is written in + Big = 2, +}; + +// One store, width as an argument. Every call site passes a literal width, so +// the loop folds to a store (and a byte swap on the foreign order); a name per +// width would claim four §11 names to save nothing. +inline void table_cook_put( uint8_t * at, uint64_t value, int32_t width, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * i ) ); } + } + else + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * ( width - 1 - i ) ) ); } + } +} + +// A 128-bit store as two lanes: sixteen bytes, the low lane first in the +// little order and the high lane first — each lane big-endian — in the big +// order, exactly as a u64 is one lane of eight (docs/SPEC-TABLES.md §7.2). +inline void table_cook_put128( uint8_t * at, uint64_t lo, uint64_t hi, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) { table_cook_put( at, lo, 8, order ); table_cook_put( at + 8, hi, 8, order ); } + else { table_cook_put( at, hi, 8, order ); table_cook_put( at + 8, lo, 8, order ); } +} + +// A buffer piece: the USED bytes and nothing else. The tail is already zero — +// the whole extent was zeroed before any field was written — so this copies the +// used prefix and leaves the rest, which is what makes a string's unused tail a +// consequence of one memset rather than a rule per buffer. A used length past +// the buffer, or below zero, is a value no reader could have produced and it is +// clamped rather than trusted: this writes inside the caller's buffer on every +// input. +inline void table_cook_bytes( uint8_t * at, const void * source, int64_t used, int64_t capacity ) +{ + if ( used <= 0 ) { return; } + const int64_t n = used < capacity ? used : capacity; + memcpy( at, source, (size_t) n ); +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE +#define LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// ---- the cooked form's WRITE side for a POINTERED root (docs/SPEC-TABLES.md §7.6) ---- +// +// A pointered root's cook is the region of §7.2: every node the numbering +// reached (§3.1), once, at its own type's alignment, in index order, the root +// at offset zero. This is that region while it is being laid out and written — +// the tool's own Layout and Write, in one struct. +// +// The OFFSETS are one per node, the root's zero at position 0 and node index k +// at position k - 1, which is the directory's own order (§6.3); they are the +// one allocation the write makes beyond the numbering, and they go through the +// same pair. A measure needs no offsets and leaves the pointer NULL. +struct TableCookRegion +{ + const TableNumbering * numbering = NULL; // node -> index, from the walk that placed it + int64_t * offsets = NULL; // index - 1 -> the node's region offset; NULL while measuring + int64_t count = 0; // nodes, the root included + int64_t bytes = 0; // the data part's length, rounded to align + int64_t align = 0; // the region's alignment: the nodes' greatest, never below eight + uint8_t * base = NULL; // where the data part is being written; NULL while measuring +}; + +// A reference slot: the SELF-RELATIVE delta from the slot's own address to the +// node's start (§6.3), and zero for null. The node is found by the address the +// numbering keyed it under, which is the same address the walk resolved through +// the same context — so a reference the numbering does not carry is a slot the +// walk never reached (a counted array's slot past its count, an absent +// optional's value) holding a node the region will not hold, and it is refused +// rather than written as a delta to nowhere. +inline bool table_cook_ref( const TableCookRegion & region, uint8_t * at, const void * pointee, TableByteOrder order ) +{ + if ( pointee == NULL ) { table_cook_put( at, 0, 8, order ); return true; } + uint64_t index = 0; + if ( !TableNumberingIndex( *region.numbering, pointee, index ) ) { return false; } + if ( index == 0 || index > (uint64_t) region.count ) { return false; } + const int64_t delta = region.offsets[index - 1] - (int64_t) ( at - region.base ); + table_cook_put( at, (uint64_t) delta, 8, order ); + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// table Placement — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Placement { + float x = 0.0f; + float y = 0.0f; + uint32_t model = 0; +}; + +// table LogEntry — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct LogEntry { + uint32_t tick = 0; +}; + +// table Save — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Save { + TableList placements; // Placement: the element array, empty until an Add + TableList log; // LogEntry: the element array, empty until an Add + TableList scores; // int32: the element array, empty until an Add +}; + +// table Point — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Point { + int32_t x = 0; + int32_t y = 0; +}; + +// HitType: union Hit's tag — None = 0, then each variant in declared order (SPEC §4.8) +enum class HitType : uint8_t { + None = 0, + Point = 1, + Damage = 2, + Max = 2, // the exported extent (SPEC §4.2) +}; + +// union Hit — at most one of the arms; the tag says which. AN ARM IS A FIELD +// LINE (docs/SPEC-TABLES.md §2.6), so an arm's storage is the field's storage +// overlaid — and an arm whose storage needs a companion, a string's length or +// a counted array's count, is one member of an unnamed struct, `value` beside +// `value_length` or `value_count`. Such a union has no packet wire and lives +// here, after its arms. Construction is None: the tag alone is initialized; an +// arm's storage is established when the arm is selected — by HitLoadBody +// before it decodes, or by assigning it: value.point = Point{}. +// Bytes of unselected arms are indeterminate. +struct Hit +{ + HitType type; + + union + { + Point point; + int32_t damage; + }; + + Hit() : type( HitType::None ) {} // the tag only — arms are established at selection +}; + +// table Mixed — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Mixed { + TableList grades; // Grade: the element array, empty until an Add + TableList perms; // Perm: the element array, empty until an Add + TableList hits; // Hit: the element array, empty until an Add + TableList bounds; // int32: the element array, empty until an Add +}; + +// Grade on the TABLE wire: a value rides under its OWN kind 30, carrying the +// REFERENCE to its variant name's id, whatever the declaration-side storage +// width — so a variant may be added anywhere, removed, or reordered and old +// data still reads (docs/SPEC-TABLES.md §3, §5). None is the ZERO REFERENCE, +// the one value that names no id, so no declared variant can be mistaken for it. +#ifndef LISTDEMO_SCHEMA_TABLE_ENUM_GRADE +#define LISTDEMO_SCHEMA_TABLE_ENUM_GRADE +inline bool TableEnumRef( TableIds & ids, Grade value, uint64_t & ref ) +{ + switch ( value ) + { + case Grade::None: ref = 0; return true; + case Grade::A: ref = ids.ref( 0xaf63fc4c860222ecull, 33 ); return true; + case Grade::B: ref = ids.ref( 0xaf63ff4c86022805ull, 34 ); return true; + case Grade::C: ref = ids.ref( 0xaf63fe4c86022652ull, 35 ); return true; + default: return false; // no variant names this value: no wire identity + } +} +inline bool TableEnumNamed( Grade value ) +{ + switch ( value ) + { + case Grade::None: return true; + case Grade::A: return true; + case Grade::B: return true; + case Grade::C: return true; + default: return false; + } +} +inline bool TableEnumId( Grade value, uint64_t & id ) +{ + switch ( value ) + { + case Grade::None: id = 0; return true; + case Grade::A: id = 0xaf63fc4c860222ecull; return true; + case Grade::B: id = 0xaf63ff4c86022805ull; return true; + case Grade::C: id = 0xaf63fe4c86022652ull; return true; + default: return false; // no variant names this value: no wire identity + } +} +inline bool TableEnumValue( uint64_t id, Grade & out ) +{ + switch ( id ) + { + case 0xaf63fc4c860222ecull: out = Grade::A; return true; + case 0xaf63ff4c86022805ull: out = Grade::B; return true; + case 0xaf63fe4c86022652ull: out = Grade::C; return true; + default: return false; // an id this build cannot name + } +} +#endif // LISTDEMO_SCHEMA_TABLE_ENUM_GRADE + +// ---- prefill: the declared defaults, in place (docs/SPEC-TABLES.md) ---- + +inline void PlacementReset( Placement & value ); +inline void LogEntryReset( LogEntry & value ); +inline void SaveReset( Save & value ); +inline void PointReset( Point & value ); +inline void MixedReset( Mixed & value ); + +inline void PlacementReset( Placement & value ) +{ + value.x = 0.0f; + value.y = 0.0f; + value.model = 0; +} + +inline void LogEntryReset( LogEntry & value ) +{ + value.tick = 0; +} + +inline void SaveReset( Save & value ) +{ + value.placements.elements.value = 0; // Placement: empty + value.placements.count = 0; + value.placements.padding = 0; + value.log.elements.value = 0; // LogEntry: empty + value.log.count = 0; + value.log.padding = 0; + value.scores.elements.value = 0; // int32: empty + value.scores.count = 0; + value.scores.padding = 0; +} + +inline void PointReset( Point & value ) +{ + value.x = 0; + value.y = 0; +} + +inline void MixedReset( Mixed & value ) +{ + value.grades.elements.value = 0; // Grade: empty + value.grades.count = 0; + value.grades.padding = 0; + value.perms.elements.value = 0; // Perm: empty + value.perms.count = 0; + value.perms.padding = 0; + value.hits.elements.value = 0; // Hit: empty + value.hits.count = 0; + value.hits.padding = 0; + value.bounds.elements.value = 0; // int32: empty + value.bounds.count = 0; + value.bounds.padding = 0; +} + +// ---- the arena's reset hook (docs/SPEC-TABLES.md §6) ---- +// +// TableWorker::Alloc is a template and cannot name a member's Reset, so +// the arena reaches it through this overload set by argument-dependent +// lookup. It is how a node born in raw arena storage comes to hold the +// declared defaults without value-initialising the whole aggregate. + +inline void TableReset( Placement & value ) { PlacementReset( value ); } +inline void TableReset( LogEntry & value ) { LogEntryReset( value ); } +inline void TableReset( Save & value ) { SaveReset( value ); } +inline void TableReset( Point & value ) { PointReset( value ); } +inline void TableReset( Mixed & value ) { MixedReset( value ); } + +// ---- pointer targets: allocation and resolution (docs/SPEC-TABLES.md §2) ---- +// +// A reference resolves differently in the two forms, and the CONTEXT says +// which: in the arena it is an offset; in a region it is a self-relative +// delta, so the const deref below is one add and needs no base pointer. + +// LogEntry is a pointer target. +inline const LogEntry * LogEntryAt( const TableRef & ref ) // the const form's hot path: one add, no base +{ + return ref.value != 0 ? (const LogEntry *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline LogEntry * LogEntryAt( TableRef & ref ) +{ + return ref.value != 0 ? (LogEntry *) ( (uint8_t *) &ref + ref.value ) : NULL; +} +inline const LogEntry * LogEntryAt( const TableRegionCtx &, const TableRef & ref ) { return LogEntryAt( ref ); } +inline const LogEntry * LogEntryAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const LogEntry *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +// while the builder is mutable, resolve against the arena itself +inline LogEntry * LogEntryAt( TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (LogEntry *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} +inline const LogEntry * LogEntryAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const LogEntry *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} +// allocate one LogEntry in the arena; the slot holds the arena offset +inline LogEntry * LogEntryEmplace( TableWorker & worker, TableRef & slot ) +{ + TableSlot allocated = worker.Alloc(); + slot = allocated.ref; + return allocated.ptr; +} + +// ---- codecs: measure/save/load per closure member ---- + +inline int64_t PlacementMeasureBody( TableIds & ids, const Placement & value ); +LISTDEMO_TABLE_INLINE bool PlacementSaveBody( TableWriter & w, TableIds & ids, const Placement & value ); +LISTDEMO_TABLE_INLINE bool PlacementLoadBody( TableReader & r, Placement & value ); +inline int64_t LogEntryMeasureBody( TableIds & ids, const LogEntry & value ); +LISTDEMO_TABLE_INLINE bool LogEntrySaveBody( TableWriter & w, TableIds & ids, const LogEntry & value ); +LISTDEMO_TABLE_INLINE bool LogEntryLoadBody( TableReader & r, LogEntry & value ); +template inline int64_t SaveMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Save & value ); +template inline bool SaveSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Save & value ); +template inline bool SaveSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Save & value ); +inline bool SaveLoadBody( TableReader & r, const TableNodeMap & nodes, Save & value ); +inline int64_t PointMeasureBody( TableIds & ids, const Point & value ); +LISTDEMO_TABLE_INLINE bool PointSaveBody( TableWriter & w, TableIds & ids, const Point & value ); +LISTDEMO_TABLE_INLINE bool PointLoadBody( TableReader & r, Point & value ); +template inline int64_t MixedMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Mixed & value ); +template inline bool MixedSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Mixed & value ); +template inline bool MixedSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Mixed & value ); +inline bool MixedLoadBody( TableReader & r, const TableNodeMap & nodes, Mixed & value ); + +// ---- pointer-graph walkers: number (measure/save), pack (Lock) ---- + +template inline bool LogEntryNumber( const Ctx & ctx, TableNumbering & numbering, const LogEntry & value ); +template inline int64_t LogEntryPackMeasure( const Ctx & ctx, TablePackMap & seen, const LogEntry & value ); +template inline bool LogEntryPack( const Ctx & ctx, TablePackMap & seen, const LogEntry & src, LogEntry & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool SaveNumber( const Ctx & ctx, TableNumbering & numbering, const Save & value ); +template inline int64_t SavePackMeasure( const Ctx & ctx, TablePackMap & seen, const Save & value ); +template inline bool SavePack( const Ctx & ctx, TablePackMap & seen, const Save & src, Save & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool MixedNumber( const Ctx & ctx, TableNumbering & numbering, const Mixed & value ); +template inline int64_t MixedPackMeasure( const Ctx & ctx, TablePackMap & seen, const Mixed & value ); +template inline bool MixedPack( const Ctx & ctx, TablePackMap & seen, const Mixed & src, Mixed & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +// ---- the numbering's bridge to each member's codec (docs/SPEC-TABLES.md §3.1) ---- + +template inline int64_t TableNodeMeasure( const Ctx &, const TableNumbering &, TableIds & ids, const LogEntry & value ) { return LogEntryMeasureBody( ids, value ); } +template inline bool TableNodeSave( const Ctx &, const TableNumbering &, TableWriter & w, TableIds & ids, const LogEntry & value ) { return LogEntrySaveBody( w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Save & value ) { return SaveMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Save & value ) { return SaveSaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Mixed & value ) { return MixedMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Mixed & value ) { return MixedSaveBody( ctx, numbering, w, ids, value ); } + +inline int64_t PlacementMeasureBody( TableIds & ids, const Placement & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.x != 0.0f ) { bytes += TableLebBytes( ids.ref( 0xaf63f54c86021707ull, 19 ) ) + 1 + 4; } // x + if ( value.y != 0.0f ) { bytes += TableLebBytes( ids.ref( 0xaf63f44c86021554ull, 20 ) ) + 1 + 4; } // y + if ( value.model != 0 ) { bytes += TableLebBytes( ids.ref( 0x9de543933e6e703aull, 21 ) ) + 1 + 4; } // model + return bytes; +} + +inline int64_t PlacementMeasure( const Placement & value ) +{ + TableIds ids; + const int64_t body = PlacementMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool PlacementSaveBody( TableWriter & w, TableIds & ids, const Placement & value ) +{ + if ( value.x != 0.0f ) + { + w.putleb( ids.ref( 0xaf63f54c86021707ull, 19 ) ); w.put8( 10 ); // x + w.put32( table_float_to_bits( value.x ) ); + } + if ( value.y != 0.0f ) + { + w.putleb( ids.ref( 0xaf63f44c86021554ull, 20 ) ); w.put8( 10 ); // y + w.put32( table_float_to_bits( value.y ) ); + } + if ( value.model != 0 ) + { + w.putleb( ids.ref( 0x9de543933e6e703aull, 21 ) ); w.put8( 8 ); // model + w.put32( uint32_t( value.model ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t PlacementSave( const Placement & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !PlacementSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == PlacementMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool PlacementLoadBody( TableReader & r, Placement & value ) +{ + PlacementReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xaf63f54c86021707ull: // x + { + if ( kind != 10 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + value.x = table_bits_to_float( r.get32() ); + break; + } + case 0xaf63f44c86021554ull: // y + { + if ( kind != 10 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + value.y = table_bits_to_float( r.get32() ); + break; + } + case 0x9de543933e6e703aull: // model + { + if ( kind != 8 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + uint32_t decoded_v = uint32_t( r.get32( ) ); + value.model = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict PlacementLoadVerdict( Placement & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + PlacementReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + PlacementReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !PlacementLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool PlacementLoad( Placement & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return PlacementLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t PlacementMeasureMessage( const Placement & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = PlacementMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t PlacementSaveMessage( const Placement & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !PlacementSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == PlacementMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool PlacementLoadMessage( Placement & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + PlacementReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return PlacementLoadBody( r, value ); +} + +inline int64_t LogEntryMeasureBody( TableIds & ids, const LogEntry & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.tick != 0 ) { bytes += TableLebBytes( ids.ref( 0x1e7683ef2ebc7684ull, 12 ) ) + 1 + 4; } // tick + return bytes; +} + +inline int64_t LogEntryMeasure( const LogEntry & value ) +{ + TableIds ids; + const int64_t body = LogEntryMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool LogEntrySaveBody( TableWriter & w, TableIds & ids, const LogEntry & value ) +{ + if ( value.tick != 0 ) + { + w.putleb( ids.ref( 0x1e7683ef2ebc7684ull, 12 ) ); w.put8( 8 ); // tick + w.put32( uint32_t( value.tick ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t LogEntrySave( const LogEntry & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !LogEntrySaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == LogEntryMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool LogEntryLoadBody( TableReader & r, LogEntry & value ) +{ + LogEntryReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x1e7683ef2ebc7684ull: // tick + { + if ( kind != 8 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + uint32_t decoded_v = uint32_t( r.get32( ) ); + value.tick = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict LogEntryLoadVerdict( LogEntry & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + LogEntryReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + LogEntryReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !LogEntryLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool LogEntryLoad( LogEntry & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return LogEntryLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t LogEntryMeasureMessage( const LogEntry & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = LogEntryMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t LogEntrySaveMessage( const LogEntry & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !LogEntrySaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == LogEntryMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool LogEntryLoadMessage( LogEntry & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + LogEntryReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return LogEntryLoadBody( r, value ); +} + +template +inline int64_t SaveMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Save & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // placements: a kind 14 array of kind 13 elements, INDEX order (§2.9) + TableListCursor cursor_placements = TableListElements( ctx, value.placements ); + if ( !cursor_placements.ok ) { return -1; } // the slot and the head disagree + if ( cursor_placements.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_placements = ids.ref( 0xd24733aa574d4b09ull, 24 ); + int64_t body_placements = 0; + body_placements += 1 + TableLebBytes( (uint64_t) ( cursor_placements.count ) ); // the element kind byte and the count + for ( int32_t elem_i_placements = 0; elem_i_placements < cursor_placements.count; elem_i_placements++ ) + { + const int64_t elem_bytes_placements = PlacementMeasureBody( ids, cursor_placements[elem_i_placements] ); + if ( elem_bytes_placements < 0 ) { return -1; } + body_placements += TableLebBytes( (uint64_t) ( elem_bytes_placements ) ) + ( elem_bytes_placements ); + } + bytes += TableLebBytes( ref_placements ) + 1 + TableLebBytes( (uint64_t) ( body_placements ) ) + ( body_placements ); + } + } + { + // log: a kind 14 array of kind 17 elements, INDEX order (§2.9) + TableListCursor cursor_log = TableListElements( ctx, value.log ); + if ( !cursor_log.ok ) { return -1; } // the slot and the head disagree + if ( cursor_log.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_log = ids.ref( 0x125073191daf5431ull, 25 ); + int64_t body_log = 0; + body_log += 1 + TableLebBytes( (uint64_t) ( cursor_log.count ) ); // the element kind byte and the count + for ( int32_t elem_i_log = 0; elem_i_log < cursor_log.count; elem_i_log++ ) + { + { + const LogEntry * slot_pointee_log = LogEntryAt( ctx, cursor_log[elem_i_log] ); + uint64_t slot_index_log = 0; + if ( slot_pointee_log != NULL && !TableNumberingIndex( numbering, (const void *) slot_pointee_log, slot_index_log ) ) { return -1; } + body_log += TableLebBytes( slot_index_log ); + } + } + bytes += TableLebBytes( ref_log ) + 1 + TableLebBytes( (uint64_t) ( body_log ) ) + ( body_log ); + } + } + { + // scores: a kind 14 array of kind 4 elements, INDEX order (§2.9) + TableListCursor cursor_scores = TableListElements( ctx, value.scores ); + if ( !cursor_scores.ok ) { return -1; } // the slot and the head disagree + if ( cursor_scores.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_scores = ids.ref( 0x01986b0b27400fb2ull, 26 ); + int64_t body_scores = 0; + body_scores += 1 + TableLebBytes( (uint64_t) ( cursor_scores.count ) ); // the element kind byte and the count + body_scores += (int64_t) ( cursor_scores.count ) * 4; + bytes += TableLebBytes( ref_scores ) + 1 + TableLebBytes( (uint64_t) ( body_scores ) ) + ( body_scores ); + } + } + return bytes; +} + +template +inline bool SaveSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Save & value ) +{ + { + TableListCursor cursor_placements = TableListElements( ctx, value.placements ); // placements + if ( !cursor_placements.ok ) { return false; } + if ( cursor_placements.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_placements = ids.ref( 0xd24733aa574d4b09ull, 24 ); + int64_t body_placements = 0; + body_placements += 1 + TableLebBytes( (uint64_t) ( cursor_placements.count ) ); // the element kind byte and the count + for ( int32_t elem_i_placements = 0; elem_i_placements < cursor_placements.count; elem_i_placements++ ) + { + const int64_t elem_bytes_placements = PlacementMeasureBody( ids, cursor_placements[elem_i_placements] ); + if ( elem_bytes_placements < 0 ) { return false; } + body_placements += TableLebBytes( (uint64_t) ( elem_bytes_placements ) ) + ( elem_bytes_placements ); + } + w.putleb( ref_placements ); w.put8( 14 ); w.putleb( (uint64_t) body_placements ); // placements + w.put8( 13 ); w.putleb( (uint64_t) ( cursor_placements.count ) ); + for ( int32_t elem_i_placements = 0; elem_i_placements < cursor_placements.count; elem_i_placements++ ) + { + { + const int64_t elem_len_placements = PlacementMeasureBody( ids, cursor_placements[elem_i_placements] ); + if ( elem_len_placements < 0 ) return false; + w.putleb( (uint64_t) elem_len_placements ); + if ( !PlacementSaveBody( w, ids, cursor_placements[elem_i_placements] ) ) return false; + } + } + } + } + { + TableListCursor cursor_log = TableListElements( ctx, value.log ); // log + if ( !cursor_log.ok ) { return false; } + if ( cursor_log.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_log = ids.ref( 0x125073191daf5431ull, 25 ); + int64_t body_log = 0; + body_log += 1 + TableLebBytes( (uint64_t) ( cursor_log.count ) ); // the element kind byte and the count + for ( int32_t elem_i_log = 0; elem_i_log < cursor_log.count; elem_i_log++ ) + { + { + const LogEntry * slot_pointee_log = LogEntryAt( ctx, cursor_log[elem_i_log] ); + uint64_t slot_index_log = 0; + if ( slot_pointee_log != NULL && !TableNumberingIndex( numbering, (const void *) slot_pointee_log, slot_index_log ) ) { return false; } + body_log += TableLebBytes( slot_index_log ); + } + } + w.putleb( ref_log ); w.put8( 14 ); w.putleb( (uint64_t) body_log ); // log + w.put8( 17 ); w.putleb( (uint64_t) ( cursor_log.count ) ); + for ( int32_t elem_i_log = 0; elem_i_log < cursor_log.count; elem_i_log++ ) + { + { + const LogEntry * slot_pointee_log = LogEntryAt( ctx, cursor_log[elem_i_log] ); + uint64_t slot_index_log = 0; + if ( slot_pointee_log != NULL && !TableNumberingIndex( numbering, (const void *) slot_pointee_log, slot_index_log ) ) { return false; } + w.putleb( slot_index_log ); + } + } + } + } + { + TableListCursor cursor_scores = TableListElements( ctx, value.scores ); // scores + if ( !cursor_scores.ok ) { return false; } + if ( cursor_scores.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_scores = ids.ref( 0x01986b0b27400fb2ull, 26 ); + int64_t body_scores = 0; + body_scores += 1 + TableLebBytes( (uint64_t) ( cursor_scores.count ) ); // the element kind byte and the count + body_scores += (int64_t) ( cursor_scores.count ) * 4; + w.putleb( ref_scores ); w.put8( 14 ); w.putleb( (uint64_t) body_scores ); // scores + w.put8( 4 ); w.putleb( (uint64_t) ( cursor_scores.count ) ); + for ( int32_t elem_i_scores = 0; elem_i_scores < cursor_scores.count; elem_i_scores++ ) + { + w.put32( uint32_t( cursor_scores[elem_i_scores] ) ); + } + } + } + return !w.overflow; +} + +template +inline bool SaveSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Save & value ) +{ + if ( !SaveSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool SaveLoadBody( TableReader & r, const TableNodeMap & nodes, Save & value ) +{ + SaveReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xd24733aa574d4b09ull: // placements + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.placements, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Placement * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + uint64_t elem_len_placements = 0; + if ( !sub.getleb( elem_len_placements ) || !sub.room( elem_len_placements ) ) { r.report->malformed = true; break; } + { + TableReader elem_placements( sub.buffer + sub.offset, (int64_t) elem_len_placements, r.report, r.ids ); + PlacementLoadBody( elem_placements, ( *slot ) ); + } + sub.offset += (int64_t) elem_len_placements; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x125073191daf5431ull: // log + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 17 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.log, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + TableRef * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + { + uint64_t node_index_log = 0; + if ( !sub.getleb( node_index_log ) ) { r.report->malformed = true; break; } + TableNodeResolve( nodes, ( *slot ), node_index_log, 0x5e781536ac58825full, r.report ); // *LogEntry + } + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x01986b0b27400fb2ull: // scores + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 4 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.scores, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + int32_t * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + if ( !sub.has( 4 ) ) { r.report->malformed = true; break; } + int32_t decoded_v = int32_t( sub.get32( ) ); + ( *slot ) = decoded_v; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +inline int64_t PointMeasureBody( TableIds & ids, const Point & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.x != 0 ) { bytes += TableLebBytes( ids.ref( 0xaf63f54c86021707ull, 19 ) ) + 1 + 4; } // x + if ( value.y != 0 ) { bytes += TableLebBytes( ids.ref( 0xaf63f44c86021554ull, 20 ) ) + 1 + 4; } // y + return bytes; +} + +inline int64_t PointMeasure( const Point & value ) +{ + TableIds ids; + const int64_t body = PointMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool PointSaveBody( TableWriter & w, TableIds & ids, const Point & value ) +{ + if ( value.x != 0 ) + { + w.putleb( ids.ref( 0xaf63f54c86021707ull, 19 ) ); w.put8( 4 ); // x + w.put32( uint32_t( value.x ) ); + } + if ( value.y != 0 ) + { + w.putleb( ids.ref( 0xaf63f44c86021554ull, 20 ) ); w.put8( 4 ); // y + w.put32( uint32_t( value.y ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t PointSave( const Point & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !PointSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == PointMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool PointLoadBody( TableReader & r, Point & value ) +{ + PointReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xaf63f54c86021707ull: // x + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.x = decoded_v; + break; + } + case 0xaf63f44c86021554ull: // y + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.y = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict PointLoadVerdict( Point & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + PointReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + PointReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !PointLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool PointLoad( Point & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return PointLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t PointMeasureMessage( const Point & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = PointMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t PointSaveMessage( const Point & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !PointSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == PointMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool PointLoadMessage( Point & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + PointReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return PointLoadBody( r, value ); +} + +template +inline int64_t MixedMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Mixed & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // grades: a kind 14 array of kind 30 elements, INDEX order (§2.9) + TableListCursor cursor_grades = TableListElements( ctx, value.grades ); + if ( !cursor_grades.ok ) { return -1; } // the slot and the head disagree + if ( cursor_grades.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_grades = ids.ref( 0xd90a4e7682f799c5ull, 13 ); + int64_t body_grades = 0; + body_grades += 1 + TableLebBytes( (uint64_t) ( cursor_grades.count ) ); // the element kind byte and the count + for ( int32_t elem_i_grades = 0; elem_i_grades < cursor_grades.count; elem_i_grades++ ) + { + uint64_t elem_ref_grades = 0; + if ( !TableEnumRef( ids, cursor_grades[elem_i_grades], elem_ref_grades ) ) { return -1; } // no variant names this value + body_grades += TableLebBytes( elem_ref_grades ); + } + bytes += TableLebBytes( ref_grades ) + 1 + TableLebBytes( (uint64_t) ( body_grades ) ) + ( body_grades ); + } + } + { + // perms: a kind 14 array of kind 9 elements, INDEX order (§2.9) + TableListCursor cursor_perms = TableListElements( ctx, value.perms ); + if ( !cursor_perms.ok ) { return -1; } // the slot and the head disagree + if ( cursor_perms.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_perms = ids.ref( 0x4af2ed8470862ea8ull, 14 ); + int64_t body_perms = 0; + body_perms += 1 + TableLebBytes( (uint64_t) ( cursor_perms.count ) ); // the element kind byte and the count + body_perms += (int64_t) ( cursor_perms.count ) * 8; + bytes += TableLebBytes( ref_perms ) + 1 + TableLebBytes( (uint64_t) ( body_perms ) ) + ( body_perms ); + } + } + { + // hits: a kind 14 array of kind 15 elements, INDEX order (§2.9) + TableListCursor cursor_hits = TableListElements( ctx, value.hits ); + if ( !cursor_hits.ok ) { return -1; } // the slot and the head disagree + if ( cursor_hits.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_hits = ids.ref( 0x732dfbcc9b0cf0bbull, 15 ); + int64_t body_hits = 0; + body_hits += 1 + TableLebBytes( (uint64_t) ( cursor_hits.count ) ); // the element kind byte and the count + for ( int32_t elem_i_hits = 0; elem_i_hits < cursor_hits.count; elem_i_hits++ ) + { + if ( cursor_hits[elem_i_hits].type == HitType::None ) { body_hits += 1; } // a None element is the zero reference in its place + else + { + switch ( cursor_hits[elem_i_hits].type ) + { + case HitType::None: break; + case HitType::Point: + { + int64_t arm_payload_hitsu = 0; + const uint64_t arm_ref_hitsu = ids.ref( 0x73feab3544c345b1ull, 36 ); + { + const int64_t arm_body_hitsu = PointMeasureBody( ids, cursor_hits[elem_i_hits].point ); + if ( arm_body_hitsu < 0 ) { return -1; } + arm_payload_hitsu += arm_body_hitsu; // the arm's own table body (§3) + } + body_hits += TableLebBytes( arm_ref_hitsu ) + 1 + TableLebBytes( (uint64_t) ( arm_payload_hitsu ) ) + ( arm_payload_hitsu ); + break; + } + case HitType::Damage: + { + int64_t arm_payload_hitsu = 0; + const uint64_t arm_ref_hitsu = ids.ref( 0x7f6308be8ab37fc0ull, 37 ); + arm_payload_hitsu += 4; // int32 + body_hits += TableLebBytes( arm_ref_hitsu ) + 1 + TableLebBytes( (uint64_t) ( arm_payload_hitsu ) ) + ( arm_payload_hitsu ); + break; + } + default: return -1; // invalid tag — the write side refuses it too + } + } + } + bytes += TableLebBytes( ref_hits ) + 1 + TableLebBytes( (uint64_t) ( body_hits ) ) + ( body_hits ); + } + } + { + // bounds: a kind 14 array of kind 4 elements, INDEX order (§2.9) + TableListCursor cursor_bounds = TableListElements( ctx, value.bounds ); + if ( !cursor_bounds.ok ) { return -1; } // the slot and the head disagree + if ( cursor_bounds.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_bounds = ids.ref( 0x52f60c4caef0b768ull, 16 ); + int64_t body_bounds = 0; + body_bounds += 1 + TableLebBytes( (uint64_t) ( cursor_bounds.count ) ); // the element kind byte and the count + body_bounds += (int64_t) ( cursor_bounds.count ) * 4; + bytes += TableLebBytes( ref_bounds ) + 1 + TableLebBytes( (uint64_t) ( body_bounds ) ) + ( body_bounds ); + } + } + return bytes; +} + +template +inline bool MixedSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Mixed & value ) +{ + (void) ctx; (void) numbering; + { + TableListCursor cursor_grades = TableListElements( ctx, value.grades ); // grades + if ( !cursor_grades.ok ) { return false; } + if ( cursor_grades.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_grades = ids.ref( 0xd90a4e7682f799c5ull, 13 ); + int64_t body_grades = 0; + body_grades += 1 + TableLebBytes( (uint64_t) ( cursor_grades.count ) ); // the element kind byte and the count + for ( int32_t elem_i_grades = 0; elem_i_grades < cursor_grades.count; elem_i_grades++ ) + { + uint64_t elem_ref_grades = 0; + if ( !TableEnumRef( ids, cursor_grades[elem_i_grades], elem_ref_grades ) ) { return false; } // no variant names this value + body_grades += TableLebBytes( elem_ref_grades ); + } + w.putleb( ref_grades ); w.put8( 14 ); w.putleb( (uint64_t) body_grades ); // grades + w.put8( 30 ); w.putleb( (uint64_t) ( cursor_grades.count ) ); + for ( int32_t elem_i_grades = 0; elem_i_grades < cursor_grades.count; elem_i_grades++ ) + { + { + uint64_t element_ref_grades = 0; + if ( !TableEnumRef( ids, cursor_grades[elem_i_grades], element_ref_grades ) ) { return false; } + w.putleb( element_ref_grades ); + } + } + } + } + { + TableListCursor cursor_perms = TableListElements( ctx, value.perms ); // perms + if ( !cursor_perms.ok ) { return false; } + if ( cursor_perms.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_perms = ids.ref( 0x4af2ed8470862ea8ull, 14 ); + int64_t body_perms = 0; + body_perms += 1 + TableLebBytes( (uint64_t) ( cursor_perms.count ) ); // the element kind byte and the count + body_perms += (int64_t) ( cursor_perms.count ) * 8; + w.putleb( ref_perms ); w.put8( 14 ); w.putleb( (uint64_t) body_perms ); // perms + w.put8( 9 ); w.putleb( (uint64_t) ( cursor_perms.count ) ); + for ( int32_t elem_i_perms = 0; elem_i_perms < cursor_perms.count; elem_i_perms++ ) + { + w.put64( uint64_t( cursor_perms[elem_i_perms] ) ); + } + } + } + { + TableListCursor cursor_hits = TableListElements( ctx, value.hits ); // hits + if ( !cursor_hits.ok ) { return false; } + if ( cursor_hits.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_hits = ids.ref( 0x732dfbcc9b0cf0bbull, 15 ); + int64_t body_hits = 0; + body_hits += 1 + TableLebBytes( (uint64_t) ( cursor_hits.count ) ); // the element kind byte and the count + for ( int32_t elem_i_hits = 0; elem_i_hits < cursor_hits.count; elem_i_hits++ ) + { + if ( cursor_hits[elem_i_hits].type == HitType::None ) { body_hits += 1; } // a None element is the zero reference in its place + else + { + switch ( cursor_hits[elem_i_hits].type ) + { + case HitType::None: break; + case HitType::Point: + { + int64_t arm_payload_hitsu = 0; + const uint64_t arm_ref_hitsu = ids.ref( 0x73feab3544c345b1ull, 36 ); + { + const int64_t arm_body_hitsu = PointMeasureBody( ids, cursor_hits[elem_i_hits].point ); + if ( arm_body_hitsu < 0 ) { return -1; } + arm_payload_hitsu += arm_body_hitsu; // the arm's own table body (§3) + } + body_hits += TableLebBytes( arm_ref_hitsu ) + 1 + TableLebBytes( (uint64_t) ( arm_payload_hitsu ) ) + ( arm_payload_hitsu ); + break; + } + case HitType::Damage: + { + int64_t arm_payload_hitsu = 0; + const uint64_t arm_ref_hitsu = ids.ref( 0x7f6308be8ab37fc0ull, 37 ); + arm_payload_hitsu += 4; // int32 + body_hits += TableLebBytes( arm_ref_hitsu ) + 1 + TableLebBytes( (uint64_t) ( arm_payload_hitsu ) ) + ( arm_payload_hitsu ); + break; + } + default: return -1; // invalid tag — the write side refuses it too + } + } + } + w.putleb( ref_hits ); w.put8( 14 ); w.putleb( (uint64_t) body_hits ); // hits + w.put8( 15 ); w.putleb( (uint64_t) ( cursor_hits.count ) ); + for ( int32_t elem_i_hits = 0; elem_i_hits < cursor_hits.count; elem_i_hits++ ) + { + if ( cursor_hits[elem_i_hits].type == HitType::None ) { w.putleb( 0 ); } // a None element rides in its place + else + { + switch ( cursor_hits[elem_i_hits].type ) + { + case HitType::Point: + { + const uint64_t arm_ref_hitsu = ids.ref( 0x73feab3544c345b1ull, 36 ); + int64_t arm_payload_hitsu = 0; + { + const int64_t arm_body_hitsu = PointMeasureBody( ids, cursor_hits[elem_i_hits].point ); + if ( arm_body_hitsu < 0 ) { return false; } + arm_payload_hitsu += arm_body_hitsu; // the arm's own table body (§3) + } + w.putleb( arm_ref_hitsu ); w.put8( 13 ); w.putleb( (uint64_t) arm_payload_hitsu ); // point + if ( !PointSaveBody( w, ids, cursor_hits[elem_i_hits].point ) ) { return false; } + break; + } + case HitType::Damage: + { + const uint64_t arm_ref_hitsu = ids.ref( 0x7f6308be8ab37fc0ull, 37 ); + int64_t arm_payload_hitsu = 0; + arm_payload_hitsu += 4; // int32 + w.putleb( arm_ref_hitsu ); w.put8( 4 ); w.putleb( (uint64_t) arm_payload_hitsu ); // damage + w.put32( uint32_t( cursor_hits[elem_i_hits].damage ) ); + break; + } + default: return false; // write validates the tag before it rides + } + } + } + } + } + { + TableListCursor cursor_bounds = TableListElements( ctx, value.bounds ); // bounds + if ( !cursor_bounds.ok ) { return false; } + if ( cursor_bounds.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_bounds = ids.ref( 0x52f60c4caef0b768ull, 16 ); + int64_t body_bounds = 0; + body_bounds += 1 + TableLebBytes( (uint64_t) ( cursor_bounds.count ) ); // the element kind byte and the count + body_bounds += (int64_t) ( cursor_bounds.count ) * 4; + w.putleb( ref_bounds ); w.put8( 14 ); w.putleb( (uint64_t) body_bounds ); // bounds + w.put8( 4 ); w.putleb( (uint64_t) ( cursor_bounds.count ) ); + for ( int32_t elem_i_bounds = 0; elem_i_bounds < cursor_bounds.count; elem_i_bounds++ ) + { + w.put32( uint32_t( cursor_bounds[elem_i_bounds] ) ); + } + } + } + return !w.overflow; +} + +template +inline bool MixedSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Mixed & value ) +{ + if ( !MixedSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool MixedLoadBody( TableReader & r, const TableNodeMap & nodes, Mixed & value ) +{ + (void) nodes; + MixedReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xd90a4e7682f799c5ull: // grades + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 30 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.grades, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Grade * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + { + uint64_t variant_ref = 0; + if ( !sub.getleb( variant_ref ) ) { r.report->malformed = true; break; } + if ( variant_ref == 0 ) { ( *slot ) = Grade::None; } // the zero reference is the enum's None + else if ( variant_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; break; } + else if ( !TableEnumValue( r.ids->at( variant_ref ), ( *slot ) ) ) + { + ( *slot ) = Grade::None; + r.report->unknown++; + } + } + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x4af2ed8470862ea8ull: // perms + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 9 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.perms, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Perm * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + if ( !sub.has( 8 ) ) { r.report->malformed = true; break; } + uint64_t decoded_v = uint64_t( sub.get64( ) ); + ( *slot ) = decoded_v; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x732dfbcc9b0cf0bbull: // hits + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 15 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.hits, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Hit * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + { + uint64_t elem_arm_ref_hits = 0; + if ( !sub.getleb( elem_arm_ref_hits ) ) { r.report->malformed = true; break; } + ( *slot ).type = HitType::None; + if ( elem_arm_ref_hits != 0 ) // the zero reference is a None element in its place + { + if ( elem_arm_ref_hits > (uint64_t) r.ids->count ) { r.report->malformed = true; break; } + const uint64_t elem_arm_id_hits = r.ids->at( elem_arm_ref_hits ); + if ( !sub.has( 1 ) ) { r.report->malformed = true; break; } + const uint8_t elem_arm_kind_hits = sub.get8(); + uint64_t elem_arm_len_hits = 0; + if ( !sub.getleb( elem_arm_len_hits ) || !sub.room( elem_arm_len_hits ) ) { r.report->malformed = true; break; } + TableReader elem_arm_hits( sub.buffer + sub.offset, (int64_t) elem_arm_len_hits, r.report, r.ids ); + switch ( elem_arm_id_hits ) // the arm's NAME hash (docs/SPEC-TABLES.md §5) + { + case 0x73feab3544c345b1ull: // point + { + if ( elem_arm_kind_hits != 13 ) { ( *slot ).type = HitType::None; r.report->kind_mismatch++; break; } + ( *slot ).type = HitType::Point; + PointLoadBody( elem_arm_hits, ( *slot ).point ); + if ( elem_arm_hits.offset != elem_arm_hits.size ) { ( *slot ).type = HitType::None; r.report->malformed = true; break; } + break; + } + case 0x7f6308be8ab37fc0ull: // damage + { + if ( elem_arm_kind_hits != 4 ) { ( *slot ).type = HitType::None; r.report->kind_mismatch++; break; } + ( *slot ).type = HitType::Damage; + if ( elem_arm_hits.size != 4 ) { ( *slot ).type = HitType::None; r.report->malformed = true; break; } // an L that is not the kind's width is that arm's own framing damage (§3) + if ( !elem_arm_hits.has( 4 ) ) { ( *slot ).type = HitType::None; r.report->malformed = true; break; } + int32_t decoded_v = int32_t( elem_arm_hits.get32( ) ); + ( *slot ).damage = decoded_v; + break; + } + default: r.report->unknown++; break; // an arm this reader cannot name: the element reads None, the body skips by its length + } + sub.offset += (int64_t) elem_arm_len_hits; + } + } + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x52f60c4caef0b768ull: // bounds + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 4 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.bounds, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + int32_t * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + if ( !sub.has( 4 ) ) { r.report->malformed = true; break; } + int32_t decoded_v = int32_t( sub.get32( ) ); + if ( decoded_v < 0 ) { decoded_v = 0; r.report->clamped++; } + else if ( decoded_v > 100 ) { decoded_v = 100; r.report->clamped++; } + ( *slot ) = decoded_v; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// LogEntryWireExtent: the extent LogEntry's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool LogEntryWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + (void) body; (void) length; (void) at; (void) ids; (void) reason; // no list or map below this record + return true; +} + +// LogEntryExtentAt: the node extent LogEntry's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as LogEntryExtentPack advances it (§2.8, §2.9). +template +inline bool LogEntryExtentAt( const Ctx & ctx, const LogEntry & value, int64_t & at ) +{ + (void) ctx; (void) value; (void) at; // no list or map below this record + return true; +} + +// LogEntryExtentPack: carve LogEntry's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset LogEntryExtentAt advances (§2.8, §2.9). +template +inline bool LogEntryExtentPack( const Ctx & ctx, const LogEntry & src, LogEntry & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no list or map below this record + return true; +} + +// SaveWireExtent: the extent Save's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool SaveWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0xd24733aa574d4b09ull && field_kind == 14 ) // placements: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Placement ), (int64_t) alignof( Placement ), 13, 2, NULL, ids, reason ) ) { return false; } + continue; + } + if ( field_id == 0x125073191daf5431ull && field_kind == 14 ) // log: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( TableRef ), (int64_t) alignof( TableRef ), 17, 1, NULL, ids, reason ) ) { return false; } + continue; + } + if ( field_id == 0x01986b0b27400fb2ull && field_kind == 14 ) // scores: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( int32_t ), (int64_t) alignof( int32_t ), 4, 4, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// SaveExtentAt: the node extent Save's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as SaveExtentPack advances it (§2.8, §2.9). +template +inline bool SaveExtentAt( const Ctx & ctx, const Save & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.placements ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Placement ) - 1 ) & ~( (int64_t) alignof( Placement ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Placement ); // the whole array FIRST + } + { + TableListCursor cursor = TableListElements( ctx, value.log ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( TableRef ) - 1 ) & ~( (int64_t) alignof( TableRef ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( TableRef ); // the whole array FIRST + } + { + TableListCursor cursor = TableListElements( ctx, value.scores ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( int32_t ) - 1 ) & ~( (int64_t) alignof( int32_t ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( int32_t ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t SaveExtent( const Ctx & ctx, const Save & value ) +{ + int64_t at = 0; + if ( !SaveExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// SaveExtentPack: carve Save's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset SaveExtentAt advances (§2.8, §2.9). +template +inline bool SaveExtentPack( const Ctx & ctx, const Save & src, Save & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.placements ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Placement ) - 1 ) & ~( (int64_t) alignof( Placement ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Placement ); + if ( at + bytes > capacity ) { return false; } + Placement * placed = (Placement *) ( extent + at ); + at += bytes; + dst.placements.count = cursor.count; + dst.placements.padding = 0; + dst.placements.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.placements.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Placement ) ); // trivially copyable, by construction + } + } + { + TableListCursor cursor = TableListElements( ctx, src.log ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( TableRef ) - 1 ) & ~( (int64_t) alignof( TableRef ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( TableRef ); + if ( at + bytes > capacity ) { return false; } + TableRef * placed = (TableRef *) ( extent + at ); + at += bytes; + dst.log.count = cursor.count; + dst.log.padding = 0; + dst.log.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.log.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( TableRef ) ); // trivially copyable, by construction + } + } + { + TableListCursor cursor = TableListElements( ctx, src.scores ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( int32_t ) - 1 ) & ~( (int64_t) alignof( int32_t ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( int32_t ); + if ( at + bytes > capacity ) { return false; } + int32_t * placed = (int32_t *) ( extent + at ); + at += bytes; + dst.scores.count = cursor.count; + dst.scores.padding = 0; + dst.scores.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.scores.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( int32_t ) ); // trivially copyable, by construction + } + } + return true; +} + +// MixedWireExtent: the extent Mixed's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool MixedWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0xd90a4e7682f799c5ull && field_kind == 14 ) // grades: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Grade ), (int64_t) alignof( Grade ), 30, 1, NULL, ids, reason ) ) { return false; } + continue; + } + if ( field_id == 0x4af2ed8470862ea8ull && field_kind == 14 ) // perms: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Perm ), (int64_t) alignof( Perm ), 9, 8, NULL, ids, reason ) ) { return false; } + continue; + } + if ( field_id == 0x732dfbcc9b0cf0bbull && field_kind == 14 ) // hits: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Hit ), (int64_t) alignof( Hit ), 15, 1, NULL, ids, reason ) ) { return false; } + continue; + } + if ( field_id == 0x52f60c4caef0b768ull && field_kind == 14 ) // bounds: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( int32_t ), (int64_t) alignof( int32_t ), 4, 4, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// MixedExtentAt: the node extent Mixed's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as MixedExtentPack advances it (§2.8, §2.9). +template +inline bool MixedExtentAt( const Ctx & ctx, const Mixed & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.grades ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Grade ) - 1 ) & ~( (int64_t) alignof( Grade ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Grade ); // the whole array FIRST + } + { + TableListCursor cursor = TableListElements( ctx, value.perms ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Perm ) - 1 ) & ~( (int64_t) alignof( Perm ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Perm ); // the whole array FIRST + } + { + TableListCursor cursor = TableListElements( ctx, value.hits ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Hit ) - 1 ) & ~( (int64_t) alignof( Hit ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Hit ); // the whole array FIRST + } + { + TableListCursor cursor = TableListElements( ctx, value.bounds ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( int32_t ) - 1 ) & ~( (int64_t) alignof( int32_t ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( int32_t ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t MixedExtent( const Ctx & ctx, const Mixed & value ) +{ + int64_t at = 0; + if ( !MixedExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// MixedExtentPack: carve Mixed's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset MixedExtentAt advances (§2.8, §2.9). +template +inline bool MixedExtentPack( const Ctx & ctx, const Mixed & src, Mixed & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.grades ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Grade ) - 1 ) & ~( (int64_t) alignof( Grade ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Grade ); + if ( at + bytes > capacity ) { return false; } + Grade * placed = (Grade *) ( extent + at ); + at += bytes; + dst.grades.count = cursor.count; + dst.grades.padding = 0; + dst.grades.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.grades.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Grade ) ); // trivially copyable, by construction + } + } + { + TableListCursor cursor = TableListElements( ctx, src.perms ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Perm ) - 1 ) & ~( (int64_t) alignof( Perm ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Perm ); + if ( at + bytes > capacity ) { return false; } + Perm * placed = (Perm *) ( extent + at ); + at += bytes; + dst.perms.count = cursor.count; + dst.perms.padding = 0; + dst.perms.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.perms.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Perm ) ); // trivially copyable, by construction + } + } + { + TableListCursor cursor = TableListElements( ctx, src.hits ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Hit ) - 1 ) & ~( (int64_t) alignof( Hit ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Hit ); + if ( at + bytes > capacity ) { return false; } + Hit * placed = (Hit *) ( extent + at ); + at += bytes; + dst.hits.count = cursor.count; + dst.hits.padding = 0; + dst.hits.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.hits.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Hit ) ); // trivially copyable, by construction + } + } + { + TableListCursor cursor = TableListElements( ctx, src.bounds ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( int32_t ) - 1 ) & ~( (int64_t) alignof( int32_t ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( int32_t ); + if ( at + bytes > capacity ) { return false; } + int32_t * placed = (int32_t *) ( extent + at ); + at += bytes; + dst.bounds.count = cursor.count; + dst.bounds.padding = 0; + dst.bounds.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.bounds.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( int32_t ) ); // trivially copyable, by construction + } + } + return true; +} + +// ---- Save.placements: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Placement * SavePlacementsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool SavePlacementsErase( TableArena & arena, TableList & list, const Placement * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach SavePlacementsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Save.log: the builder's three (§2.9) ---- + +// ADD: the element is appended and handed back to fill. On a []*T that is +// the SLOT at null, which LogEntryEmplace fills as it fills any pointer slot, +// and a second slot may hold the same reference: two slots, one node. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline TableRef * SaveLogAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool SaveLogErase( TableArena & arena, TableList & list, const TableRef * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach SaveLogEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Save.scores: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline int32_t * SaveScoresAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool SaveScoresErase( TableArena & arena, TableList & list, const int32_t * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach SaveScoresEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Mixed.grades: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Grade * MixedGradesAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool MixedGradesErase( TableArena & arena, TableList & list, const Grade * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach MixedGradesEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Mixed.perms: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Perm * MixedPermsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool MixedPermsErase( TableArena & arena, TableList & list, const Perm * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach MixedPermsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Mixed.hits: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Hit * MixedHitsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool MixedHitsErase( TableArena & arena, TableList & list, const Hit * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach MixedHitsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Mixed.bounds: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline int32_t * MixedBoundsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool MixedBoundsErase( TableArena & arena, TableList & list, const int32_t * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach MixedBoundsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// LogEntryNumber: number everything LogEntry POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool LogEntryNumber( const Ctx & ctx, TableNumbering & numbering, const LogEntry & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// LogEntryPackMeasure: the packed region bytes of everything LogEntry POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t LogEntryPackMeasure( const Ctx & ctx, TablePackMap & seen, const LogEntry & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// LogEntryPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool LogEntryPackEdges( const Ctx & ctx, TablePackMap & seen, const LogEntry & src, LogEntry & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool LogEntryPack( const Ctx & ctx, TablePackMap & seen, const LogEntry & src, LogEntry & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( LogEntry ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( LogEntry ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !LogEntryExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return LogEntryPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool LogEntryPackEdges( const Ctx & ctx, TablePackMap & seen, const LogEntry & src, LogEntry & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// SaveNumber: number everything Save POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool SaveNumber( const Ctx & ctx, TableNumbering & numbering, const Save & value ) +{ + { // log: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_log = TableListElements( ctx, value.log ); + if ( !cursor_log.ok ) { return false; } + for ( int32_t i = 0; i < cursor_log.count; i++ ) + { + { + const LogEntry * pointee = LogEntryAt( ctx, cursor_log[i] ); // log + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( numbering.seen, (const void *) pointee, + (int64_t) ( numbering.count + 2 ), taken, slot ); // its index, if this is its first visit + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + } + else + { + TableNodeEntry node; + node.node = (const void *) pointee; + node.type_id = 0x5e781536ac58825full; // fnv1a64( "LogEntry" ) + node.type_slot = 50; // its slot in the unit's vocabulary (§3.3) + node.measure = &TableNodeMeasureThunk; + node.save = &TableNodeSaveThunk; + if ( !TableNumberingAppend( numbering, node ) ) { return false; } + if ( !LogEntryNumber( ctx, numbering, *pointee ) ) { return false; } + TablePackMapClose( numbering.seen, (const void *) pointee, slot ); + } + } + } + } + } + return true; +} + +// SavePackMeasure: the packed region bytes of everything Save POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t SavePackMeasure( const Ctx & ctx, TablePackMap & seen, const Save & value ) +{ + int64_t bytes = 0; + { // log: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_log = TableListElements( ctx, value.log ); + if ( !cursor_log.ok ) { return -1; } + for ( int32_t i = 0; i < cursor_log.count; i++ ) + { + { + const LogEntry * pointee = LogEntryAt( ctx, cursor_log[i] ); // log + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, 0, taken, slot ); + if ( entry == NULL ) { return -1; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return -1; } // a data cycle + } + else + { + int64_t inner = LogEntryPackMeasure( ctx, seen, *pointee ); + if ( inner < 0 ) { return -1; } + TablePackMapClose( seen, (const void *) pointee, slot ); + bytes += TableAlignUp64( (int64_t) sizeof( LogEntry ) ) + inner; + } + } + } + } + } + return bytes; +} + +// SavePack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool SavePackEdges( const Ctx & ctx, TablePackMap & seen, const Save & src, Save & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool SavePack( const Ctx & ctx, TablePackMap & seen, const Save & src, Save & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Save ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Save ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !SaveExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return SavePackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool SavePackEdges( const Ctx & ctx, TablePackMap & seen, const Save & src, Save & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + { // log: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_log = TableListElements( ctx, src.log ); + if ( !cursor_log.ok ) { return false; } + TableRef * placed_log = (TableRef *) ( dst.log.elements.value != 0 ? ( (uint8_t *) &dst.log.elements + dst.log.elements.value ) : NULL ); + for ( int32_t i = 0; i < cursor_log.count; i++ ) + { + { + placed_log[i].value = 0; // log + const LogEntry * pointee = LogEntryAt( ctx, cursor_log[i] ); + if ( pointee != NULL ) + { + int64_t at = TableAlignUp64( used ); // where it WOULD land, if this is its first visit + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, at, taken, slot ); + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + placed_log[i].value = (int64_t) ( ( base + entry->offset ) - (const uint8_t *) &placed_log[i] ); // the one body it already has + } + else + { + if ( at + (int64_t) sizeof( LogEntry ) > capacity ) { return false; } + used = at + TableAlignUp64( (int64_t) sizeof( LogEntry ) ); + LogEntry * child = new ( base + at ) LogEntry; // lifetime only: the Pack below memcpy's the whole node over it + placed_log[i].value = (int64_t) ( ( base + at ) - (const uint8_t *) &placed_log[i] ); + if ( !LogEntryPack( ctx, seen, *pointee, *child, base, capacity, used ) ) { return false; } + TablePackMapClose( seen, (const void *) pointee, slot ); + } + } + } + } + } + return true; +} + +// MixedNumber: number everything Mixed POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool MixedNumber( const Ctx & ctx, TableNumbering & numbering, const Mixed & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// MixedPackMeasure: the packed region bytes of everything Mixed POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t MixedPackMeasure( const Ctx & ctx, TablePackMap & seen, const Mixed & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// MixedPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool MixedPackEdges( const Ctx & ctx, TablePackMap & seen, const Mixed & src, Mixed & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool MixedPack( const Ctx & ctx, TablePackMap & seen, const Mixed & src, Mixed & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Mixed ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Mixed ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !MixedExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return MixedPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool MixedPackEdges( const Ctx & ctx, TablePackMap & seen, const Mixed & src, Mixed & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// ---- Save: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: SaveBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Save is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct SaveBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + SaveBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~SaveBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + SaveBuilder( const SaveBuilder & ) = delete; + SaveBuilder & operator=( const SaveBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Save * GetRoot() { return arena.locked ? NULL : (Save *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Save * AsConst() const { return (const Save *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool SaveBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Save & root = *(const Save *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = SavePackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = SaveExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + Save * destination = new ( packed ) Save; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !SavePack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Save on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// SaveNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t SaveNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + case 0x5e781536ac58825full: return TableAlignUp64( (int64_t) sizeof( LogEntry ) ); // LogEntry + default: break; + } + return -1; +} + +// SaveNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void SaveNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + switch ( type_id ) + { + case 0x5e781536ac58825full: { LogEntry * node = new ( at ) LogEntry; LogEntryReset( *node ); break; } // LogEntry + default: break; + } +} + +// SaveNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t SaveNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + case 0x5e781536ac58825full: return TableAlignUp64( (int64_t) sizeof( LogEntry ) ); // LogEntry + default: break; + } + return 0; +} + +// SaveNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t SaveNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + switch ( type_id ) + { + case 0x5e781536ac58825full: return (uint32_t) worker.Alloc().ref.value; // LogEntry + default: break; + } + return 0; +} + +// SaveNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void SaveNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = SaveNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? SaveNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + case 0x5e781536ac58825full: LogEntryLoadBody( r, *(LogEntry *) at ); break; // LogEntry + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool SaveNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Save & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return SaveNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t SaveMeasureWire( const Ctx & ctx, const Save & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( SaveNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = SaveMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t SaveSaveWire( const Ctx & ctx, const Save & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !SaveNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = SaveSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == SaveMeasure( root ) +} + +inline int64_t SaveMeasure( const Save * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SaveMeasureWire( ctx, *root, allocator ); +} + +inline int64_t SaveSave( const Save * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SaveSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t SaveMeasure( const SaveBuilder & builder ) +{ + if ( builder.region != NULL ) { return SaveMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SaveMeasureWire( ctx, *(const Save *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t SaveSave( const SaveBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return SaveSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SaveSaveWire( ctx, *(const Save *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t SaveMeasureMessage( const Save * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SaveMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t SaveSaveMessage( const Save * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SaveSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t SaveMeasureMessage( const SaveBuilder & builder ) +{ + if ( builder.region != NULL ) { return SaveMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return SaveMeasureWire( ctx, *(const Save *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t SaveSaveMessage( const SaveBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return SaveSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return SaveSaveWire( ctx, *(const Save *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// SaveLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t SaveLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !SaveWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SaveNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// SaveLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Save * SaveLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Save ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !SaveWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SaveNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x33f85f24c0f5f008ull; + Save * root = new ( region ) Save; // lifetime only: LoadBody's first act is SaveReset + SaveReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = SaveNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + SaveNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SaveNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Save ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + SaveLoadBody( r, nodes, *root ); + return root; +} + +// SaveLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t SaveLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !SaveWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SaveNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// SaveLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Save * SaveLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Save ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !SaveWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SaveNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x33f85f24c0f5f008ull; + Save * root = new ( region ) Save; // lifetime only: LoadBody's first act is SaveReset + SaveReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = SaveNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + SaveNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SaveNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Save ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + SaveLoadBody( r, nodes, *root ); + return root; +} + +// SaveLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool SaveLoadBuilder( SaveBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Save * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0x33f85f24c0f5f008ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = SaveNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SaveNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = SaveLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Mixed: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: MixedBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Mixed is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct MixedBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + MixedBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~MixedBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + MixedBuilder( const MixedBuilder & ) = delete; + MixedBuilder & operator=( const MixedBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Mixed * GetRoot() { return arena.locked ? NULL : (Mixed *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Mixed * AsConst() const { return (const Mixed *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool MixedBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Mixed & root = *(const Mixed *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = MixedPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = MixedExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + Mixed * destination = new ( packed ) Mixed; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !MixedPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Mixed on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// MixedNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t MixedNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// MixedNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void MixedNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// MixedNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t MixedNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// MixedNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t MixedNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// MixedNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void MixedNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = MixedNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? MixedNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool MixedNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Mixed & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return MixedNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t MixedMeasureWire( const Ctx & ctx, const Mixed & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( MixedNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = MixedMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t MixedSaveWire( const Ctx & ctx, const Mixed & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !MixedNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = MixedSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == MixedMeasure( root ) +} + +inline int64_t MixedMeasure( const Mixed * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return MixedMeasureWire( ctx, *root, allocator ); +} + +inline int64_t MixedSave( const Mixed * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return MixedSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t MixedMeasure( const MixedBuilder & builder ) +{ + if ( builder.region != NULL ) { return MixedMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return MixedMeasureWire( ctx, *(const Mixed *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t MixedSave( const MixedBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return MixedSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return MixedSaveWire( ctx, *(const Mixed *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t MixedMeasureMessage( const Mixed * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return MixedMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t MixedSaveMessage( const Mixed * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return MixedSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t MixedMeasureMessage( const MixedBuilder & builder ) +{ + if ( builder.region != NULL ) { return MixedMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return MixedMeasureWire( ctx, *(const Mixed *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t MixedSaveMessage( const MixedBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return MixedSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return MixedSaveWire( ctx, *(const Mixed *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// MixedLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t MixedLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !MixedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = MixedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// MixedLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Mixed * MixedLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Mixed ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !MixedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = MixedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xbb86c6c96f598bb8ull; + Mixed * root = new ( region ) Mixed; // lifetime only: LoadBody's first act is MixedReset + MixedReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = MixedNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + MixedNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + MixedNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Mixed ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + MixedLoadBody( r, nodes, *root ); + return root; +} + +// MixedLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t MixedLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !MixedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = MixedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// MixedLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Mixed * MixedLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Mixed ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !MixedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = MixedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xbb86c6c96f598bb8ull; + Mixed * root = new ( region ) Mixed; // lifetime only: LoadBody's first act is MixedReset + MixedReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = MixedNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + MixedNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + MixedNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Mixed ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + MixedLoadBody( r, nodes, *root ); + return root; +} + +// MixedLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool MixedLoadBuilder( MixedBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Mixed * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0xbb86c6c96f598bb8ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = MixedNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + MixedNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = MixedLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- the cooked form: point at a cook (docs/SPEC-TABLES.md §7) ---- + +// PlacementOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Placement IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Placement * PlacementOpen( const void * bytes, uint64_t length ) +{ + return (const Placement *) TableCookOpen( bytes, length, (uint64_t) sizeof( Placement ), (uint64_t) alignof( Placement ) ); +} + +// LogEntryOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// LogEntry IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const LogEntry * LogEntryOpen( const void * bytes, uint64_t length ) +{ + return (const LogEntry *) TableCookOpen( bytes, length, (uint64_t) sizeof( LogEntry ), (uint64_t) alignof( LogEntry ) ); +} + +// SaveOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH SaveAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Save * SaveOpen( const void * bytes, uint64_t length ) +{ + return (const Save *) TableCookOpen( bytes, length, (uint64_t) sizeof( Save ), (uint64_t) alignof( Save ) ); +} + +// PointOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Point IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Point * PointOpen( const void * bytes, uint64_t length ) +{ + return (const Point *) TableCookOpen( bytes, length, (uint64_t) sizeof( Point ), (uint64_t) alignof( Point ) ); +} + +// MixedOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH MixedAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Mixed * MixedOpen( const void * bytes, uint64_t length ) +{ + return (const Mixed *) TableCookOpen( bytes, length, (uint64_t) sizeof( Mixed ), (uint64_t) alignof( Mixed ) ); +} + +// ---- the cooked form: WRITE a cook (docs/SPEC-TABLES.md §7.6) ---- +// +// The bytes are `schema cook`'s, and the tool stays the reference: the two +// writers are held to one file, byte for byte, in both byte orders. A cook is +// content-addressed by (asset hash, build version), so two writers of one +// instance produce ONE artifact or the pair means nothing. + +inline void PlacementCookBody( uint8_t * at, const Placement & value, TableByteOrder order ); +inline void LogEntryCookBody( uint8_t * at, const LogEntry & value, TableByteOrder order ); +template inline bool SaveCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Save & value, TableByteOrder order ); +inline void PointCookBody( uint8_t * at, const Point & value, TableByteOrder order ); +template inline bool MixedCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Mixed & value, TableByteOrder order ); + +inline void PlacementCookBody( uint8_t * at, const Placement & value, TableByteOrder order ) +{ + { uint32_t bits = 0; memcpy( &bits, &value.x, 4 ); table_cook_put( at + 0, (uint64_t) bits, 4, order ); } + { uint32_t bits = 0; memcpy( &bits, &value.y, 4 ); table_cook_put( at + 4, (uint64_t) bits, 4, order ); } + table_cook_put( at + 8, (uint64_t) value.model, 4, order ); +} + +inline void LogEntryCookBody( uint8_t * at, const LogEntry & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.tick, 4, order ); +} + +template inline bool SaveCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Save & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + (void) value; + table_cook_put( at + 0, 0, 8, order ); // placements: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, 0, 8, order ); // log: the array's delta, filled by the extent writer + table_cook_put( at + 24, 0, 4, order ); // and its count + table_cook_put( at + 32, 0, 8, order ); // scores: the array's delta, filled by the extent writer + table_cook_put( at + 40, 0, 4, order ); // and its count + return true; +} + +inline void PointCookBody( uint8_t * at, const Point & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.x, 4, order ); + table_cook_put( at + 4, (uint64_t) value.y, 4, order ); +} + +template inline bool MixedCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Mixed & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + (void) value; + table_cook_put( at + 0, 0, 8, order ); // grades: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, 0, 8, order ); // perms: the array's delta, filled by the extent writer + table_cook_put( at + 24, 0, 4, order ); // and its count + table_cook_put( at + 32, 0, 8, order ); // hits: the array's delta, filled by the extent writer + table_cook_put( at + 40, 0, 4, order ); // and its count + table_cook_put( at + 48, 0, 8, order ); // bounds: the array's delta, filled by the extent writer + table_cook_put( at + 56, 0, 4, order ); // and its count + return true; +} + +template inline bool PlacementCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Placement & value, TableByteOrder order ); +template inline bool LogEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const LogEntry & value, TableByteOrder order ); +template inline bool SaveCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Save & value, TableByteOrder order ); +template inline bool PointCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Point & value, TableByteOrder order ); +template inline bool MixedCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Mixed & value, TableByteOrder order ); + +// PlacementCookExtent: Placement's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool PlacementCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Placement & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// LogEntryCookExtent: LogEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool LogEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const LogEntry & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// SaveCookExtent: Save's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SaveCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Save & value, TableByteOrder order ) +{ + { // placements: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.placements ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( Placement ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 12; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + PlacementCookBody( array + i * 12, cursor[i], order ); + } + } + { // log: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.log ); + if ( !cursor.ok ) { return false; } + at = ( at + 7 ) & ~(int64_t) 7; // at alignof( TableRef ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 8; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 16, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 16 ) ) : 0, 8, order ); + table_cook_put( record + 24, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + if ( !table_cook_ref( region, array + i * 8, (const void *) LogEntryAt( ctx, cursor[i] ), order ) ) { return false; } + } + } + { // scores: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.scores ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( int32_t ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 4; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 32, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 32 ) ) : 0, 8, order ); + table_cook_put( record + 40, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + table_cook_put( array + i * 4, (uint64_t) cursor[i], 4, order ); + } + } + return true; +} + +// PointCookExtent: Point's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool PointCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Point & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// MixedCookExtent: Mixed's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool MixedCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Mixed & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // grades: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.grades ); + if ( !cursor.ok ) { return false; } + at = ( at + 0 ) & ~(int64_t) 0; // at alignof( Grade ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 1; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + table_cook_put( array + i * 1, (uint64_t) cursor[i], 1, order ); + } + } + { // perms: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.perms ); + if ( !cursor.ok ) { return false; } + at = ( at + 7 ) & ~(int64_t) 7; // at alignof( Perm ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 8; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 16, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 16 ) ) : 0, 8, order ); + table_cook_put( record + 24, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + table_cook_put( array + i * 8, (uint64_t) cursor[i], 8, order ); // a mask rides raw, in every target + } + } + { // hits: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.hits ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( Hit ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 12; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 32, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 32 ) ) : 0, 8, order ); + table_cook_put( record + 40, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + { + table_cook_put( array + i * 12, (uint64_t) cursor[i].type, 1, order ); // the tag; None is the tag alone + switch ( cursor[i].type ) + { + case HitType::Point: PointCookBody( array + i * 12 + 4, cursor[i].point, order ); break; + case HitType::Damage: + { + table_cook_put( array + i * 12 + 4, (uint64_t) cursor[i].damage, 4, order ); + break; + } + default: break; // every byte outside the set arm stays zero + } + } + } + } + { // bounds: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.bounds ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( int32_t ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 4; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 48, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 48 ) ) : 0, 8, order ); + table_cook_put( record + 56, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + table_cook_put( array + i * 4, (uint64_t) cursor[i], 4, order ); + } + } + return true; +} + +// PlacementCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool PlacementCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Placement & value, TableByteOrder order ) +{ + PlacementCookBody( at, value, order ); + int64_t extent_at = 0; + return PlacementCookExtent( ctx, region, at + 16, extent_at, at, value, order ); +} + +// LogEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool LogEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const LogEntry & value, TableByteOrder order ) +{ + LogEntryCookBody( at, value, order ); + int64_t extent_at = 0; + return LogEntryCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// SaveCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool SaveCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Save & value, TableByteOrder order ) +{ + if ( !SaveCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return SaveCookExtent( ctx, region, at + 48, extent_at, at, value, order ); +} + +// PointCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool PointCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Point & value, TableByteOrder order ) +{ + PointCookBody( at, value, order ); + int64_t extent_at = 0; + return PointCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// MixedCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool MixedCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Mixed & value, TableByteOrder order ) +{ + if ( !MixedCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return MixedCookExtent( ctx, region, at + 64, extent_at, at, value, order ); +} + +// PlacementCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Placement IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t PlacementCookMeasure( const Placement & value ) +{ + (void) value; + return 96; // 64 header + 16 data + 16 attribution +} + +// PlacementCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract PlacementMeasure/PlacementSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool PlacementCook( const Placement & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) PlacementCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 16, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + PlacementCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 80, 0, 8, order ); + table_cook_put( raw + 88, 0x41f721f1b93aea80ull, 8, order ); + return true; +} + +// LogEntryCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// LogEntry IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t LogEntryCookMeasure( const LogEntry & value ) +{ + (void) value; + return 88; // 64 header + 8 data + 16 attribution +} + +// LogEntryCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract LogEntryMeasure/LogEntrySave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool LogEntryCook( const LogEntry & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) LogEntryCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 8, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + LogEntryCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 72, 0, 8, order ); + table_cook_put( raw + 80, 0x5e781536ac58825full, 8, order ); + return true; +} + +// SaveCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool SaveCookLayout( const Ctx & ctx, const Save & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = SaveExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 48 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + case 0x5e781536ac58825full: size = 4; node_align = 4; break; // LogEntry + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// SaveCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t SaveCookMeasureFrom( const Ctx & ctx, const Save & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( SaveNumberFrom( ctx, numbering, root ) && SaveCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// SaveCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool SaveCookFrom( const Ctx & ctx, const Save & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = SaveNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && SaveCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = SaveCookNode( ctx, region, region.base, root, order ); + for ( int64_t k = 0; ok && k < numbering.count; k++ ) + { + uint8_t * at = region.base + region.offsets[k + 1]; + const void * node = numbering.entries[k].node; + switch ( numbering.entries[k].type_id ) + { + case 0x5e781536ac58825full: ok = LogEntryCookNode( ctx, region, at, *(const LogEntry *) node, order ); break; // LogEntry + default: ok = false; break; + } + } + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0x33f85f24c0f5f008ull, 8, order ); // the root: fnv1a64( "Save" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// SaveCookMeasure / SaveCook over a REGION root — a locked builder's AsConst, a +// region SaveLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t SaveCookMeasure( const Save * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SaveCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool SaveCook( const Save * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return SaveCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t SaveCookMeasure( const SaveBuilder & builder ) +{ + if ( builder.region != NULL ) { return SaveCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SaveCookMeasureFrom( ctx, *(const Save *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool SaveCook( const SaveBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return SaveCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SaveCookFrom( ctx, *(const Save *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// PointCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Point IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t PointCookMeasure( const Point & value ) +{ + (void) value; + return 88; // 64 header + 8 data + 16 attribution +} + +// PointCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract PointMeasure/PointSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool PointCook( const Point & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) PointCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 8, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + PointCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 72, 0, 8, order ); + table_cook_put( raw + 80, 0x8a439296ced9ed11ull, 8, order ); + return true; +} + +// MixedCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool MixedCookLayout( const Ctx & ctx, const Mixed & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = MixedExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 64 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// MixedCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t MixedCookMeasureFrom( const Ctx & ctx, const Mixed & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( MixedNumberFrom( ctx, numbering, root ) && MixedCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// MixedCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool MixedCookFrom( const Ctx & ctx, const Mixed & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = MixedNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && MixedCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = MixedCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0xbb86c6c96f598bb8ull, 8, order ); // the root: fnv1a64( "Mixed" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// MixedCookMeasure / MixedCook over a REGION root — a locked builder's AsConst, a +// region MixedLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t MixedCookMeasure( const Mixed * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return MixedCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool MixedCook( const Mixed * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return MixedCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t MixedCookMeasure( const MixedBuilder & builder ) +{ + if ( builder.region != NULL ) { return MixedCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return MixedCookMeasureFrom( ctx, *(const Mixed *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool MixedCook( const MixedBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return MixedCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return MixedCookFrom( ctx, *(const Mixed *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ---- relocatability, enforced: the wire is a pure length-prefixed +// stream AND the decoded storage is pointer-free — every closure type +// must stay trivially copyable and standard-layout, so instances can be +// memcpy'd, mmap'd, shared across processes, and walked through +// descriptor offsets. A failure here means a pointer, virtual or +// non-trivial member crept into generated storage. +// +// They ask the COMPILER ITSELF, which is what every C++ standard library +// answers the same two questions with — and it costs this header no +// include at all. +// A pointer FIELD is a TableRef — eight bytes and no address — so the +// property holds in BOTH forms: a fixed-size table is one relocatable +// struct, and a packed region is one relocatable block whose references +// are self-relative and therefore survive a plain memcpy. +static_assert( __is_trivially_copyable( Placement ), "Placement must stay relocatable" ); +static_assert( __is_standard_layout( Placement ), "Placement must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( LogEntry ), "LogEntry must stay relocatable" ); +static_assert( __is_standard_layout( LogEntry ), "LogEntry must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Save ), "Save must stay relocatable" ); +static_assert( __is_standard_layout( Save ), "Save must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Point ), "Point must stay relocatable" ); +static_assert( __is_standard_layout( Point ), "Point must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Mixed ), "Mixed must stay relocatable" ); +static_assert( __is_standard_layout( Mixed ), "Mixed must stay standard-layout for offsetof" ); + +// ---- the cook's layout contract (docs/SPEC-TABLES.md §20.3) ---- +// +// The compiler derived every number below from the declaration and folded it +// into the BUILD VERSION; these asserts are this compiler saying whether it +// agrees. The model is not self-evidently right — on 32-bit System V +// alignof(uint64_t) is 4, not 8 — which is precisely why it is asserted +// rather than assumed. +static_assert( sizeof( Placement ) == 12, "Placement's sizeof moved: the build version was taken over 12, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Placement ) == 4, "Placement's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Placement, x ) == 0, "Placement's field x moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Placement, y ) == 4, "Placement's field y moved: the build version was taken over offset 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Placement, model ) == 8, "Placement's field model moved: the build version was taken over offset 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( LogEntry ) == 4, "LogEntry's sizeof moved: the build version was taken over 4, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( LogEntry ) == 4, "LogEntry's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( LogEntry, tick ) == 0, "LogEntry's field tick moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Save ) == 48, "Save's sizeof moved: the build version was taken over 48, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Save ) == 8, "Save's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Save, placements ) == 0, "Save's field placements moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Save, log ) == 16, "Save's field log moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Save, scores ) == 32, "Save's field scores moved: the build version was taken over offset 32 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Point ) == 8, "Point's sizeof moved: the build version was taken over 8, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Point ) == 4, "Point's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Point, x ) == 0, "Point's field x moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Point, y ) == 4, "Point's field y moved: the build version was taken over offset 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Mixed ) == 64, "Mixed's sizeof moved: the build version was taken over 64, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Mixed ) == 8, "Mixed's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Mixed, grades ) == 0, "Mixed's field grades moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Mixed, perms ) == 16, "Mixed's field perms moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Mixed, hits ) == 32, "Mixed's field hits moved: the build version was taken over offset 32 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Mixed, bounds ) == 48, "Mixed's field bounds moved: the build version was taken over offset 48 (docs/SPEC-TABLES.md §20.3)" ); + +static_assert( alignof( Placement ) <= kTableAlign, "Save.placements: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( TableRef ) <= kTableAlign, "Save.log: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( int32_t ) <= kTableAlign, "Save.scores: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( Grade ) <= kTableAlign, "Mixed.grades: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( Perm ) <= kTableAlign, "Mixed.perms: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( Hit ) <= kTableAlign, "Mixed.hits: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( int32_t ) <= kTableAlign, "Mixed.bounds: an unbounded array's element alignment must fit the arena's" ); + +// ---- reflection descriptors (tables only, docs/SPEC-TABLES.md) ---- + +inline const TableTypeInfo * PlacementTableType(); +inline const TableTypeInfo * LogEntryTableType(); +inline const TableTypeInfo * SaveTableType(); +inline const TableTypeInfo * PointTableType(); +inline const TableTypeInfo * MixedTableType(); +// The descriptors are CONSTANT-INITIALISED data, and a field's target is +// the ADDRESS of another descriptor. These declarations are what let a +// self- or mutually-referential graph — Node naming itself through *Node — +// be expressed as constant data instead of a lazy link, which could not +// have been written race-free OR recursion-safe. The whole reflection +// surface is therefore immutable: read it from any thread, any time. +extern const TableTypeInfo PlacementTableInfo; +extern const TableTypeInfo LogEntryTableInfo; +extern const TableTypeInfo SaveTableInfo; +extern const TableTypeInfo PointTableInfo; +extern const TableTypeInfo MixedTableInfo; + +inline const TableFieldInfo PlacementTableFields[] = { + { "x", "x", "float32", 0xaf63f54c86021707ull, 10, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Placement, x ), (uint32_t) sizeof( Placement::x ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "y", "y", "float32", 0xaf63f44c86021554ull, 10, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Placement, y ), (uint32_t) sizeof( Placement::y ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "model", "model", "uint32", 0x9de543933e6e703aull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Placement, model ), (uint32_t) sizeof( Placement::model ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo PlacementTableInfo = { "Placement", (uint32_t) sizeof( Placement ), 3, PlacementTableFields, +[]( void * p ) { PlacementReset( *(Placement *) p ); }, false }; +inline const TableTypeInfo * PlacementTableType() { return &PlacementTableInfo; } + +inline const TableFieldInfo LogEntryTableFields[] = { + { "tick", "tick", "uint32", 0x1e7683ef2ebc7684ull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( LogEntry, tick ), (uint32_t) sizeof( LogEntry::tick ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo LogEntryTableInfo = { "LogEntry", (uint32_t) sizeof( LogEntry ), 1, LogEntryTableFields, +[]( void * p ) { LogEntryReset( *(LogEntry *) p ); }, false }; +inline const TableTypeInfo * LogEntryTableType() { return &LogEntryTableInfo; } + +inline const TableFieldInfo SaveTableFields[] = { + { "placements", "placements", "Placement", 0xd24733aa574d4b09ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Save, placements ), (uint32_t) sizeof( Placement ), (uint32_t) offsetof( Save, placements.count ), 0xffffffffu, &PlacementTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "log", "log", "LogEntry", 0x125073191daf5431ull, 17, true, true, []( const void * slot ) -> const void * { return (const void *) LogEntryAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) LogEntryEmplace( worker, *(TableRef *) slot ); }, true, false, 0, (uint32_t) offsetof( Save, log ), (uint32_t) sizeof( TableRef ), (uint32_t) offsetof( Save, log.count ), 0xffffffffu, &LogEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "scores", "scores", "int32", 0x01986b0b27400fb2ull, 4, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Save, scores ), (uint32_t) sizeof( int32_t ), (uint32_t) offsetof( Save, scores.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, +}; +inline const TableTypeInfo SaveTableInfo = { "Save", (uint32_t) sizeof( Save ), 3, SaveTableFields, +[]( void * p ) { SaveReset( *(Save *) p ); }, true }; +inline const TableTypeInfo * SaveTableType() { return &SaveTableInfo; } + +inline const TableFieldInfo PointTableFields[] = { + { "x", "x", "int32", 0xaf63f54c86021707ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Point, x ), (uint32_t) sizeof( Point::x ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "y", "y", "int32", 0xaf63f44c86021554ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Point, y ), (uint32_t) sizeof( Point::y ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo PointTableInfo = { "Point", (uint32_t) sizeof( Point ), 2, PointTableFields, +[]( void * p ) { PointReset( *(Point *) p ); }, false }; +inline const TableTypeInfo * PointTableType() { return &PointTableInfo; } + +inline const TableFieldInfo MixedTableFields[] = { + { "grades", "grades", "Grade", 0xd90a4e7682f799c5ull, 30, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Mixed, grades ), (uint32_t) sizeof( Grade ), (uint32_t) offsetof( Mixed, grades.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, 3, +[]( uint64_t v ) { return EnumName( Grade( v ) ); }, +[]( uint64_t v ) -> uint64_t { uint64_t id = 0; TableEnumId( Grade( v ), id ); return id; }, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "perms", "perms", "Perm", 0x4af2ed8470862ea8ull, 9, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Mixed, perms ), (uint32_t) sizeof( Perm ), (uint32_t) offsetof( Mixed, perms.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, 2, +[]( uint64_t v ) { return FlagNamePerm( (int) v ); }, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "hits", "hits", "Hit", 0x732dfbcc9b0cf0bbull, 15, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Mixed, hits ), (uint32_t) sizeof( Hit ), (uint32_t) offsetof( Mixed, hits.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, 2, +[]( uint64_t v ) -> const char * { switch ( v ) { case 0: return "None"; case 1: return "point"; case 2: return "damage"; default: return "???"; } }, +[]( uint64_t v ) -> uint64_t { switch ( v ) { case 0: return 0; case 1: return 0x73feab3544c345b1ull; case 2: return 0x7f6308be8ab37fc0ull; default: return 0; } }, NULL, NULL, NULL, +[]() -> const TableUnionInfo * { static const TableFieldInfo arm_fields_Hit[] = { { "damage", "damage", "int32", 0x7f6308be8ab37fc0ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Hit, damage ), (uint32_t) sizeof( Hit::damage ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; static const TableUnionArmInfo arms[] = { { 0, NULL, NULL, 0 }, { (uint32_t) offsetof( Hit, point ), &PointTableInfo, NULL, 8 }, { (uint32_t) offsetof( Hit, damage ), NULL, &arm_fields_Hit[0], 4 }, }; static const TableUnionInfo info = { (uint32_t) offsetof( Hit, type ), (uint32_t) sizeof( Hit::type ), arms }; return &info; }, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "bounds", "bounds", "int32", 0x52f60c4caef0b768ull, 4, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Mixed, bounds ), (uint32_t) sizeof( int32_t ), (uint32_t) offsetof( Mixed, bounds.count ), 0xffffffffu, NULL, true, 0.0, 100.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, +}; +inline const TableTypeInfo MixedTableInfo = { "Mixed", (uint32_t) sizeof( Mixed ), 4, MixedTableFields, +[]( void * p ) { MixedReset( *(Mixed *) p ); }, true }; +inline const TableTypeInfo * MixedTableType() { return &MixedTableInfo; } + +// ---- the text form (docs/SPEC-TABLES.md §16) ---- + +// Placement in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// SaveTable.cpp; link it to use them. +bool PlacementFromJson( Placement & value, const char * text, int64_t bytes, TableReport * report ); +int64_t PlacementToJsonMeasure( const Placement & value ); +int64_t PlacementToJson( const Placement & value, char * buffer, int64_t capacity ); + +// LogEntry in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// SaveTable.cpp; link it to use them. +bool LogEntryFromJson( LogEntry & value, const char * text, int64_t bytes, TableReport * report ); +int64_t LogEntryToJsonMeasure( const LogEntry & value ); +int64_t LogEntryToJson( const LogEntry & value, char * buffer, int64_t capacity ); + +// Save in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in SaveTable.cpp; link it to use them. +bool SaveFromJson( SaveBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t SaveToJsonMeasure( const Save * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t SaveToJson( const Save * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Point in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// SaveTable.cpp; link it to use them. +bool PointFromJson( Point & value, const char * text, int64_t bytes, TableReport * report ); +int64_t PointToJsonMeasure( const Point & value ); +int64_t PointToJson( const Point & value, char * buffer, int64_t capacity ); + +// Mixed in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in SaveTable.cpp; link it to use them. +bool MixedFromJson( MixedBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t MixedToJsonMeasure( const Mixed * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t MixedToJson( const Mixed * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/SharedTable.cpp b/testdata/golden/tables/lists/SharedTable.cpp new file mode 100644 index 000000000..5cba2ef74 --- /dev/null +++ b/testdata/golden/tables/lists/SharedTable.cpp @@ -0,0 +1,3134 @@ +// Code generated by the schema compiler from Shared.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — the TABLE wire's text form (docs/SPEC-TABLES.md §16). +// Compile this file to use FromJson / ToJson; a project that +// never reads or writes a text does not compile it and pays nothing. + +#include "SharedTable.h" + +#include // the text form: number formatting +#include // the text form: exact number conversion +#include // the text form: the runtime's decimal point + +// The guard is not vestigial. Several listdemo Table.cpp files may be +// concatenated into ONE translation unit — a unity build — and without it +// each would redefine the walk. It is also why the walk's functions may be +// weak (vague linkage) across separate objects: ODR requires their +// definitions to be token-identical, and the generic-walk gate is what +// proves that, byte for byte, across every generated .cpp. +#ifndef LISTDEMO_SCHEMA_TABLE_JSON +#define LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +// ---- the pointer adapters (docs/SPEC-TABLES.md §16.7) ---- +// +// The walk below is ONE walk, byte-identical in every generated .cpp, and a +// pointer is the one kind it cannot walk alone: reading one needs the +// builder's arena and writing one needs a region's deref, and neither exists +// in a unit that declares no pointer. So the walk calls these three and does +// not define them. A unit with no pointer defines them as stubs no field ever +// reaches; a pointered unit defines them in the graph half that follows the +// walk. + +struct TableJsonIn; +struct TableJsonOut; + +// a pointer field's object, or the `&node` reference standing in for it, into +// the slot; the cursor is on the opening brace +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// the node a pointer slot names, in place — or as `&node` when it is shared +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// the FIRST key of an object the walk is skipping begins with `&`: the cursor is +// on its value. A dropped definition still takes its label (§16.7); a fixed reader +// skips the value whole, as it skips everything else it does not place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); + +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- +// +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map +inline bool TableJsonIsMap( const TableFieldInfo * f ); +// the map as a plain JSON object keyed by the KEY, in ASCENDING key order +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that object back into the slot, in whatever order the text gives it +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); + +// ---- json walk: begin ---- +// +// The TEXT form (docs/SPEC-TABLES.md §16): one table, one text, one walk over the +// reflection descriptors (§8). Reading fills ONE caller-owned instance and +// allocates nothing beyond it; writing targets a caller buffer with the +// wire's measure/write symmetry. Everything AROUND this — which file goes +// with which instance, what key an instance is filed under, how instances +// link into a root table's collections — is a packer's opinion and stays +// with the tool that holds it. +// +// The dialect: trailing commas are accepted on read (the authoring files +// this exists for carry them) and never written; comments are not JSON and +// are refused; unknown keys are skipped and counted; a duplicate key is +// last-wins and counted; a key present with the wrong JSON type is skipped +// and counted, never coerced. + +static const int32_t kTableJsonMaxDepth = 128; + +// A key longer than this cannot name a field, so it is skipped as unknown. +static const int32_t kTableJsonMaxKey = 256; + +// The longest numeric token the walk will convert. Anything longer is a +// value no field can hold and counts as a kind mismatch. +static const int32_t kTableJsonMaxNumber = 512; + +// The decimal point the C runtime is CURRENTLY using. Number conversion is +// the one locale-sensitive corner of the grammar — JSON's point is always +// '.', the runtime's is whatever the program set — so every number crosses +// this one character on the way out and on the way back in. Nothing else in +// the walk consults the locale. +inline char TableJsonDecimalPoint() +{ + const struct lconv * conv = localeconv(); + if ( conv != NULL && conv->decimal_point != NULL && conv->decimal_point[0] != 0 ) + { + return conv->decimal_point[0]; + } + return '.'; +} + +// ---- storage access: the descriptors give an offset and a width, and the +// ---- storage is the HOST's, so every load and store goes through a width +// ---- switch rather than a memcpy into the low bytes of a wider word + +// finite: not a NaN, not an infinity. Written without — the walk's +// runtime surface stays the handful of functions it already names. +// A vocabulary entry the descriptor could not spell. The generated name +// functions answer "???" for a value outside the declared set, and that is +// not a name — writing it would put a spelling in the text that the reader +// then counts as unknown, turning a refusal into a silent loss. +inline bool TableJsonNamed( const char * name ) +{ + return name != NULL && strcmp( name, "???" ) != 0; +} + +inline bool TableJsonFinite( double v ) +{ + return v == v && v <= 1.7976931348623157e308 && v >= -1.7976931348623157e308; +} + +inline uint64_t TableJsonGetRaw( const void * storage, uint32_t width ) +{ + switch ( width ) + { + case 1: { uint8_t v = 0; memcpy( &v, storage, 1 ); return v; } + case 2: { uint16_t v = 0; memcpy( &v, storage, 2 ); return v; } + case 4: { uint32_t v = 0; memcpy( &v, storage, 4 ); return v; } + case 8: { uint64_t v = 0; memcpy( &v, storage, 8 ); return v; } + } + return 0; +} + +inline void TableJsonSetRaw( void * storage, uint32_t width, uint64_t value ) +{ + switch ( width ) + { + case 1: { uint8_t v = (uint8_t) value; memcpy( storage, &v, 1 ); break; } + case 2: { uint16_t v = (uint16_t) value; memcpy( storage, &v, 2 ); break; } + case 4: { uint32_t v = (uint32_t) value; memcpy( storage, &v, 4 ); break; } + case 8: { uint64_t v = value; memcpy( storage, &v, 8 ); break; } + } +} + +inline int64_t TableJsonGetSigned( const void * storage, uint32_t width ) +{ + uint64_t raw = TableJsonGetRaw( storage, width ); + if ( width < 8 ) + { + uint64_t sign = uint64_t( 1 ) << ( width * 8 - 1 ); + if ( ( raw & sign ) != 0 ) + { + raw |= ~( ( sign << 1 ) - 1 ); + } + } + return (int64_t) raw; +} + +// ---- the WIDE kinds (docs/SPEC-TABLES.md §3, §16.2) ---- +// +// The 128-bit integers and the fixed-point family convert EXACTLY, over two +// 64-bit lanes: a 128-bit integer is a decimal integer, a fixed value a +// decimal in WHOLE UNITS (1.0, -0.25, 3.0000152587890625) and nothing +// on either path passes through a double. Nothing here needs a 128-bit type +// either, which is what keeps this walk one text for every unit. +struct TableJsonWide +{ + uint64_t lo; + uint64_t hi; +}; + +inline bool TableJsonKindWide( uint8_t kind ) { return kind >= 18 && kind <= 29; } +inline bool TableJsonKindWideSigned( uint8_t kind ) { return kind == 18 || ( kind >= 20 && kind <= 24 ); } +inline bool TableJsonKindFixed( uint8_t kind ) { return kind >= 20 && kind <= 29; } + +inline bool TableJsonWideZero( TableJsonWide v ) { return v.lo == 0 && v.hi == 0; } +inline bool TableJsonWideNegative( TableJsonWide v ) { return ( v.hi >> 63 ) != 0; } + +inline int TableJsonWideCompare( TableJsonWide a, TableJsonWide b, bool is_signed ) +{ + if ( is_signed && TableJsonWideNegative( a ) != TableJsonWideNegative( b ) ) { return TableJsonWideNegative( a ) ? -1 : 1; } + if ( a.hi != b.hi ) { return a.hi < b.hi ? -1 : 1; } + if ( a.lo != b.lo ) { return a.lo < b.lo ? -1 : 1; } + return 0; +} + +inline TableJsonWide TableJsonWideShl( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.hi = v.lo << ( n - 64 ); return r; } + r.hi = ( v.hi << n ) | ( v.lo >> ( 64 - n ) ); + r.lo = v.lo << n; + return r; +} + +inline TableJsonWide TableJsonWideShr( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.lo = v.hi >> ( n - 64 ); return r; } + r.lo = ( v.lo >> n ) | ( v.hi << ( 64 - n ) ); + r.hi = v.hi >> n; + return r; +} + +inline TableJsonWide TableJsonWideNeg( TableJsonWide v ) +{ + TableJsonWide r; + r.lo = ~v.lo + 1; + r.hi = ~v.hi + ( r.lo == 0 ? 1 : 0 ); + return r; +} + +// v = v * m + a; the return is the carry out of 128 bits +inline uint32_t TableJsonWideMulAdd( TableJsonWide * v, uint32_t m, uint32_t a ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t carry = a; + for ( int i = 0; i < 4; i++ ) + { + uint64_t p = limb[i] * m + carry; + limb[i] = p & 0xffffffffull; + carry = p >> 32; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) carry; +} + +// v = v / d; the return is the remainder +inline uint32_t TableJsonWideDiv( TableJsonWide * v, uint32_t d ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t rem = 0; + for ( int i = 3; i >= 0; i-- ) + { + uint64_t cur = ( rem << 32 ) | limb[i]; + limb[i] = cur / d; + rem = cur % d; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) rem; +} + +// The storage of a wide kind, as lanes. A sixteen-byte storage is serialize's +// pair — native __int128 in the host's byte order, or the emulated struct with +// its low lane first — so the lanes are read in the host's order; a narrower +// storage is one lane, sign-extended for a signed kind. +inline TableJsonWide TableJsonWideLoad( const void * storage, uint32_t width, bool is_signed ) +{ + TableJsonWide v = { 0, 0 }; + if ( width == 16 ) + { + uint64_t half[2]; + memcpy( half, storage, 16 ); + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + v.lo = little ? half[0] : half[1]; + v.hi = little ? half[1] : half[0]; + return v; + } + v.lo = is_signed ? (uint64_t) TableJsonGetSigned( storage, width ) : TableJsonGetRaw( storage, width ); + v.hi = ( is_signed && ( v.lo >> 63 ) != 0 ) ? ~uint64_t( 0 ) : 0; + return v; +} + +inline void TableJsonWideStore( void * storage, uint32_t width, TableJsonWide v ) +{ + if ( width == 16 ) + { + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + uint64_t half[2]; + half[0] = little ? v.lo : v.hi; + half[1] = little ? v.hi : v.lo; + memcpy( storage, half, 16 ); + return; + } + TableJsonSetRaw( storage, width, v.lo ); +} + +// a counted field's companion: a string's length, a bytes' length, a counted +// array's count. Bounded by the declared extent on the way out, so a storage +// invariant a caller broke cannot walk off the end of the array. +inline int32_t TableJsonCount( const void * base, const TableFieldInfo * f ) +{ + if ( !f->counted ) + { + return f->array_bound; + } + int32_t count = 0; + memcpy( &count, (const uint8_t *) base + f->count_offset, sizeof( count ) ); + if ( count < 0 ) { count = 0; } + if ( count > f->array_bound ) { count = f->array_bound; } + return count; +} + +inline void TableJsonSetCount( void * base, const TableFieldInfo * f, int32_t count ) +{ + if ( f->counted ) + { + memcpy( (uint8_t *) base + f->count_offset, &count, sizeof( count ) ); + } +} + +// ---- what a field's kind expects to see in the text ---- +// +// One classifier, consulted by both directions, so a reader and a writer can +// never disagree about a kind's JSON form. 'o' object, 'a' array, 's' +// string, 'n' number, 'b' boolean. +// +// A vocabulary field is spelled by NAME: an enum is one name, a flags mask +// is the array of the names of its set bits. The two are told apart by the +// id column — an enum variant rides under a wire id, a flags BIT never does +// (docs/SPEC-TABLES.md §4), so a name function with no id function is flags. +// +// bytes(N) is the one kind whose element kind does not decide its form: it +// shares u8 with a plain array of u8, and rides as base64. The schema type +// name settles it, and "bytes" is a keyword no declaration can claim. +inline bool TableJsonIsBytes( const TableFieldInfo * f ) +{ + return f->is_array && f->kind == 6 && strcmp( f->type_name, "bytes" ) == 0; +} + +// An ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): its JSON form is an OBJECT +// keyed by variant name, not a positional array, because that is what the +// storage is — one slot per variant, addressed by the variant. +inline bool TableJsonIsKeyed( const TableFieldInfo * f ) +{ + return f->key_name != NULL; +} + +// THE KEY A STORAGE SLOT HOLDS (§2.4, §8): the storage shifts left, so slot i +// holds the key i + 1 and nothing is stored for None. This is the ONE place +// the walker spells the shift. +inline uint64_t TableJsonKeyedSlotKey( int64_t slot ) +{ + return (uint64_t) ( slot + 1 ); +} + +// A slot whose key names a variant of the keying enum. Every slot in +// [0, array_bound) does, unless the enum carries max-headroom variants outside +// a table closure, where a reserved value names nothing and its key id is 0 — +// the reserved id no declared name can fold to (§5). +inline bool TableJsonKeyedSlotValid( const TableFieldInfo * f, int64_t slot ) +{ + return f->key_id( TableJsonKeyedSlotKey( slot ) ) != 0; +} + +inline bool TableJsonIsFlags( const TableFieldInfo * f ) +{ + return f->enum_name != NULL && f->variant_id == NULL; +} + +inline bool TableJsonIsEnum( const TableFieldInfo * f ) +{ + return f->variant_id != NULL && f->arms == NULL; +} + +inline char TableJsonShape( const TableFieldInfo * f ) +{ + if ( TableJsonIsMap( f ) ) return 'o'; // a MAP: an object keyed by the KEY (§2.8) + if ( f->kind == 12 ) return 's'; // string + if ( TableJsonIsBytes( f ) ) return 's'; // bytes: base64 + if ( TableJsonIsKeyed( f ) ) return 'o'; // an object keyed by variant NAME + if ( f->is_array ) return 'a'; + if ( f->arms != NULL ) return 'o'; // union: an object with ONE key + if ( f->kind == 13 ) return 'o'; // nested table or type + if ( f->kind == 17 ) return f->table != NULL ? 'o' : 's'; // a pointer: the pointee's object in place, or null (§16.7); a byte buffer's string (§2.5) + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// the ELEMENT shape of an array field — the same classifier one level down +inline char TableJsonElementShape( const TableFieldInfo * f ) +{ + if ( f->arms != NULL ) return 'o'; // an element of an array of unions: one key, the arm (§2.6) + if ( f->kind == 13 ) return 'o'; + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// A guarded group rides only when its guard reads true — the wire's own +// elision (§4), carried into the text so a text and a wire written from one +// instance say the same thing. The guard is spelled as its branch condition +// over bool fields of the SAME type ("at_rest", "!at_rest", +// "active && has_target"), so evaluating it is a walk of the same +// descriptor. Nothing is inferred in the other direction: reading places +// every key it can name, and the guard is a plain bool key (§16.2). +inline bool TableJsonGuardHolds( const void * base, const TableTypeInfo * info, const char * guard ) +{ + const char * p = guard; + for ( ;; ) + { + while ( *p == ' ' || *p == '&' ) { p++; } + if ( *p == 0 ) { return true; } + bool want = true; + if ( *p == '!' ) { want = false; p++; } + const char * start = p; + while ( *p != 0 && *p != ' ' && *p != '&' ) { p++; } + size_t length = (size_t) ( p - start ); + bool value = false; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( strlen( f->name ) == length && strncmp( f->name, start, length ) == 0 ) + { + value = TableJsonGetRaw( (const uint8_t *) base + f->offset, f->elem_size ) != 0; + break; + } + } + if ( value != want ) { return false; } + } +} + +// ---- writing ---- + +// The writer sink MEASURES when the buffer is NULL and WRITES when it is +// not, over one code path — so measure and write agree byte for byte, the +// wire's invariant (§9) carried across. +struct TableJsonOut +{ + char * buffer; + int64_t capacity; + int64_t offset; + bool overflow; + void * graph; // the pointered write's identity map (§16.7); NULL for a fixed table + + void raw( const char * data, int64_t count ) + { + if ( buffer != NULL ) + { + if ( offset + count > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) count ); + } + offset += count; + } + void put( char c ) { raw( &c, 1 ); } + void text( const char * s ) { raw( s, (int64_t) strlen( s ) ); } + void line( int32_t depth ) + { + put( '\n' ); + for ( int32_t i = 0; i < depth; i++ ) { raw( " ", 2 ); } + } +}; + +inline const char * TableJsonBase64Alphabet() +{ + return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; +} + +inline void TableJsonWriteBase64( TableJsonOut & out, const uint8_t * data, int32_t length ) +{ + const char * alphabet = TableJsonBase64Alphabet(); + out.put( '"' ); + int32_t i = 0; + for ( ; i + 3 <= length; i += 3 ) + { + uint32_t triple = ( uint32_t( data[i] ) << 16 ) | ( uint32_t( data[i+1] ) << 8 ) | uint32_t( data[i+2] ); + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], + alphabet[ ( triple >> 6 ) & 0x3f ], alphabet[ triple & 0x3f ] }; + out.raw( quad, 4 ); + } + if ( i < length ) + { + int32_t left = length - i; + uint32_t triple = uint32_t( data[i] ) << 16; + if ( left == 2 ) { triple |= uint32_t( data[i+1] ) << 8; } + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], '=', '=' }; + if ( left == 2 ) { quad[2] = alphabet[ ( triple >> 6 ) & 0x3f ]; } + out.raw( quad, 4 ); + } + out.put( '"' ); +} + +// One UTF-8 sequence at s, or -1 when the bytes there are not one. Rejects +// the lot: a stray continuation, an overlong form, a surrogate half, and +// anything past U+10FFFF. +inline int32_t TableJsonUtf8( const char * s, int32_t remaining, int32_t * width ) +{ + unsigned char lead = (unsigned char) s[0]; + int32_t want = 0; + int32_t code = 0; + if ( lead < 0x80 ) { *width = 1; return lead; } + else if ( lead >= 0xc2 && lead <= 0xdf ) { want = 2; code = lead & 0x1f; } + else if ( lead >= 0xe0 && lead <= 0xef ) { want = 3; code = lead & 0x0f; } + else if ( lead >= 0xf0 && lead <= 0xf4 ) { want = 4; code = lead & 0x07; } + else { return -1; } + if ( remaining < want ) { return -1; } + for ( int32_t i = 1; i < want; i++ ) + { + unsigned char next = (unsigned char) s[i]; + if ( ( next & 0xc0 ) != 0x80 ) { return -1; } + code = ( code << 6 ) | ( next & 0x3f ); + } + if ( want == 3 && code < 0x800 ) { return -1; } // overlong + if ( want == 4 && code < 0x10000 ) { return -1; } // overlong + if ( code >= 0xd800 && code <= 0xdfff ) { return -1; } // a surrogate half + if ( code > 0x10ffff ) { return -1; } + *width = want; + return code; +} + +// A JSON text MUST be valid UTF-8 (RFC 8259 §8.1). The read path is +// byte-transparent — the wire imposes no encoding (§3) and a string may hold +// anything — so the WRITER is where that obligation is met: a byte that is +// not part of a well-formed sequence is written as U+FFFD, one per bad byte, +// and never raw. A text this walk writes is therefore readable by any +// conforming parser, which a raw byte would not be. The cost is stated +// plainly: for a string holding invalid UTF-8, the round trip is NOT +// byte-identical, because the alternative is emitting a text that is not +// JSON. +inline void TableJsonWriteString( TableJsonOut & out, const char * s, int32_t length ) +{ + static const char hex[] = "0123456789abcdef"; + out.put( '"' ); + for ( int32_t i = 0; i < length; i++ ) + { + unsigned char c = (unsigned char) s[i]; + switch ( c ) + { + case '"': out.raw( "\\\"", 2 ); break; + case '\\': out.raw( "\\\\", 2 ); break; + case '\b': out.raw( "\\b", 2 ); break; + case '\f': out.raw( "\\f", 2 ); break; + case '\n': out.raw( "\\n", 2 ); break; + case '\r': out.raw( "\\r", 2 ); break; + case '\t': out.raw( "\\t", 2 ); break; + default: + if ( c < 0x20 ) + { + char escape[6] = { '\\', 'u', '0', '0', hex[ c >> 4 ], hex[ c & 0xf ] }; + out.raw( escape, 6 ); + } + else if ( c < 0x80 ) + { + out.put( (char) c ); + } + else + { + int32_t width = 0; + if ( TableJsonUtf8( s + i, length - i, &width ) < 0 ) + { + out.raw( "\xef\xbf\xbd", 3 ); // U+FFFD, one per bad byte + } + else + { + out.raw( s + i, width ); + i += width - 1; + } + } + break; + } + } + out.put( '"' ); +} + +inline void TableJsonWriteUnsigned( TableJsonOut & out, uint64_t value ) +{ + char digits[24]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) ( value % 10 ) ); + value /= 10; + } while ( value != 0 ); + char text[24]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); +} + +inline void TableJsonWriteSigned( TableJsonOut & out, int64_t value ) +{ + if ( value < 0 ) + { + out.put( '-' ); + TableJsonWriteUnsigned( out, uint64_t( 0 ) - (uint64_t) value ); + return; + } + TableJsonWriteUnsigned( out, (uint64_t) value ); +} + +// A wide kind writes its raw storage as §16.2's text: a 128-bit integer as a +// decimal integer; a fixed value in WHOLE UNITS as the shortest exact decimal +// with at least one fractional digit (1.0, -0.25), the spelling the schema text +// gives a fixed default. The fraction terminates because a dyadic fraction has +// a finite decimal expansion — at most F digits. +inline void TableJsonWriteWide( TableJsonOut & out, const void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + TableJsonWide v = TableJsonWideLoad( storage, f->elem_size, is_signed ); + if ( is_signed && TableJsonWideNegative( v ) ) + { + out.put( '-' ); + v = TableJsonWideNeg( v ); + } + int frac = f->frac_bits; + TableJsonWide whole = TableJsonWideShr( v, frac ); + char digits[40]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) TableJsonWideDiv( &whole, 10 ) ); + } while ( !TableJsonWideZero( whole ) ); + char text[40]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); + if ( !TableJsonKindFixed( f->kind ) ) { return; } + out.put( '.' ); + // the fraction bits alone: v with everything at and above bit F cleared + TableJsonWide fraction = v; + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + if ( frac == 0 ) { fraction.lo = 0; } + if ( TableJsonWideZero( fraction ) ) + { + out.put( '0' ); + return; + } + while ( !TableJsonWideZero( fraction ) ) + { + // ×10: the digit is what lands at and above bit F, including the + // carry out of 128 bits when F leaves no room for it below + uint32_t carry = TableJsonWideMulAdd( &fraction, 10, 0 ); + uint64_t digit = TableJsonWideShr( fraction, frac ).lo; + if ( frac > 64 ) { digit |= uint64_t( carry ) << ( 128 - frac ); } + out.put( (char) ( '0' + (int) digit ) ); + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + } +} + +// A float writes at the SHORTEST precision that reads back as the same value +// at the field's own width, so a round trip is exact and a text stays +// readable. Non-finite values have no JSON spelling at all, and the writer +// REFUSES rather than losing one silently — the same rule measure and save +// already apply to an enum value no variant names (§5). +inline bool TableJsonWriteFloat( TableJsonOut & out, double value, bool single ) +{ + if ( !TableJsonFinite( value ) ) { return false; } + char text[64]; + int low = single ? 6 : 15; + int high = single ? 9 : 17; + int length = 0; + for ( int digits = low; ; digits++ ) + { + length = snprintf( text, sizeof( text ), "%.*g", digits, value ); + if ( length <= 0 || length >= (int) sizeof( text ) ) { return false; } + if ( digits >= high ) { break; } + // the round-trip check runs BEFORE the decimal point is normalised: + // the token still carries whatever point snprintf just produced + if ( single ) + { + if ( (double) strtof( text, NULL ) == value ) { break; } + } + else + { + if ( strtod( text, NULL ) == value ) { break; } + } + } + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int i = 0; i < length; i++ ) + { + if ( text[i] == point ) { text[i] = '.'; } + } + } + out.raw( text, length ); + return true; +} + +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration writes through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ); + +// one scalar, at one storage address: a nested object, a union, a +// vocabulary, or a number +inline bool TableJsonWriteScalar( TableJsonOut & out, const void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; None is {} + const TableUnionInfo * arms = f->arms(); + uint64_t tag = TableJsonGetRaw( (const uint8_t *) storage + arms->tag_offset, arms->tag_size ); + if ( tag == 0 ) + { + out.raw( "{}", 2 ); + return true; + } + if ( (int64_t) tag > f->enum_max ) + { + return false; // a tag no arm names, exactly as measure refuses it + } + const char * arm = f->enum_name( tag ); + // and refuse on the NAME, not merely on the bound: §16.2 says a value + // no variant NAMES is refused, so the check is the name. Writing + // whatever came back would emit "???", a spelling the reader counts + // as unknown — a silent round-trip loss in place of a refusal. + if ( !TableJsonNamed( arm ) ) { return false; } + out.put( '{' ); + out.line( depth + 1 ); + TableJsonWriteString( out, arm, (int32_t) strlen( arm ) ); + out.raw( ": ", 2 ); + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2): an arm that names + // no declaration carries the FIELD descriptor a field of its type + // would carry, offsets taken inside the union storage (§2.6), so the + // value walks through the field writer one key down. + if ( arms->arms[tag].field != NULL ) + { + if ( !TableJsonWriteField( out, storage, arms->arms[tag].field, depth + 1 ) ) + { + return false; + } + } + else if ( arms->arms[tag].table == NULL ) + { + out.raw( "null", 4 ); // a payload-free arm: the name selects it (§2.6) + } + else if ( !TableJsonWriteValue( out, (const uint8_t *) storage + arms->arms[tag].offset, arms->arms[tag].table, depth + 1 ) ) + { + return false; + } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->kind == 13 ) + { + return TableJsonWriteValue( out, storage, f->table, depth ); + } + if ( TableJsonIsEnum( f ) ) + { + uint64_t value = TableJsonGetRaw( storage, f->elem_size ); + // a value no variant names has no text spelling, exactly as it has no + // wire identity: the writer REFUSES rather than writing None over it, + // the rule measure and save already apply (docs/SPEC-TABLES.md §5) + if ( (int64_t) value > f->enum_max ) { return false; } + if ( value != 0 && f->variant_id( value ) == 0 ) { return false; } + const char * name = f->enum_name( value ); + if ( !TableJsonNamed( name ) ) { return false; } + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + return true; + } + if ( TableJsonIsFlags( f ) ) + { + uint64_t bits = TableJsonGetRaw( storage, f->elem_size ); + if ( bits == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + bool first = true; + for ( int64_t bit = 0; bit < 64; bit++ ) + { + if ( ( bits & ( uint64_t( 1 ) << bit ) ) == 0 ) { continue; } + if ( bit > f->enum_max ) + { + return false; // a bit no variant names has no text spelling + } + const char * name = f->enum_name( (uint64_t) bit ); + if ( !TableJsonNamed( name ) ) { return false; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + } + out.line( depth ); + out.put( ']' ); + return true; + } + switch ( f->kind ) + { + case 1: + out.text( TableJsonGetRaw( storage, f->elem_size ) != 0 ? "true" : "false" ); + return true; + case 10: + { + float v = 0.0f; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, (double) v, true ); + } + case 11: + { + double v = 0.0; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, v, false ); + } + case 2: case 3: case 4: case 5: + TableJsonWriteSigned( out, TableJsonGetSigned( storage, f->elem_size ) ); + return true; + default: + if ( TableJsonKindWide( f->kind ) ) + { + TableJsonWriteWide( out, storage, f ); + return true; + } + TableJsonWriteUnsigned( out, TableJsonGetRaw( storage, f->elem_size ) ); + return true; + } +} + +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ) +{ + const uint8_t * storage = (const uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonWriteMap( out, (const void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } + if ( f->kind == 17 && !f->is_array ) + { + return TableJsonWritePointer( out, storage, f, depth ); + } + if ( f->kind == 17 ) + { + // an ARRAY OF POINTERS (§2.1): the pointer row per element — the + // pointee's object in place, null, or `&node` for a shared one (§16.7) + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWritePointer( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; + } + if ( f->kind == 12 ) + { + TableJsonWriteString( out, (const char *) storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + TableJsonWriteBase64( out, storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + // one entry per SLOT, keyed by the variant that owns it, so inserting + // a variant next season moves nothing in the text either. Slot i holds + // the key i + 1: nothing is stored for None, so nothing is written for it. + out.put( '{' ); + bool first = true; + for ( int64_t slot = 0; slot < f->array_bound; slot++ ) + { + if ( !TableJsonKeyedSlotValid( f, slot ) ) { continue; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + const char * key = f->key_name( TableJsonKeyedSlotKey( slot ) ); + TableJsonWriteString( out, key, (int32_t) strlen( key ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteScalar( out, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + if ( first ) { out.raw( "}", 1 ); return true; } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->is_array ) + { + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWriteScalar( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + out.line( depth ); + out.put( ']' ); + return true; + } + return TableJsonWriteScalar( out, storage, f, depth ); +} + +// One instance's fields, in DECLARATION ORDER, defaults included — a text is +// for people and tools, and a text that elides is a text a reader has to know +// the schema to complete. `any` says whether the object is already open on +// entry — a shared node's `&node` opens it before the fields (§16.7) — and +// whether it is open on return. +inline bool TableJsonWriteFields( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth, bool & any ) +{ + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( f->guard[0] != 0 && !TableJsonGuardHolds( base, info, f->guard ) ) { continue; } + // an ABSENT optional writes no key: presence of the key IS the + // presence (§16.2), so an absent field is an absent key and nothing + // else would read back as absent + if ( f->optional && + TableJsonGetRaw( (const uint8_t *) base + f->present_offset, 1 ) == 0 ) + { + continue; + } + if ( !any ) { out.put( '{' ); } + else { out.put( ',' ); } + any = true; + out.line( depth + 1 ); + TableJsonWriteString( out, f->json, (int32_t) strlen( f->json ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, base, f, depth + 1 ) ) { return false; } + } + return true; +} + +// One instance as one object. The writer carries the reader's depth cap +// (§16.2): a pointer chain nests as deep as it is long (§16.7), and a text the +// writer produced past the cap would be a text the reader refuses. +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { return false; } + bool any = false; + if ( !TableJsonWriteFields( out, base, info, depth, any ) ) { return false; } + if ( !any ) + { + out.raw( "{}", 2 ); + return true; + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- reading ---- + +struct TableJsonIn +{ + const char * text; + int64_t size; + int64_t pos; + TableReport * report; + bool bad; // the text is not JSON: the walk stops and keeps what it placed + void * graph; // the pointered read's builder and label map (§16.7); NULL for a fixed table +}; + +inline void TableJsonSpace( TableJsonIn & in ) +{ + while ( in.pos < in.size ) + { + char c = in.text[in.pos]; + if ( c == ' ' || c == '\t' || c == '\n' || c == '\r' ) { in.pos++; continue; } + // comments are not JSON, and a walk that guessed at one would be + // reading a dialect nobody wrote down + if ( c == '/' ) { in.bad = true; } + return; + } +} + +inline char TableJsonPeek( TableJsonIn & in ) +{ + TableJsonSpace( in ); + return in.pos < in.size ? in.text[in.pos] : 0; +} + +// the shape of the value sitting at the cursor, without consuming it +inline char TableJsonValueShape( TableJsonIn & in ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return 'o'; + case '[': return 'a'; + case '"': return 's'; + case 't': case 'f': return 'b'; + case 'n': return 'z'; + case 0: return 0; + default: return 'n'; + } +} + +inline bool TableJsonLiteral( TableJsonIn & in, const char * word ) +{ + int64_t length = (int64_t) strlen( word ); + if ( in.pos + length > in.size || memcmp( in.text + in.pos, word, (size_t) length ) != 0 ) + { + in.bad = true; + return false; + } + in.pos += length; + return true; +} + +// one \uXXXX escape body; -1 when the four hex digits are not there +inline int TableJsonHex4( TableJsonIn & in ) +{ + if ( in.pos + 4 > in.size ) { return -1; } + int value = 0; + for ( int i = 0; i < 4; i++ ) + { + char c = in.text[in.pos + i]; + int digit; + if ( c >= '0' && c <= '9' ) { digit = c - '0'; } + else if ( c >= 'a' && c <= 'f' ) { digit = c - 'a' + 10; } + else if ( c >= 'A' && c <= 'F' ) { digit = c - 'A' + 10; } + else { return -1; } + value = ( value << 4 ) | digit; + } + in.pos += 4; + return value; +} + +inline int32_t TableJsonEncodeUtf8( uint32_t code, char * unit ) +{ + if ( code < 0x80 ) { unit[0] = (char) code; return 1; } + if ( code < 0x800 ) + { + unit[0] = (char) ( 0xc0 | ( code >> 6 ) ); + unit[1] = (char) ( 0x80 | ( code & 0x3f ) ); + return 2; + } + if ( code < 0x10000 ) + { + unit[0] = (char) ( 0xe0 | ( code >> 12 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( code & 0x3f ) ); + return 3; + } + unit[0] = (char) ( 0xf0 | ( code >> 18 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 12 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[3] = (char) ( 0x80 | ( code & 0x3f ) ); + return 4; +} + +// Scan one JSON string into a caller buffer. Bytes are appended ONE CODE +// POINT AT A TIME — an escape's encoding, or a UTF-8 sequence read whole — +// so a string longer than the field is clamped AT A CODE POINT BOUNDARY and +// never cut through a multi-byte character. Clamping is counted, never +// fatal, exactly as it is on the wire (§4). A NULL destination scans past a +// string without keeping it. +inline bool TableJsonScanString( TableJsonIn & in, char * out, int32_t capacity, int32_t * length ) +{ + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + int32_t placed = 0; + bool clamped = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos]; + if ( c == '"' ) { in.pos++; break; } + char unit[4]; + int32_t unit_length = 0; + if ( c == '\\' ) + { + in.pos++; + if ( in.pos >= in.size ) { in.bad = true; return false; } + char escape = in.text[in.pos++]; + switch ( escape ) + { + case '"': unit[0] = '"'; unit_length = 1; break; + case '\\': unit[0] = '\\'; unit_length = 1; break; + case '/': unit[0] = '/'; unit_length = 1; break; + case 'b': unit[0] = '\b'; unit_length = 1; break; + case 'f': unit[0] = '\f'; unit_length = 1; break; + case 'n': unit[0] = '\n'; unit_length = 1; break; + case 'r': unit[0] = '\r'; unit_length = 1; break; + case 't': unit[0] = '\t'; unit_length = 1; break; + case 'u': + { + int high = TableJsonHex4( in ); + if ( high < 0 ) { in.bad = true; return false; } + uint32_t code = (uint32_t) high; + if ( high >= 0xd800 && high <= 0xdbff && in.pos + 2 <= in.size && + in.text[in.pos] == '\\' && in.text[in.pos + 1] == 'u' ) + { + int64_t mark = in.pos; + in.pos += 2; + int low = TableJsonHex4( in ); + if ( low >= 0xdc00 && low <= 0xdfff ) + { + code = 0x10000 + ( ( (uint32_t) high - 0xd800 ) << 10 ) + ( (uint32_t) low - 0xdc00 ); + } + else + { + in.pos = mark; // a lone lead surrogate rides as itself + } + } + // a surrogate half that never found its partner has no + // UTF-8 encoding: encoding it anyway would manufacture + // CESU-8 — invalid UTF-8 — out of input that was valid + // JSON, so it reads as the replacement character + if ( code >= 0xd800 && code <= 0xdfff ) { code = 0xfffd; } + unit_length = TableJsonEncodeUtf8( code, unit ); + break; + } + default: in.bad = true; return false; + } + } + else if ( (unsigned char) c < 0x20 ) + { + in.bad = true; // a raw control character is not a JSON string body + return false; + } + else + { + // a UTF-8 sequence read WHOLE, so the clamp below can only land + // between code points. Only bytes that ACTUALLY look like + // continuations are taken: the wire imposes no encoding (§3), so + // a string may legitimately hold a stray lead byte, and one at + // the end of a text must not swallow the closing quote. + unsigned char lead = (unsigned char) c; + int32_t want = 1; + if ( ( lead & 0xe0 ) == 0xc0 ) { want = 2; } + else if ( ( lead & 0xf0 ) == 0xe0 ) { want = 3; } + else if ( ( lead & 0xf8 ) == 0xf0 ) { want = 4; } + unit[0] = c; + in.pos++; + unit_length = 1; + while ( unit_length < want && in.pos < in.size && + ( (unsigned char) in.text[in.pos] & 0xc0 ) == 0x80 ) + { + unit[unit_length++] = in.text[in.pos++]; + } + } + if ( out == NULL ) + { + placed += unit_length; // measured and not kept: a byte buffer's read sizes its node this way (§2.5) + } + else if ( placed + unit_length <= capacity ) + { + memcpy( out + placed, unit, (size_t) unit_length ); + placed += unit_length; + } + else + { + clamped = true; + } + } + if ( clamped ) { in.report->clamped++; } + if ( length != NULL ) { *length = placed; } + return true; +} + +// the numeric token at the cursor, copied out whole; false = not a number +// Scan one number, to JSON's OWN grammar (RFC 8259 §6) and not to a run of +// number-ish characters: +// +// number = [ "-" ] int [ frac ] [ exp ] +// int = "0" / ( digit1-9 *digit ) +// frac = "." 1*digit +// exp = ( "e" / "E" ) [ "-" / "+" ] 1*digit +// +// Scanning the production is what makes a typo in an authoring file a +// DIAGNOSTIC rather than a value: "1-2" scans as 1 and leaves "-2" where the +// object expects a comma, so the text is malformed — which is what §16.2 +// already promises. A permissive scan would hand "1-2" to a digit loop and +// report a clamp, and a config pipeline would never hear about it. Leading +// "+", leading zeros, ".5" and "3." are not JSON either. +inline bool TableJsonWalkNumber( TableJsonIn & in, bool * integral ) +{ + TableJsonSpace( in ); + bool whole = true; + if ( in.pos < in.size && in.text[in.pos] == '-' ) { in.pos++; } + // int: a lone zero, or a non-zero digit and any digits after it + if ( in.pos >= in.size ) { return false; } + if ( in.text[in.pos] == '0' ) + { + in.pos++; + } + else if ( in.text[in.pos] >= '1' && in.text[in.pos] <= '9' ) + { + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + } + else + { + return false; + } + // frac + if ( in.pos < in.size && in.text[in.pos] == '.' ) + { + in.pos++; + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + // exp + if ( in.pos < in.size && ( in.text[in.pos] == 'e' || in.text[in.pos] == 'E' ) ) + { + in.pos++; + if ( in.pos < in.size && ( in.text[in.pos] == '-' || in.text[in.pos] == '+' ) ) { in.pos++; } + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + *integral = whole; + return true; +} + +// the same production, with the token kept for conversion +inline bool TableJsonScanNumber( TableJsonIn & in, char * token, int32_t capacity, int32_t * length, bool * integral ) +{ + TableJsonSpace( in ); + int64_t start = in.pos; + if ( !TableJsonWalkNumber( in, integral ) ) { return false; } + int64_t count = in.pos - start; + if ( count <= 0 || count >= capacity ) { return false; } + memcpy( token, in.text + start, (size_t) count ); + token[count] = 0; + *length = (int32_t) count; + return true; +} + +// the token's exact double, through the runtime's own converter — which +// speaks the LOCALE's decimal point, so the token crosses back over that +// character on its way in +inline double TableJsonTokenDouble( const char * token, int32_t length, bool single ) +{ + char work[kTableJsonMaxNumber]; + memcpy( work, token, (size_t) length ); + work[length] = 0; + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int32_t i = 0; i < length; i++ ) + { + if ( work[i] == '.' ) { work[i] = point; } + } + } + if ( single ) { return (double) strtof( work, NULL ); } + return strtod( work, NULL ); +} + +// the token's exact integer, parsed digit by digit so no width and no +// locale can move it. Saturation is reported as a clamp, the wire's rule for +// a value outside what the reader can hold (§4). +inline int64_t TableJsonTokenInteger( const char * token, int32_t length, bool is_signed, bool * saturated ) +{ + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) + { + negative = token[i] == '-'; + i++; + } + uint64_t magnitude = 0; + bool over = false; + for ( ; i < length; i++ ) + { + uint64_t digit = (uint64_t) ( token[i] - '0' ); + if ( magnitude > ( UINT64_MAX - digit ) / 10 ) { over = true; break; } + magnitude = magnitude * 10 + digit; + } + if ( !is_signed ) + { + // -0 IS zero, and clamping it would report an event that did not + // happen; only a real negative magnitude is out of range here + if ( negative ) { *saturated = magnitude != 0; return 0; } + if ( over ) { *saturated = true; return (int64_t) UINT64_MAX; } + *saturated = false; + return (int64_t) magnitude; + } + if ( negative ) + { + if ( over || magnitude > ( uint64_t( 1 ) << 63 ) ) { *saturated = true; return INT64_MIN; } + *saturated = false; + if ( magnitude == ( uint64_t( 1 ) << 63 ) ) { return INT64_MIN; } + return -(int64_t) magnitude; + } + if ( over || magnitude > (uint64_t) INT64_MAX ) { *saturated = true; return INT64_MAX; } + *saturated = false; + return (int64_t) magnitude; +} + +// A number token into a wide kind's raw storage (docs/SPEC-TABLES.md §16.2). A +// 128-bit integer takes any token whose VALUE is integral; a fixed field any +// token whose value is EXACTLY representable in its Q I.F — a finer fraction +// is the wrong shape for the field, counted as a kind mismatch and never +// rounded, the rule SPEC.md §4.6 gives a fixed default. A magnitude past 128 +// bits saturates and counts as a clamp, as an int64 field saturates at +// INT64_MAX; the declared range clamps after it, on the RAW scale, as it does +// for every bounded scalar. +// +// The token is normalized to its digits with the decimal point after "point" +// of them. An integer part past 40 digits is above 2^128 whatever the digits +// are, and a value below 10^-40 is finer than 2^-127, the finest fraction any +// F can spell — so outside that band the answer is known without the +// arithmetic, and a token spelling 1e999999999 costs nothing to refuse. +inline bool TableJsonReadWide( TableJsonIn & in, const char * token, int32_t length, void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + int frac = f->frac_bits; + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { negative = token[i] == '-'; i++; } + const char * int_digits = token + i; + int32_t int_len = 0; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { int_len++; i++; } + const char * frac_digits = token + i; + int32_t frac_len = 0; + if ( i < length && token[i] == '.' ) + { + i++; + frac_digits = token + i; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { frac_len++; i++; } + } + int64_t exp = 0; + if ( i < length && ( token[i] == 'e' || token[i] == 'E' ) ) + { + i++; + bool exp_negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { exp_negative = token[i] == '-'; i++; } + while ( i < length && token[i] >= '0' && token[i] <= '9' ) + { + if ( exp < 100000 ) { exp = exp * 10 + ( token[i] - '0' ); } + i++; + } + if ( exp_negative ) { exp = -exp; } + } + // the digits, with the point after "point" of them; leading and trailing + // zeros stripped. digit( k ) reads the k-th of the int and frac runs. + int32_t start = 0, end = int_len + frac_len; + int64_t point = int_len + exp; + while ( start < end && ( start < int_len ? int_digits[start] : frac_digits[start - int_len] ) == '0' ) { start++; point--; } + while ( end > start && ( end - 1 < int_len ? int_digits[end - 1] : frac_digits[end - 1 - int_len] ) == '0' ) { end--; } + + TableJsonWide raw = { 0, 0 }; + bool saturated = false; + TableJsonWide signed_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) >> 1 }; + TableJsonWide signed_min = { 0, uint64_t( 1 ) << 63 }; + TableJsonWide unsigned_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) }; + if ( start == end ) + { + // zero, and -0 IS zero + } + else if ( point > 40 ) + { + saturated = true; + if ( !negative ) { raw = is_signed ? signed_max : unsigned_max; } + else if ( is_signed ) { raw = signed_min; } + } + else if ( point < -40 ) + { + in.report->kind_mismatch++; // finer than any F can spell + return true; + } + else + { + // the fraction FIRST, so an inexact value is the wrong shape whatever + // its magnitude: its digits, with the zeros a negative point puts in + // front, doubled F times; each doubling's carry is the next bit, and + // the value is exact iff nothing is left after the last one + char fd[kTableJsonMaxNumber + 48]; + int32_t fn = 0; + for ( int64_t z = point; z < 0; z++ ) { fd[fn++] = 0; } + for ( int32_t k = (int32_t) ( point > 0 ? point : 0 ) + start; k < end; k++ ) + { + fd[fn++] = (char) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ); + } + TableJsonWide fraction = { 0, 0 }; + for ( int b = 0; b < frac; b++ ) + { + int carry = 0; + for ( int32_t k = fn - 1; k >= 0; k-- ) + { + int d = fd[k] * 2 + carry; + fd[k] = (char) ( d % 10 ); + carry = d / 10; + } + fraction = TableJsonWideShl( fraction, 1 ); + fraction.lo |= (uint64_t) carry; + } + for ( int32_t k = 0; k < fn; k++ ) + { + if ( fd[k] != 0 ) + { + in.report->kind_mismatch++; + return true; + } + } + // then the whole part, saturating past 128 bits + TableJsonWide whole = { 0, 0 }; + for ( int64_t k = start; k < start + point && !saturated; k++ ) + { + uint32_t digit = k < end ? (uint32_t) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ) : 0; + if ( TableJsonWideMulAdd( &whole, 10, digit ) != 0 ) { saturated = true; } + } + if ( !saturated && frac > 0 && !TableJsonWideZero( TableJsonWideShr( whole, 128 - frac ) ) ) { saturated = true; } + if ( !saturated ) + { + raw = TableJsonWideShl( whole, frac ); + raw.lo |= fraction.lo; + raw.hi |= fraction.hi; + } + if ( is_signed ) + { + if ( !saturated && !negative && TableJsonWideNegative( raw ) ) { saturated = true; } + if ( !saturated && negative && TableJsonWideCompare( raw, signed_min, false ) > 0 ) { saturated = true; } + if ( saturated ) { raw = negative ? signed_min : signed_max; } + else if ( negative ) { raw = TableJsonWideNeg( raw ); } + } + else + { + if ( saturated ) { raw = unsigned_max; } + if ( negative && !TableJsonWideZero( raw ) ) { raw.lo = 0; raw.hi = 0; saturated = true; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->wide != NULL ) + { + TableJsonWide lo = { f->wide->lo[0], f->wide->lo[1] }; + TableJsonWide hi = { f->wide->hi[0], f->wide->hi[1] }; + if ( TableJsonWideCompare( raw, lo, is_signed ) < 0 ) { raw = lo; in.report->clamped++; } + else if ( TableJsonWideCompare( raw, hi, is_signed ) > 0 ) { raw = hi; in.report->clamped++; } + } + TableJsonWideStore( storage, f->elem_size, raw ); + return true; +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ); + +inline bool TableJsonSkipContainer( TableJsonIn & in, char close, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; // the opening bracket + bool first = true; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == close ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + if ( close == '}' ) + { + // the key is kept, because a skipped OBJECT may still be a + // pointer's: an `&node` opening it names a node the storage could + // not hold, and the numbering has to survive the drop (§16.7). + // Anywhere but first, the prefix is the reserved key out of place + // — in a pointered unit; a fixed unit skips the value whole. + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( key[0] == '&' && in.graph != NULL ) + { + if ( !first ) { in.report->malformed = true; in.bad = true; return false; } + if ( !TableJsonSkippedAmpersand( in, key, depth ) ) { return false; } + first = false; + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } + } + first = false; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return TableJsonSkipContainer( in, '}', depth ); + case '[': return TableJsonSkipContainer( in, ']', depth ); + case '"': return TableJsonScanString( in, NULL, 0, NULL ); + case 't': return TableJsonLiteral( in, "true" ); + case 'f': return TableJsonLiteral( in, "false" ); + case 'n': return TableJsonLiteral( in, "null" ); + case 0: in.bad = true; return false; + default: + { + // consumed, never converted: skipping needs no buffer, and this + // is the one walk a hostile text drives to the depth cap. It is + // the SAME production the value path scans, so an unknown key + // cannot smuggle past a number a named key would refuse. + bool integral = false; + if ( !TableJsonWalkNumber( in, &integral ) ) { in.bad = true; return false; } + return true; + } + } +} + +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration reads through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ); + +// place one scalar at one storage address +inline bool TableJsonReadScalar( TableJsonIn & in, void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; {} is None, and + // two keys is a text this walk will not guess at + const TableUnionInfo * arms = f->arms(); + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, 0 ); + if ( TableJsonPeek( in ) == '}' ) { in.pos++; return true; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t tag = 0; + for ( int64_t t = 1; t <= f->enum_max; t++ ) + { + if ( strcmp( f->enum_name( (uint64_t) t ), key ) == 0 ) { tag = t; break; } + } + if ( tag == 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + void * payload = (uint8_t *) storage + arms->arms[tag].offset; + const TableFieldInfo * arm = arms->arms[tag].field; + bool placed = true; + if ( arm != NULL ) + { + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2). A value of + // the wrong shape for that row is a KIND MISMATCH: the union + // reads None, the event is counted, and the enclosing object + // continues — the rule a FIELD's value lives under, one key + // down. A pointer arm's null is a null pointer, not a shape + // error, exactly as a pointer field's is (§16.7). + char got = TableJsonValueShape( in ); + if ( arm->kind == 17 && !arm->is_array && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + memset( payload, 0, (size_t) arms->arms[tag].size ); + } + else if ( got != TableJsonShape( arm ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( arm->kind == 17 && !arm->is_array ) + { + // A POINTER ARM'S VALUE IS THE POINTEE IN PLACE, or a + // node reference to one (§16.7) — the read a pointer + // FIELD takes, which is not the scalar walk + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadPointer( in, payload, arm, depth + 1 ) ) { return false; } + } + else + { + // SELECTION ZERO-ESTABLISHES THE ARM (SPEC §5): an arm + // takes no specified default, so zero is the establish + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadField( in, storage, arm, depth + 1 ) ) { return false; } + } + } + else if ( arms->arms[tag].table != NULL ) + { + if ( TableJsonValueShape( in ) != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else + { + arms->arms[tag].table->reset( payload ); + if ( !TableJsonReadTable( in, payload, arms->arms[tag].table, depth + 1 ) ) { return false; } + } + } + else + { + // A PAYLOAD-FREE ARM'S VALUE IS null (§2.6): the arm name + // selects it and there is nothing to place + if ( TableJsonValueShape( in ) != 'z' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( !TableJsonLiteral( in, "null" ) ) + { + return false; + } + } + if ( placed ) + { + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, (uint64_t) tag ); + } + } + char c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; // a second key: a one-of with two arms is not a value + return false; + } + if ( f->kind == 13 ) + { + f->table->reset( storage ); + return TableJsonReadTable( in, storage, f->table, depth + 1 ); + } + if ( TableJsonIsEnum( f ) ) + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + for ( int64_t v = 0; v <= f->enum_max; v++ ) + { + if ( strcmp( f->enum_name( (uint64_t) v ), name ) == 0 ) + { + TableJsonSetRaw( storage, f->elem_size, (uint64_t) v ); + return true; + } + } + // a name this build cannot name reads as None and counts as unknown, + // exactly as an unknown variant id does on the wire (§4) + TableJsonSetRaw( storage, f->elem_size, 0 ); + in.report->unknown++; + return true; + } + if ( TableJsonIsFlags( f ) ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + uint64_t bits = 0; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( c != '"' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + bool found = false; + for ( int64_t bit = 0; bit <= f->enum_max; bit++ ) + { + if ( strcmp( f->enum_name( (uint64_t) bit ), name ) == 0 ) + { + bits |= uint64_t( 1 ) << bit; + found = true; + break; + } + } + if ( !found ) { in.report->unknown++; } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + TableJsonSetRaw( storage, f->elem_size, bits ); + return true; + } + if ( f->kind == 1 ) + { + char c = TableJsonPeek( in ); + if ( c == 't' ) { if ( !TableJsonLiteral( in, "true" ) ) { return false; } TableJsonSetRaw( storage, f->elem_size, 1 ); return true; } + if ( !TableJsonLiteral( in, "false" ) ) { return false; } + TableJsonSetRaw( storage, f->elem_size, 0 ); + return true; + } + char token[kTableJsonMaxNumber]; + int32_t length = 0; + bool integral = false; + if ( !TableJsonScanNumber( in, token, kTableJsonMaxNumber, &length, &integral ) ) + { + in.bad = true; + return false; + } + if ( TableJsonKindWide( f->kind ) ) + { + return TableJsonReadWide( in, token, length, storage, f ); + } + if ( f->kind == 10 || f->kind == 11 ) + { + bool single = f->kind == 10; + double value = TableJsonTokenDouble( token, length, single ); + // A magnitude the field's format cannot hold is the WRONG SHAPE for + // the kind, and it never reaches storage: 1e400 is not a float64 and + // 1e300 is not a float32. Storing the infinity the conversion + // produced would leave an instance this walk called CLEAN that + // ToJsonMeasure then refuses forever (a non-finite float has no JSON + // spelling), and §16.1's one invariant is that a text which reads + // clean writes back. + if ( !TableJsonFinite( value ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( f->has_range ) + { + if ( value < f->range_min ) { value = f->range_min; in.report->clamped++; } + else if ( value > f->range_max ) { value = f->range_max; in.report->clamped++; } + } + if ( single ) + { + float narrow = (float) value; + if ( !TableJsonFinite( (double) narrow ) ) + { + in.report->kind_mismatch++; + return true; + } + memcpy( storage, &narrow, sizeof( narrow ) ); + } + else + { + memcpy( storage, &value, sizeof( value ) ); + } + return true; + } + // JSON HAS ONE NUMBER TYPE. 2.0 IS the integer 2 and 1e3 IS 1000, and a + // library that round-trips numbers through a double emits them that way — + // this walker's own float writer emits 1e+21. So an integer field takes + // any number whose VALUE is integral, however it was spelled; only a + // genuinely fractional value is the wrong shape for it. + bool is_signed = f->kind >= 2 && f->kind <= 5; + bool saturated = false; + int64_t value = 0; + if ( integral ) + { + value = TableJsonTokenInteger( token, length, is_signed, &saturated ); + } + else + { + double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( is_signed ) + { + if ( d >= 9223372036854775808.0 ) { value = INT64_MAX; saturated = true; } + else if ( d < -9223372036854775808.0 ) { value = INT64_MIN; saturated = true; } + else if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) d; } + } + else + { + if ( d < 0.0 ) + { + // a negative for an unsigned field clamps to zero, as the + // exact digit path already does + if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + value = 0; + saturated = true; + } + else if ( d >= 18446744073709551616.0 ) { value = (int64_t) UINT64_MAX; saturated = true; } + else if ( d != (double) (uint64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) (uint64_t) d; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->has_range ) + { + if ( (double) value < f->range_min ) { value = (int64_t) f->range_min; in.report->clamped++; } + else if ( (double) value > f->range_max ) { value = (int64_t) f->range_max; in.report->clamped++; } + } + // the field's own storage width is the last bound: a value past it + // clamps rather than wrapping, which is what the wire does too + if ( f->elem_size < 8 ) + { + if ( is_signed ) + { + int64_t high = ( int64_t( 1 ) << ( f->elem_size * 8 - 1 ) ) - 1; + int64_t low = -high - 1; + if ( value > high ) { value = high; in.report->clamped++; } + else if ( value < low ) { value = low; in.report->clamped++; } + } + else + { + uint64_t high = ( uint64_t( 1 ) << ( f->elem_size * 8 ) ) - 1; + if ( value < 0 ) { value = 0; in.report->clamped++; } + else if ( (uint64_t) value > high ) { value = (int64_t) high; in.report->clamped++; } + } + } + // at eight bytes the storage IS the parser's width, and an unsigned value + // past INT64_MAX rides here as a negative int64 by design — the token + // parser already turned a NEGATIVE token for an unsigned field into a + // clamped zero, so there is nothing left to bound. + TableJsonSetRaw( storage, f->elem_size, (uint64_t) value ); + return true; +} + +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ) +{ + uint8_t * storage = (uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonReadMap( in, (void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + + if ( f->kind == 12 ) + { + int32_t length = 0; + if ( !TableJsonScanString( in, (char *) storage, f->array_bound, &length ) ) { return false; } + storage[length] = 0; + TableJsonSetCount( base, f, length ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + // base64 decodes STRAIGHT INTO the field's storage, six bits at a + // time — no window, no temporary, so a bytes(N) of any declared + // extent reads the same way. A base64 body carries no escapes, so a + // backslash in one is simply not an alphabet character. + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + memset( storage, 0, (size_t) f->array_bound ); + TableJsonSetCount( base, f, 0 ); + const char * alphabet = TableJsonBase64Alphabet(); + int32_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + bool clamped = false; + bool malformed = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + const char * at = c != 0 ? strchr( alphabet, c ) : NULL; + if ( at == NULL ) { malformed = true; continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( at - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < f->array_bound ) + { + storage[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); + } + else + { + clamped = true; + } + } + } + if ( malformed ) + { + // a body that is not base64 is the wrong shape for the kind: the + // field keeps its default and the event is counted + in.report->kind_mismatch++; + return true; + } + if ( clamped ) { in.report->clamped++; } + TableJsonSetCount( base, f, placed ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + // every slot back to its declared defaults first, so a key the text + // omits keeps them and a repeated field key cannot leave an earlier + // occurrence's slots standing + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + void * slot = storage + (int64_t) i * f->elem_size; + if ( f->kind == 13 ) { f->table->reset( slot ); } + else { memset( slot, 0, (size_t) f->elem_size ); } + } + char shape = TableJsonElementShape( f ); + // A KEYED OBJECT'S KEYS ARE KEYS: a variant named twice is a duplicate + // key like any other, last-wins and counted (§16.2). Tracked the way + // a table's own field keys are — a bounded, allocation-free bitmask; + // a vocabulary wider than this still reads, its repeats simply stop + // being counted. + uint64_t seen[8] = {}; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t slot = -1; + for ( int64_t v = 0; v < f->array_bound; v++ ) + { + // nothing is stored for None, so "None" finds no slot and is + // an unknown key like any other name this reader cannot place + if ( !TableJsonKeyedSlotValid( f, v ) ) { continue; } + if ( strcmp( f->key_name( TableJsonKeyedSlotKey( v ) ), key ) == 0 ) { slot = v; break; } + } + if ( slot >= 0 && slot < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( slot & 63 ); + if ( ( seen[slot >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[slot >> 6] |= bit; + } + if ( slot < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; + } + if ( f->is_array ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + // LAST WINS has to be true of a repeated ARRAY key too, and it is + // wire-visible: a fixed array writes every slot, so a second, shorter + // occurrence overlaying a prefix would leave the first occurrence's + // tail standing. The field goes back to its declared defaults before + // this occurrence's elements are placed — the re-establishment a nested + // table and a union arm already get. A table element's defaults are + // its own (the reset hook); every other element kind's storage + // default is zero, which is what the generated array declares. + if ( f->kind == 13 ) + { + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + f->table->reset( storage + (int64_t) i * f->elem_size ); + } + } + else + { + memset( storage, 0, (size_t) f->array_bound * (size_t) f->elem_size ); + } + TableJsonSetCount( base, f, 0 ); + int32_t placed = 0; + char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( placed >= f->array_bound ) + { + // more elements than the reader's bound: the bounded prefix + // is kept and the excess counts, the wire's rule (§4) + in.report->clamped++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( f->kind == 17 ) + { + // an element of an ARRAY OF POINTERS (§2.1): null is a null slot, an + // object is the pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( storage + (int64_t) placed * f->elem_size, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed++; + } + else + { + if ( !TableJsonReadScalar( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + // a fixed array's tail keeps the defaults the prefill left there, + // exactly as a short wire count does + TableJsonSetCount( base, f, placed ); + return true; + } + return TableJsonReadScalar( in, storage, f, depth ); +} + +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ); + +// ONE table object: keys are field keys, unknown ones are skipped and +// counted, a repeated key is last-wins and counted. The instance is already +// at its declared defaults when this is entered, so a key the text never +// mentions keeps the default an absent field takes on the wire (§4). +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + return TableJsonReadTableKeys( in, base, info, depth, NULL ); +} + +// The keys of an object whose brace is already consumed. A pointer's object +// opens the same way a table's does, but its FIRST key may be `&node` (§16.7) +// and the adapter that reads it has to scan the key to know — so it hands the +// key it scanned in as `first_key`, with the colon consumed, and this places +// it before scanning the rest. +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ) +{ + // duplicate tracking, bounded and allocation-free: a table with more + // fields than this still reads, its repeats simply stop being counted + uint64_t seen[8] = {}; + for ( ;; ) + { + char key[kTableJsonMaxKey]; + char c = 0; + if ( first_key != NULL ) + { + memcpy( key, first_key, strlen( first_key ) + 1 ); // scanned into a buffer this size by the caller + first_key = NULL; + } + else + { + c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + } + int32_t index = -1; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + if ( strcmp( info->fields[i].json, key ) == 0 ) { index = i; break; } + } + if ( key[0] == '&' ) + { + // THE AMPERSAND PREFIX IS RESERVED TO THE FORM (docs/SPEC-TABLES.md + // §16.7). No declaration may take a key beginning with it, so this + // is never a field this build lacks — it is the sharing construct + // somewhere it cannot stand: `&node` is the FIRST key of a pointer's + // object and nothing else, and the adapter that reads a pointer + // has consumed it before these keys are read. MALFORMED, refused + // and counted; never counted as unknown, never skipped. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( index < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + const TableFieldInfo * f = &info->fields[index]; + if ( index < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( index & 63 ); + if ( ( seen[index >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[index >> 6] |= bit; + } + // PRESENCE OF THE KEY IS THE PRESENCE (§16.2): reaching this line + // is the key being present, so an optional is set present + // whatever its value — with one exception the page names: a JSON + // null, which reads as ABSENT rather than as a value. + char got = TableJsonValueShape( in ); + if ( f->kind == 17 && !f->is_array ) + { + // a pointer: null is a null pointer, an object is the pointee + // in place or an `&node` reference to one (§16.7), a string is + // a BYTE BUFFER's bytes (§2.5), and anything else is the wrong + // shape for the kind + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) base + f->offset, f->elem_size, 0 ); + } + else if ( got != TableJsonShape( f ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) base + f->offset, f, depth ) ) + { + return false; + } + } + else if ( f->optional && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + // absent, and back at its defaults: a repeated key whose last + // occurrence is null must not leave an earlier value standing + if ( f->table != NULL ) { f->table->reset( (uint8_t *) base + f->offset ); } + else { memset( (uint8_t *) base + f->offset, 0, (size_t) f->elem_size ); } + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 0 ); + } + else + { + if ( got != TableJsonShape( f ) ) + { + // the wrong JSON type for the kind: skipped, never coerced + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, base, f, depth ) ) + { + return false; + } + if ( f->optional ) + { + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 1 ); + } + } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +// ---- the two entry points the per-table wrappers name ---- + +inline bool TableJsonRead( void * value, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = NULL; + info->reset( value ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, value, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +inline int64_t TableJsonWrite( const void * value, const TableTypeInfo * info, char * buffer, int64_t capacity ) +{ + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = NULL; + if ( !TableJsonWriteValue( out, value, info, 0 ) ) { return -1; } + // THE CANONICAL TEXT ENDS WITH EXACTLY ONE NEWLINE (docs/SPEC-TABLES.md + // §16.1). Every writer emits it — this walk, the C# walk and + // "schema unpack" — and every reader accepts a text with or without one, + // because the trailing whitespace a read already skips is what makes the + // two the same text. It is a byte of the FORM rather than a file + // convention: a text that is written to a file, pasted into a diff and + // handed back through a pipe has to be one text in all three places, and a + // buffer whose last byte is a closing brace is the one shape that is not. + out.put( '\n' ); + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json walk: end ---- + +// ---- json graph walk: begin ---- +// +// THE VARIABLE CLASS's half of the text form (docs/SPEC-TABLES.md §16.7). The +// walk above places every kind but one; this defines the three adapters it +// calls for that one, and the two entry points a pointered table's wrappers +// name. The text is the fixed class's — a pointee is an object in place — and a +// node named more than once carries `&node`: defined once, with its fields, +// and referenced after by `{ "&node": N }` alone. + +// ---- the identity map ---- +// +// ONE map shape serves both directions. Writing keys it by a node's ADDRESS and +// counts the slots that name the node, so the second pass knows at a node's +// first occurrence whether it will be named again; reading keys it by the +// text's own label and answers the node it defined. Open addressing, a +// multiply-shift hash and quadrupling growth — TablePackMap's shape (§6.2), on +// the same terms: proportional to nodes, never to bytes, on the authoring +// side, and released before the call returns. + +struct TableJsonGraphEntry +{ + uint64_t key; // a node's address (write) or a label (read); 0 is an empty slot + int64_t count; // write: how many slots name this node + int64_t label; // write: the `&node` label assigned at its first write, 0 until then + uint8_t open; // the descent is still open: a reference here is a cycle (write), a self-reference (read) + uint32_t node; // read: the node's arena offset; 0 for a definition the reader dropped + const TableTypeInfo * type; // read: the node's table; NULL for a dropped one +}; + +struct TableJsonGraphMap +{ + TableJsonGraphEntry * entries; + int64_t capacity; // a power of two, or zero while empty + int64_t count; + TableAllocator allocator; // the caller's pair (§6.5): the builder's on read, the one handed to ToJson on write +}; + +inline void TableJsonGraphMapInit( TableJsonGraphMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TableJsonGraphMapShutdown( TableJsonGraphMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TableJsonGraphMapInit( map, map.allocator ); +} + +inline int64_t TableJsonGraphMapSlot( const TableJsonGraphMap & map, uint64_t key ) +{ + uint64_t hash = key * 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != 0 && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TableJsonGraphEntry * TableJsonGraphMapFind( TableJsonGraphMap & map, uint64_t key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +inline bool TableJsonGraphMapGrow( TableJsonGraphMap & map ) +{ + TableJsonGraphMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 64; + grown.count = 0; + grown.entries = (TableJsonGraphEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TableJsonGraphEntry ) ); // zeroed, by the pair's contract + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == 0 ) { continue; } + grown.entries[ TableJsonGraphMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// the entry for a key, made if it was not there; `taken` says which. NULL is the +// allocator refusing, and the walk refuses with it. +inline TableJsonGraphEntry * TableJsonGraphMapReach( TableJsonGraphMap & map, uint64_t key, bool & taken ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TableJsonGraphMapGrow( map ) ) { return NULL; } + } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + taken = entry->key != key; + if ( taken ) + { + entry->key = key; + map.count++; + } + return entry; +} + +// ---- reading: into a builder ---- + +struct TableJsonGraphIn +{ + TableWorker * worker; // where every node comes from + TableJsonGraphMap labels; // a label -> the node it defined +}; + +// `&node`'s value, the LABEL: a positive integer spelled as one — digits, no sign, no +// fraction, no exponent, no leading zero (§16.7). Anything else is malformed. +inline bool TableJsonScanLabel( TableJsonIn & in, uint64_t & label ) +{ + TableJsonSpace( in ); + if ( in.pos >= in.size || in.text[in.pos] < '1' || in.text[in.pos] > '9' ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + uint64_t value = 0; + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) + { + uint64_t digit = (uint64_t) ( in.text[in.pos] - '0' ); + if ( value > ( UINT64_MAX - digit ) / 10 ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + value = value * 10 + digit; + in.pos++; + } + label = value; + return true; +} + +// A BYTE BUFFER's text (docs/SPEC-TABLES.md §2.5, §16.2): a string. For a +// *string the string's bytes become the blob; for a *bytes the string is base64 +// and its decoded bytes do. The blob is allocated at EXACTLY the decoded +// length — the string is scanned once without keeping it to learn the length, +// and once into the node — so a blob of any size reads with no window and no +// bound to clamp against. A *bytes body that is not base64 is the wrong shape +// for the kind: the reference stays null and the event is counted. +inline bool TableJsonReadBlob( TableJsonIn & in, void * slot, const TableFieldInfo * f ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + TableRef * ref = (TableRef *) slot; + ref->value = 0; + if ( strcmp( f->type_name, "string" ) == 0 ) + { + const int64_t mark = in.pos; + int32_t length = 0; + if ( !TableJsonScanString( in, NULL, 0, &length ) ) { return false; } + in.pos = mark; + char * data = TableStringEmplace( *graph->worker, *ref, NULL, (int64_t) length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int32_t placed = 0; + return TableJsonScanString( in, data, length, &placed ); + } + // base64: the alphabet characters decide the length, six bits apiece + const char * alphabet = TableJsonBase64Alphabet(); + const int64_t mark = in.pos + 1; + int64_t symbols = 0; + bool malformed = false; + in.pos++; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + if ( c == 0 || strchr( alphabet, c ) == NULL ) { malformed = true; continue; } + symbols++; + } + if ( malformed ) + { + in.report->kind_mismatch++; + return true; + } + const int64_t length = ( symbols * 6 ) / 8; + uint8_t * data = TableBytesEmplace( *graph->worker, *ref, length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int64_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + for ( int64_t at = mark; ; at++ ) + { + char c = in.text[at]; + if ( c == '"' ) { break; } + const char * symbol = c != '=' ? strchr( alphabet, c ) : NULL; + if ( symbol == NULL ) { continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( symbol - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < length ) { data[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); } + } + } + return true; +} + +// A pointer's object. Its FIRST key decides what it is: `&node` naming a label not +// yet defined, with fields after it, is a DEFINITION; `&node` naming one already +// defined, alone, is a REFERENCE; any other key is a node named once, its +// object in place. The node comes from the +// builder's arena, and the slot holds its arena offset (§6.3). A pointer whose +// target is a BYTE BUFFER — no table — takes a string instead (§2.5). +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( f->table == NULL ) { return TableJsonReadBlob( in, slot, f ); } + // the pointee nests one level down, exactly as a by-value table does, and + // takes the same cap: a chain nests as deep as it is long (§16.7) + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + char c = TableJsonPeek( in ); + if ( c == '}' ) + { + // an empty object: a node at its defaults, named once + in.pos++; + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + return true; + } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( strcmp( key, "&node" ) != 0 ) + { + // a node named once: the pointee's object in place, and this key is + // its first field — unless it is the reserved prefix under a spelling + // this form does not have, which ReadTableKeys refuses + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return TableJsonReadTableKeys( in, node, f->table, depth + 1, key ); + } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->labels, label, taken ); + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + // ONE SPELLING, and what follows the label says which half it is: fields + // after a label the text has not defined DEFINE it, and a label alone that + // the text has defined REFERS to it. The other two are malformed — a label + // alone that the text never defined, which would otherwise read as a default + // node under a silent report, and a field after a label already defined, + // which would be a second definition. That is what keeps a typo loud. + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + bool bare = c == '}'; + if ( bare == taken ) { in.report->malformed = true; in.bad = true; return false; } + if ( bare ) + { + // A REFERENCE. A label is defined when its object CLOSES, so a + // reference met inside its own definition — at any depth of by-value + // nesting — names a node whose descent is still open: the cycle the + // wire refuses (§3.1), refused here where it is written. A definition + // the reader dropped names no node, so the slot stays null with + // nothing more counted — the drop was counted where it happened. A + // node of another table than the slot declares is a kind mismatch, as + // on the wire. + in.pos++; + if ( entry->open != 0 ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + if ( entry->type == NULL ) + { + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + if ( entry->type != f->table ) + { + memcpy( slot, &ref, sizeof( ref ) ); + in.report->kind_mismatch++; + return true; + } + ref.value = (int64_t) entry->node; + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + // A DEFINITION: the node is allocated, the label is its, and the keys after + // `&node` are its fields. The entry is OPEN until the object closes, so a + // reference to the label from inside the node's own fields is refused as + // the cycle it is; the node and its table are filled in at the close. + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + entry->open = 1; + if ( !TableJsonReadTableKeys( in, node, f->table, depth + 1, NULL ) ) { return false; } + entry = TableJsonGraphMapFind( graph->labels, label ); // the map may have grown under the descent + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + memcpy( &ref, slot, sizeof( ref ) ); + entry->node = (uint32_t) ref.value; + entry->type = f->table; + entry->open = 0; + return true; +} + +// An `&`-prefixed key opening an object the walk is SKIPPING — a value past an +// array's bound, an unknown key's value, a value of the wrong shape. A +// definition in there still takes its label, so the numbering survives whatever +// the storage could not hold (§16.7): the label is registered with no node, and a +// reference to it reads null. Any other prefixed key is the reserved prefix +// out of place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL || strcmp( key, "&node" ) != 0 ) { in.report->malformed = true; in.bad = true; return false; } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + if ( TableJsonGraphMapReach( graph->labels, label, taken ) == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return true; // a fresh entry is node 0, type NULL: a definition with no node +} + +// ---- writing: from a region's const root ---- + +struct TableJsonGraphOut +{ + TableJsonGraphMap nodes; // a node's address -> how many slots name it, and its `&node` once assigned + bool counting; // PASS ONE: count the references, refuse a cycle, emit nothing + int64_t next_label; +}; + +// The node a slot names: null as `null`, a node named once as its object in +// place, and a node named more than once under the construct. Which of the +// last two it is was learned in pass one; pass two spells it. +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphOut * graph = (TableJsonGraphOut *) out.graph; + if ( graph == NULL ) { return false; } + const void * node = f->resolve( slot ); + if ( node == NULL ) + { + out.raw( "null", 4 ); + return true; + } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->nodes, (uint64_t) (uintptr_t) node, taken ); + if ( entry == NULL ) { return false; } + if ( f->table == NULL ) + { + // A BYTE BUFFER (§2.5, §16.7): its text is a string, which has no + // first key to carry `&node`, so a blob named from more than one + // slot has no spelling this form can carry and the graph is refused — + // as a shared node with nothing to write is. A blob named once is its + // bytes in place: base64 for a *bytes, the string itself for a *string. + if ( graph->counting ) { entry->count++; return true; } + if ( entry->count > 1 ) { return false; } + const TableBlob * blob = (const TableBlob *) node; + if ( blob->length > (uint32_t) 0x7fffffff ) { return false; } + if ( strcmp( f->type_name, "string" ) == 0 ) { TableJsonWriteString( out, (const char *) ( blob + 1 ), (int32_t) blob->length ); } + else { TableJsonWriteBase64( out, (const uint8_t *) ( blob + 1 ), (int32_t) blob->length ); } + return true; + } + if ( graph->counting ) + { + // PASS ONE: one visit per node, every slot that names it counted, and + // a reference to a node whose descent is still open is a cycle — + // refused here as the wire refuses it (§3.1) + entry->count++; + if ( !taken ) { return entry->open == 0; } + entry->open = 1; + if ( !TableJsonWriteValue( out, node, f->table, depth ) ) { return false; } + entry = TableJsonGraphMapFind( graph->nodes, (uint64_t) (uintptr_t) node ); // the map may have grown under the descent + if ( entry == NULL ) { return false; } + entry->open = 0; + return true; + } + // PASS TWO: a node named once is its object in place; a node named more + // than once is DEFINED at its first occurrence — `&node` first, then its + // fields — and REFERENCED by `&node` alone after that, spelled the same way at + // every site. Labels run from 1 in first-write order and are the text's own, + // so a stray number in a hand-edited text is most often one never defined. + if ( entry->count <= 1 ) + { + return TableJsonWriteValue( out, node, f->table, depth ); + } + if ( depth > kTableJsonMaxDepth ) { return false; } + if ( entry->label != 0 ) + { + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + out.line( depth ); + out.put( '}' ); + return true; + } + entry->label = ++graph->next_label; + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + bool any = true; + int64_t before = out.offset; + if ( !TableJsonWriteFields( out, node, f->table, depth, any ) ) { return false; } + // a definition carries at least one field, because a label alone is a + // reference: a shared node with nothing to write has no definition this + // form can spell, and the writer refuses it as it refuses any value it + // cannot spell (§16.3) + if ( out.offset == before ) { return false; } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- the two entry points a pointered table's wrappers name ---- + +// The text into the builder's root. Every node the text names is allocated in +// the builder's arena through the field's own Emplace; the label map is the +// walk's, released before this returns. The root itself takes no label — nothing +// may name it (§16.7) — so an `&node` at the root is refused like any other key +// of the prefix. +inline bool TableJsonReadGraph( TableWorker & worker, void * root, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + if ( worker.arena == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } + TableJsonGraphIn graph; + graph.worker = &worker; + TableJsonGraphMapInit( graph.labels, worker.arena->allocator ); + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = &graph; + info->reset( root ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, root, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + TableJsonGraphMapShutdown( graph.labels ); + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +// The text of a region's const root: measured when the buffer is NULL, written +// when it is not, over one code path. Two passes over one walk — the first +// counts how many slots name each node and refuses a cycle, the second writes +// — so a node's first occurrence knows whether it will be named again. The +// ROOT's entry is open for the whole first pass, so a reference back at it is +// the cycle it is (§3.1), and it takes no label. +inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * info, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + if ( root == NULL ) { return -1; } + TableJsonGraphOut graph; + TableJsonGraphMapInit( graph.nodes, allocator ); + graph.counting = true; + graph.next_label = 0; + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph.nodes, (uint64_t) (uintptr_t) root, taken ); + if ( entry == NULL ) { TableJsonGraphMapShutdown( graph.nodes ); return -1; } + entry->open = 1; + TableJsonOut count; + count.buffer = NULL; + count.capacity = 0; + count.offset = 0; + count.overflow = false; + count.graph = &graph; + bool ok = TableJsonWriteValue( count, root, info, 0 ); + graph.counting = false; + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = &graph; + if ( ok ) { ok = TableJsonWriteValue( out, root, info, 0 ); } + TableJsonGraphMapShutdown( graph.nodes ); + if ( !ok ) { return -1; } + out.put( '\n' ); // the canonical text ends with exactly one newline (§16.1) + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json graph walk: end ---- + +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + +// ---- json map walk: begin ---- + +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} + +// the entry's two rows: fields[0] IS the key and fields[1] IS the value, which +// is what makes a user's own table of pairs the same bytes (§2.8) +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } + +inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } +inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } + +// AN INTEGER KEY IS THE INTEGER'S DECIMAL SPELLING, QUOTED, because a JSON +// object's keys are strings. Written digit by digit so no locale can move it. +inline void TableJsonWriteMapIntegerKey( TableJsonOut & out, const void * storage, const TableFieldInfo * key ) +{ + uint64_t magnitude = 0; + bool negative = false; + if ( TableJsonMapKeySigned( key ) ) + { + int64_t value = 0; + switch ( key->kind ) + { + case 2: value = (int64_t) *(const int8_t *) storage; break; + case 3: value = (int64_t) *(const int16_t *) storage; break; + case 4: value = (int64_t) *(const int32_t *) storage; break; + default: value = *(const int64_t *) storage; break; + } + negative = value < 0; + magnitude = negative ? ( ~(uint64_t) value ) + 1 : (uint64_t) value; + } + else + { + switch ( key->kind ) + { + case 6: magnitude = (uint64_t) *(const uint8_t *) storage; break; + case 7: magnitude = (uint64_t) *(const uint16_t *) storage; break; + case 8: magnitude = (uint64_t) *(const uint32_t *) storage; break; + default: magnitude = *(const uint64_t *) storage; break; + } + } + char digits[24]; + int32_t at = (int32_t) sizeof( digits ); + do { digits[--at] = (char) ( '0' + ( magnitude % 10 ) ); magnitude /= 10; } while ( magnitude != 0 ); + if ( negative ) { digits[--at] = '-'; } + TableJsonWriteString( out, digits + at, (int32_t) sizeof( digits ) - at ); +} + +inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const TableFieldInfo * key ) +{ + const uint8_t * storage = (const uint8_t *) entry + key->offset; + if ( TableJsonMapKeyIsString( key ) ) + { + // A STRING KEY IS THE STRING (§2.8): every JSON key of a map object is + // a KEY OF THE MAP and none is a field key, so the `&` prefix §16.7 + // reserves for field keys is ordinary data here. + TableJsonWriteString( out, (const char *) storage, *(const int32_t *) ( (const uint8_t *) entry + key->count_offset ) ); + return; + } + TableJsonWriteMapIntegerKey( out, (const void *) storage, key ); +} + +// ToJson WRITES ENTRIES IN ASCENDING KEY ORDER, so unpack then pack is +// byte-stable and a diff of two texts is a diff of two maps (§2.8, §17.2). +// A region holds them in that order already, so this is the array in place. +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "{}", 2 ); return true; } + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); + out.put( '{' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); + TableJsonWriteMapKey( out, entry, key ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// AN INTEGER KEY IS READ BY §16.2's INTEGER RULE AND BY NOTHING ELSE, so +// "2.0" and "1e3" are the integers 2 and 1000 and "-0" is zero. The token is +// walked as a JSON number over its own bytes; a token that rule calls +// malformed makes the KEY malformed, and a genuinely fractional value, or one +// outside the key kind's range, is kind_mismatch for that entry. +inline bool TableJsonMapKeyValue( const char * token, int32_t length, const TableFieldInfo * key, + int64_t & value, bool & fits ) +{ + fits = false; + TableReport scratch; + TableJsonIn probe = { token, (int64_t) length, 0, &scratch, false, NULL }; + bool integral = false; + if ( !TableJsonWalkNumber( probe, &integral ) ) { return false; } + if ( probe.pos != (int64_t) length ) { return false; } // trailing bytes: not a number + if ( !integral ) + { + const double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) { return true; } // a value no key kind holds + const double whole = d < 0 ? -d : d; + if ( whole != (double) (int64_t) whole ) { return true; } // genuinely fractional + } + bool saturated = false; + const bool is_signed = TableJsonMapKeySigned( key ); + value = integral ? TableJsonTokenInteger( token, length, is_signed, &saturated ) + : (int64_t) TableJsonTokenDouble( token, length, false ); + if ( saturated ) { return true; } // outside every width: kind_mismatch, never clamped + switch ( key->kind ) + { + case 2: fits = value >= -128 && value <= 127; break; + case 3: fits = value >= -32768 && value <= 32767; break; + case 4: fits = value >= -2147483647 - 1 && value <= 2147483647; break; + case 5: fits = true; break; + case 6: fits = value >= 0 && value <= 255; break; + case 7: fits = value >= 0 && value <= 65535; break; + case 8: fits = value >= 0 && (uint64_t) value <= 4294967295ull; break; + default: fits = integral; break; // uint64: the token's own magnitude + } + return true; +} + +// FromJson READS KEYS IN WHATEVER ORDER THE TEXT GIVES THEM. A repeated key is +// last-wins and counted duplicate, the object rule (§16.2) applied inside the +// map. An empty object is an empty map, and null is kind_mismatch. +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const char shape = TableJsonShape( value ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char token[kTableJsonMaxKey]; + int32_t token_length = 0; + if ( !TableJsonScanString( in, token, kTableJsonMaxKey - 1, &token_length ) ) { return false; } + token[token_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t key_value = 0; + bool place = true; + if ( !TableJsonMapKeyIsString( key ) ) + { + bool fits = false; + if ( !TableJsonMapKeyValue( token, token_length, key, key_value, fits ) ) + { + // A MALFORMED KEY STOPS THE READ where §16.1's rule stops it, + // with the instance holding what was placed before the stop. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( !fits ) { in.report->kind_mismatch++; place = false; } + } + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; + if ( place && entry == NULL ) + { + // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the + // wire's rule, because a clamped key is a merged entry (§2.8). + in.report->clamped++; + } + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) + { + in.report->duplicate++; // last-wins, the object rule inside the map + } + const char got = TableJsonValueShape( in ); + if ( entry == NULL ) + { + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( value->kind == 17 && !value->is_array ) + { + // A POINTER VALUE IS SHARED EXACTLY AS A POINTER FIELD IS (§2.8): + // null is a null slot, an object is the pointee in place or an + // &node reference to one (§16.7), anything else is the wrong shape — + // the same three the field-key loop gives a pointer field, because + // an entry's value IS a field line. + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) entry + value->offset, value->elem_size, 0 ); + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) entry + value->offset, value, depth + 1 ) ) + { + return false; + } + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, entry, value, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json map walk: end ---- + +// ---- json list walk: begin ---- + +// an unbounded array is the out-of-line array that is not a map (§8.1) +inline bool TableJsonIsList( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && !TableJsonIsMap( f ); +} + +// ToJson WRITES THE ELEMENTS IN INDEX ORDER, which is the only order there is, +// so unpack then pack is byte-stable without a rule of its own (§2.9, §17.2). +// A region holds the array in place, so this steps it at the descriptor's pitch. +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + const uint8_t * elements = TableJsonExtentElements( slot ); + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const uint8_t * element = elements + (int64_t) i * f->elem_size; + if ( f->kind == 17 ) + { + // a []*T's elements take the pointer row (§16.7): the pointee's + // object in place, null, or `&node` for a shared one + if ( !TableJsonWritePointer( out, element, f, depth + 1 ) ) { return false; } + } + else if ( !TableJsonWriteScalar( out, element, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; +} + +// FromJson READS EVERY ELEMENT THE TEXT CARRIES, appending each through the +// descriptor's place resolver: `[]` is an empty list, and null is +// kind_mismatch, the array row's own rule (§16.2). LAST WINS holds for a +// repeated key: the list goes back to EMPTY before this occurrence's elements +// land, the builder's storage being reclaimed at reset (§2.9). +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) slot, 8, 0 ); + TableJsonSetRaw( (uint8_t *) slot + 8, 4, 0 ); + const char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + void * element = f->place( *graph->worker, slot, NULL, 0, 0 ); + if ( element == NULL ) + { + // NOT ADDED: the arena could not carve another segment, or the + // count met the int32 cap. The text cannot be placed whole, and + // the read stops where §16.1's rule stops it. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( f->kind == 17 ) + { + // an element of a []*T (§2.9): null is a null slot, an object is the + // pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) element, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, element, f, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + // the wrong shape for the element kind: the slot keeps its + // defaults and the event counts, the array row's rule (§16.2) + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, element, f, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json list walk: end ---- + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +bool PhotoFromJson( Photo & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, PhotoTableType(), text, bytes, report ); +} + +int64_t PhotoToJsonMeasure( const Photo & value ) +{ + return TableJsonWrite( &value, PhotoTableType(), NULL, 0 ); +} + +int64_t PhotoToJson( const Photo & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, PhotoTableType(), buffer, capacity ); +} + +bool AlbumFromJson( AlbumBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Album * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, AlbumTableType(), text, bytes, report ); +} + +int64_t AlbumToJsonMeasure( const Album * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, AlbumTableType(), NULL, 0, allocator ); +} + +int64_t AlbumToJson( const Album * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, AlbumTableType(), buffer, capacity, allocator ); +} + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/SharedTable.h b/testdata/golden/tables/lists/SharedTable.h new file mode 100644 index 000000000..1442d4542 --- /dev/null +++ b/testdata/golden/tables/lists/SharedTable.h @@ -0,0 +1,5530 @@ +// Code generated by the schema compiler from Shared.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — protocol id 0xa5fbe602c119cdd9 (packets only: tables version by field id, not by protocol id) +// The TABLE wire (evolution-tolerant, docs/SPEC-TABLES.md): no serialize +// dependency — includable from any TU. + +#pragma once + +#include +#include // the prefill's scalar-array fills +#include // offsetof, for the reflection descriptors + +// ---- the hooks (docs/USAGE.md, "the C++ table runtime's hooks") ---- +// +// schema_assert — the runtime's own assert, and the refusal a debugger reads. +// NDEBUG removes it, exactly as it removes assert. A caller who already routes +// serialize's asserts writes `#define schema_assert serialize_assert` before +// including this header and both halves land in one handler. +#ifndef schema_assert +#include +#define schema_assert assert +#endif // #ifndef schema_assert + +// schema_fatal — what stands after the assert on a path that cannot continue. +// NDEBUG does not remove it. Supply it and is never included. +#ifndef schema_fatal +#include // abort +#define schema_fatal abort +#endif // #ifndef schema_fatal + +// schema_allocate / schema_release — what "no allocator handed in" means for +// this program. schema_allocate hands back ZEROED bytes and NULL on failure: +// an arena segment is copied whole, padding included, so anything left +// uninitialized here would reach a packed region. Supply both and +// is never included; hand a TableAllocator to a builder to route one +// structure's allocations somewhere else again. +#ifndef schema_allocate +#include // calloc, free +#define schema_allocate( bytes ) calloc( (size_t) 1, (size_t) ( bytes ) ) +#define schema_release( pointer ) free( pointer ) +#endif // #ifndef schema_allocate +#include // a node's lifetime starts in arena storage (placement new) +#include // one atomic per slab: the arena is lock-free by ownership + +#include "Shared.h" + +#ifndef LISTDEMO_SCHEMA_TABLE_PRIMITIVES +#define LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +// THE CODEC DOES NOT DEPEND ON THE COMPILER'S INLINING BUDGET. A table of a +// realistic field count emits one large body per type, and the cursor a body +// writes through lives in the caller's `TableWriter`: across a call boundary +// that cursor round-trips through memory, and a `uint8_t *` store may alias the +// writer itself, so every put reloads it. When a budget runs out mid-body the +// codec silently degrades to that shape. Forcing the primitives and the +// fixed-class bodies inline is what keeps the cursor in registers and lets +// adjacent constant framing bytes merge into one store. +#if defined( _MSC_VER ) +#define LISTDEMO_TABLE_INLINE __forceinline +#elif defined( __GNUC__ ) || defined( __clang__ ) +#define LISTDEMO_TABLE_INLINE inline __attribute__(( always_inline )) +#else +#define LISTDEMO_TABLE_INLINE inline +#endif + +namespace listdemo { + +// WHY A READ WAS REFUSED, by name (docs/SPEC-TABLES.md §3.3, §11). A REFUSAL +// is not one of §4's events: nothing is decoded, no counter moves and no +// damage is reported, so five zero counters and a false flag are what a clean +// read prints too and only the verdict tells them apart. The reason says which +// refusal it was. +// +// This is the MESSAGE PATH's vocabulary and not the cooked form's (§7.4): a +// caller meeting one of these has been refused a MESSAGE on a connection, +// which is a different recovery with a different owner than a file a header +// match turned down. +enum TableMessageReason +{ + newer_form, // a FORM BYTE this reader does not carry (§3) + no_vocabulary, // no table for this connection: the message arrived before the announcement, or after a refused one + second_announcement, // a second announcement on a connection: it sets nothing, amends nothing, and the connection closes + vocabulary_too_large, // an announcement above the receiver's declared bound, refused before an entry is touched + message_form_as_file // a form 2 wire where a FILE was expected: its table is somewhere else +}; + +// The table-wire read report — the permissive contract's ledger. Silence +// (all zero) means the data matched this reader's schema exactly. +struct TableReport +{ + int32_t unknown = 0; // unknown field ids skipped (newer data) + int32_t kind_mismatch = 0; // known id, changed type — skipped, never misdecoded + int32_t clamped = 0; // out-of-range values clamped to declared bounds + // a key the TEXT form saw twice: last wins, and the repeat is counted + // (docs/SPEC-TABLES.md §16.2). The wire never raises it — a body carrying an + // id twice is legal input whose last occurrence wins, silently (§3). + int32_t duplicate = 0; + bool malformed = false; // framing damage; decode stopped, partial result kept + // THE REFUSAL VERDICT, which is not one of §4's events and moves no counter + // (docs/SPEC-TABLES.md §3): a FORM BYTE this reader does not carry. Five + // zero counters and a false flag are what a clean read prints too, so the + // verdict is what tells the two apart. + bool refused = false; + // WHICH refusal, and it is read only when refused is set: a read that + // was not refused has no reason, and this member is the one the caller + // must not look at then (docs/SPEC-TABLES.md §3.3). + TableMessageReason reason = newer_form; +}; + + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; +// ---- reflection (tables only, docs/SPEC-TABLES.md) ---- +// +// Static field descriptors for every type in the table closure: name, wire +// id/kind, storage offset, bounds, ranges, enum names and branch guards — +// enough to walk, print, diff, edit or bind any table value at runtime with +// no RTTI and no schema files. TableType() returns X's descriptor. + +struct TableTypeInfo; + +// One arm of a union field: where its payload sits inside the union's storage +// and what its payload looks like. The arm's NAME and its table-wire id come +// from the field's enum_name/variant_id functions at the same tag, so nothing +// is spelled twice (docs/SPEC-TABLES.md §8). +struct TableFieldInfo; + +struct TableUnionArmInfo +{ + uint32_t offset; // offsetof the arm's payload within the union storage + const TableTypeInfo * table; // the arm payload's descriptor, or NULL + // AN ARM IS A FIELD LINE (docs/SPEC-TABLES.md §2.6): an arm that names no + // declared type or table carries the FIELD descriptor a field of that + // type would carry instead — offsets taken within the union storage — so + // a generic walk meets an arm's kind, width, bounds and companions where + // it meets a field's. Exactly one of the two is non-NULL on a set arm. + const TableFieldInfo * field; + uint32_t size; // the arm's whole storage, which selection zero-establishes +}; + +// A union field's shape: the tag, and the arms indexed by it. Arms run +// [0, enum_max]; index 0 is the EMPTY arm and carries no payload. +struct TableUnionInfo +{ + uint32_t tag_offset; // offsetof the tag within the union storage + uint32_t tag_size; // sizeof the tag + const TableUnionArmInfo * arms; +}; + +// The exact raw range of a wide-kind field (docs/SPEC-TABLES.md §8.2): two 128-bit +// values as 64-bit lanes, low lane first, two's complement for the signed kinds. +struct TableWideRange +{ + uint64_t lo[2]; + uint64_t hi[2]; +}; + +// the arena's allocation front, defined with the variable-length runtime +// below; a descriptor names it only through a pointer parameter. +struct TableWorker; + +struct TableFieldInfo +{ + const char * name; // schema field name, e.g. "health" + const char * json; // the TEXT form's key: the json = "key" attribute, else name (§16.3) + const char * type_name; // schema type name, e.g. "float32", "Grade" + uint64_t id; // table-wire field id: fnv1a64 of the name, of the was alias after a rename (§5) + uint8_t kind; // table-wire kind; for arrays/strings/bytes, the ELEMENT kind + bool is_array; // fixed or counted array (bytes included) + bool is_pointer; // a *T pointer field: storage is an 8-byte TableRef; the target is a table + // THE TWO THE TEXT FORM NEEDS (docs/SPEC-TABLES.md §16.7), and they + // are here for the same reason is_pointer is: the walk is ONE walk + // over descriptors and cannot spell a target's own At or + // Emplace. `resolve` reads a slot in a REGION and answers the + // node it names, or NULL; `emplace` allocates one in a BUILDER's + // arena and points the slot at it. NULL on every field that is not + // a pointer, and emitted only in a unit that declares one. + const void * (*resolve)( const void * slot ); + void * (*emplace)( TableWorker & worker, void * slot ); + bool counted; // a _count/_length int32 companion exists (counted arrays, strings, bytes) + bool optional; // a ?T field: a _present bool companion decides whether it rides + int32_t array_bound; // array capacity / string max length; 0 for plain scalars + uint32_t offset; // offsetof the storage member + uint32_t elem_size; // sizeof the member (element size for arrays) + uint32_t count_offset; // offsetof the _count/_length companion, or 0xffffffff + uint32_t present_offset; // offsetof the _present companion, or 0xffffffff + const TableTypeInfo * table; // nested table's descriptor, or NULL + bool has_range; // a declared [min, max] (int or float) + double range_min; // NOTE: int64 ranges beyond 2^53 lose precision here + double range_max; + // the WIDE kinds (18-29, docs/SPEC-TABLES.md §3, §8.2): frac_bits is a fixed + // field's F — its storage holds units × 2^F — and wide is the declared + // range on that RAW scale, exact, as two 128-bit two's-complement values + // in 64-bit lanes (low lane first). NULL where the declaration bounds + // nothing (a bare uint128) and for every other kind; frac_bits is 0 for + // every kind that is not fixed-point. range_min/range_max still carry + // the declared bounds as doubles — whole units for a fixed field — for + // a walker that only shows them. + uint8_t frac_bits; + const TableWideRange * wide; + int64_t enum_max; // enums: highest valid value (None = 0 always valid); + // unions: the arm count (tag range [0, enum_max]); + // flags: the highest declared BIT INDEX; else -1 + // the vocabulary's names, indexed the same way enum_max bounds: an enum's + // value -> name, a union's tag -> arm name, a FLAGS field's bit index -> + // variant name. NULL for every other kind. + const char * (*enum_name)( uint64_t value ); + // the TABLE-WIRE id of one variant (docs/SPEC-TABLES.md §5): for an enum, the + // hash of the variant's name; for a union, the hash of the arm's name. + // 0 is the reserved id — an enum's None, a union's empty. NULL for every + // other kind — a FLAGS field's variants have no per-variant wire id (§4), + // so a NULL here beside a non-NULL enum_name is what says "flags". + // Walk [0, enum_max] to enumerate a vocabulary and its ids. + uint64_t (*variant_id)( uint64_t value ); + // an ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): the array has one slot per + // variant of key_type_name, indexed by the variant's value, and its slots + // ride under variant ids rather than positions. key_name and key_id are + // the key's vocabulary — walk [0, array_bound) to print slots by name. + // NULL on every other field. + const char * key_type_name; + const char * (*key_name)( uint64_t value ); + uint64_t (*key_id)( uint64_t value ); + // union fields: the tag and its arms, behind a function so the whole + // descriptor stays CONSTANT-INITIALISED (a captureless lambda converts to + // a function pointer at compile time; the arms themselves are a static + // inside it). NULL for every other kind. + const TableUnionInfo * (*arms)(); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded +}; + +struct TableTypeInfo +{ + const char * name; // schema type name + uint32_t size; // sizeof the storage struct + int32_t num_fields; + const TableFieldInfo * fields; + // put one instance back at its declared defaults, in place. A generic + // walker that fills a value has to be able to establish the defaults an + // absent field takes, and it holds no type to spell — this is the one + // thing the descriptors could not express without it. Placement-new + // value-init, exactly what the wire's read path does, and no temporary. + void (*reset)( void * storage ); + // the DERIVED mode (docs/SPEC-TABLES.md): false = fixed-size, a plain + // relocatable struct; true = variable-length, built through a Builder + // and read through a region root. Nobody declares it; the compiler + // works it out. + bool variable; +}; + +struct TableWriter +{ + uint8_t * buffer; + int64_t capacity; + int64_t offset = 0; + bool overflow = false; + + // the parameters do not repeat the member names: a parameter that hides a + // member is a warning the estate's compilers disagree about (gcc's + // -Wshadow and cl's C4458 refuse it, clang's -Wshadow does not), and this + // is a header a consumer compiles under its OWN flags + TableWriter( uint8_t * to_buffer, int64_t to_capacity ) : buffer( to_buffer ), capacity( to_capacity ) {} + + LISTDEMO_TABLE_INLINE void raw( const void * data, int64_t bytes ) + { + if ( offset + bytes > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) bytes ); + offset += bytes; + } + LISTDEMO_TABLE_INLINE void put8( uint8_t v ) { raw( &v, 1 ); } + LISTDEMO_TABLE_INLINE void put16( uint16_t v ) { uint8_t b[2] = { uint8_t( v ), uint8_t( v >> 8 ) }; raw( b, 2 ); } + LISTDEMO_TABLE_INLINE void put32( uint32_t v ) { uint8_t b[4] = { uint8_t( v ), uint8_t( v >> 8 ), uint8_t( v >> 16 ), uint8_t( v >> 24 ) }; raw( b, 4 ); } + LISTDEMO_TABLE_INLINE void put64( uint64_t v ) { put32( uint32_t( v ) ); put32( uint32_t( v >> 32 ) ); } + // a 128-bit value as two lanes, the low half first (docs/SPEC-TABLES.md §3) + LISTDEMO_TABLE_INLINE void put128( uint64_t lo, uint64_t hi ) { put64( lo ); put64( hi ); } + // EVERY LENGTH, COUNT, INDEX AND ID REFERENCE IS ONE CANONICAL UNSIGNED + // LEB128 (docs/SPEC-TABLES.md §3): seven value bits a byte, the lowest + // group first, the high bit set on every byte but the last. One value has + // one spelling, so two conforming writers agree byte for byte. + LISTDEMO_TABLE_INLINE void putleb( uint64_t v ) + { + while ( v >= 0x80 ) { put8( uint8_t( v ) | 0x80 ); v >>= 7; } + put8( uint8_t( v ) ); + } +}; + +// TableLebBytes is one value's spelling length, which a MEASURE needs before +// the bytes exist — the length of a body has to be known before it is written, +// because a length whose own width moves cannot be patched in place. +inline int64_t TableLebBytes( uint64_t v ) +{ + int64_t n = 1; + while ( v >= 0x80 ) { v >>= 7; n++; } + return n; +} + +// THE ID TABLE, WRITER SIDE (docs/SPEC-TABLES.md §3). It holds every id the +// body used, once each, in FIRST-USE order over the whole wire, and the body +// names them by position: reference k is the kth entry, counted from 1, and +// reference 0 names NO ID. +// +// Its capacity is a COMPILE-TIME fact of the unit — the distinct names its +// table closure can spell — so a save allocates nothing: the table is a local +// of Measure and of Save. The bucket chain makes ref constant time and makes +// truncate constant time too, which is what an ELIDED field needs: a field +// that turns out not to ride costs nothing in the id table either, so the walk +// interns its id, builds the payload that decides, and undoes the entry when +// nothing rides. +struct TableIds +{ + static const int32_t kCapacity = 60; + static const int32_t kBuckets = 128; + + uint64_t ids[ kCapacity ]; + int32_t chain[ kCapacity ]; + int32_t head[ kBuckets ]; + int32_t count; + bool overflow; + // THE MESSAGE FORM'S SLOTS (docs/SPEC-TABLES.md §3.3). A form 2 wire + // names ids through the CONNECTION's table, which is the unit's whole + // vocabulary in a compiler-settled order — so every reference is known at + // compile time and rides at the header as a literal beside the id. This + // flag is what selects it: false interns the id in first-use order and + // writes a trailer, true answers the slot and writes none, and the walk + // that decides is one walk. + bool vocabulary; + + TableIds() : count( 0 ), overflow( false ), vocabulary( false ) + { + for ( int32_t i = 0; i < kBuckets; i++ ) { head[i] = -1; } + } + + static LISTDEMO_TABLE_INLINE uint32_t bucket_of( uint64_t id ) + { + return uint32_t( ( id * 0x9E3779B97F4A7C15ull ) >> 57 ) & uint32_t( kBuckets - 1 ); + } + + // the reference an id takes: its message-form SLOT under the connection's + // table, or the file's own first-use entry + LISTDEMO_TABLE_INLINE uint64_t ref( uint64_t id, uint64_t slot ) + { + if ( vocabulary ) { return slot; } + return intern( id ); + } + + // the FILE form's half, appending the id on first use + uint64_t intern( uint64_t id ) + { + const uint32_t b = bucket_of( id ); + for ( int32_t i = head[b]; i >= 0; i = chain[i] ) + { + if ( ids[i] == id ) { return uint64_t( i ) + 1; } + } + if ( count >= kCapacity ) { overflow = true; return 1; } + ids[count] = id; chain[count] = head[b]; head[b] = count; count++; + return uint64_t( count ); + } + + // undo every entry appended since mark. An entry removed is the most + // recent one in its bucket, so it sits at that bucket's head. + void truncate( int32_t mark ) + { + // a SLOT costs no entry, so an elided field has nothing to undo + if ( vocabulary ) { return; } + while ( count > mark ) + { + count--; + head[ bucket_of( ids[count] ) ] = chain[count]; + } + } +}; + +// TableIdsBytes is the trailer's own size: the entries, each a fixed +// little-endian u64, and the ENTRY COUNT, the one fixed-width number on the +// wire (docs/SPEC-TABLES.md §3). +inline int64_t TableIdsBytes( const TableIds & ids ) { return int64_t( ids.count ) * 8 + 8; } + +// TableIdsWrite puts the trailer where the walk ended: a writer never patches, +// because first-use order is known only when the walk ends. +inline void TableIdsWrite( TableWriter & w, const TableIds & ids ) +{ + for ( int32_t i = 0; i < ids.count; i++ ) { w.put64( ids.ids[i] ); } + w.put64( uint64_t( ids.count ) ); +} + +// THE ID TABLE, READER SIDE (docs/SPEC-TABLES.md §3). A reader locates it from +// the END of the wire and resolves it ONCE, at open: the entries are eight +// bytes each and a body names them by position, so every field dispatches +// through an index rather than through a search over hashes. +struct TableIdTable +{ + const uint8_t * entries = NULL; + int64_t count = 0; + + // the id a reference names. ref is 1-based and bounds-checked by the + // caller: a reference ABOVE the entry count is framing damage on the body + // that carries it, and 0 names no id at all. + uint64_t at( uint64_t ref ) const + { + const uint8_t * e = entries + ( ref - 1 ) * 8; + uint64_t lo = uint64_t( e[0] ) | uint64_t( e[1] ) << 8 | uint64_t( e[2] ) << 16 | uint64_t( e[3] ) << 24; + uint64_t hi = uint64_t( e[4] ) | uint64_t( e[5] ) << 8 | uint64_t( e[6] ) << 16 | uint64_t( e[7] ) << 24; + return lo | ( hi << 32 ); + } +}; + +struct TableReader +{ + const uint8_t * buffer; + int64_t size; + int64_t offset = 0; + TableReport * report; + const TableIdTable * ids = NULL; + // ONLY THE ROOT BODY CARRIES THE NODE TABLE (docs/SPEC-TABLES.md §3.1), so + // a body has to know which it is: the reserved id inside a NESTED body is + // malformed, because a second numbering cannot exist. Every reader made + // for a payload is nested; the two the wire surfaces make for a root say so. + bool nested = true; + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report ) + : buffer( from_buffer ), size( from_size ), report( to_report ) {} + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report, const TableIdTable * to_ids ) + : buffer( from_buffer ), size( from_size ), report( to_report ), ids( to_ids ) {} + + LISTDEMO_TABLE_INLINE bool has( int64_t bytes ) const { return offset + bytes <= size; } + // A LENGTH IS A 64-BIT NUMBER AND A BUFFER IS NOT (docs/SPEC-TABLES.md + // §3): every length, count and index on this wire has sixty-four bits of + // capability, so one past what remains must be compared UNSIGNED. Casting + // it to int64 first turns 0xFFFFFFFFFFFFFFFF into -1, and a negative + // length looks like room. + LISTDEMO_TABLE_INLINE bool room( uint64_t bytes ) const { return bytes <= (uint64_t) ( size - offset ); } + LISTDEMO_TABLE_INLINE uint8_t get8() { return buffer[offset++]; } + LISTDEMO_TABLE_INLINE uint16_t get16() { uint16_t v = uint16_t( buffer[offset] ) | uint16_t( buffer[offset+1] ) << 8; offset += 2; return v; } + LISTDEMO_TABLE_INLINE uint32_t get32() { uint32_t v = uint32_t( buffer[offset] ) | uint32_t( buffer[offset+1] ) << 8 | uint32_t( buffer[offset+2] ) << 16 | uint32_t( buffer[offset+3] ) << 24; offset += 4; return v; } + LISTDEMO_TABLE_INLINE uint64_t get64() { uint64_t lo = get32(); uint64_t hi = get32(); return lo | ( hi << 32 ); } + LISTDEMO_TABLE_INLINE void get128( uint64_t & lo, uint64_t & hi ) { lo = get64(); hi = get64(); } + + // ONE CANONICAL UNSIGNED LEB128 (docs/SPEC-TABLES.md §3), and a + // non-minimal spelling is MALFORMED: 0x80 0x00 and 0x00 both spell zero, + // and only the second is legal input. An encoding past ten bytes, or a + // tenth byte with a bit above the 64th value bit, is malformed on the same + // rule. false = framing damage on the body carrying it. + bool getleb( uint64_t & value ) + { + // A NUMBER THIS READER REFUSES LEAVES THE CURSOR WHERE IT WAS. The + // caller's next question is often "did this body end exactly at its + // L", and a rejected number that had moved the cursor would answer + // that question with the damage already stepped over. + const int64_t at = offset; + value = 0; + uint32_t shift = 0; + for ( int32_t i = 0; i < 10; i++ ) + { + if ( !has( 1 ) ) { offset = at; return false; } + const uint8_t b = get8(); + if ( i == 9 && b > 1 ) { offset = at; return false; } + value |= uint64_t( b & 0x7F ) << shift; + if ( ( b & 0x80 ) == 0 ) + { + if ( i > 0 && b == 0 ) { offset = at; return false; } // a redundant continuation + return true; + } + shift += 7; + } + offset = at; + return false; + } + + // resolve one id reference against the file's table. false = a reference + // ABOVE the entry count, or a 0 where an id is required, both of which + // are framing damage on the body that carries it. + bool getid( uint64_t & id ) + { + uint64_t ref = 0; + if ( !getleb( ref ) ) { return false; } + if ( ref == 0 || ids == NULL || ref > (uint64_t) ids->count ) { return false; } + id = ids->at( ref ); + return true; + } + + // skip one payload by kind; false = framing damage. FOUR RULES COVER THE + // SET (docs/SPEC-TABLES.md §3), and a kind outside it is not skippable — + // which is why the set is closed and why kind 31 exists. + bool skip( uint8_t kind ) + { + switch ( kind ) + { + // the fixed-width kinds, each by its width: 18-29 are the 128-bit integers and + // the fixed-point family at every storage width (docs/SPEC-TABLES.md §3) + case 1: case 2: case 6: case 20: case 25: return has( 1 ) ? ( offset += 1, true ) : false; + case 3: case 7: case 21: case 26: return has( 2 ) ? ( offset += 2, true ) : false; + case 4: case 8: case 10: case 22: case 27: return has( 4 ) ? ( offset += 4, true ) : false; + case 5: case 9: case 11: case 23: case 28: return has( 8 ) ? ( offset += 8, true ) : false; + case 18: case 19: case 24: case 29: return has( 16 ) ? ( offset += 16, true ) : false; + case 17: case 30: // a NODE INDEX (§3.1) and an ENUM's variant reference: one LEB128 and stop + { + uint64_t ignored = 0; + return getleb( ignored ); + } + case 12: case 13: case 14: case 16: case 31: case 32: // 31 is the ESCAPE, 32 the payload-free kind + { + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + case 15: // union: the arm id reference, then its kind, its L and its payload (reference 0 = empty) + { + uint64_t arm = 0; + if ( !getleb( arm ) ) return false; + if ( arm == 0 ) return true; + if ( !has( 1 ) ) return false; + offset += 1; // the arm's kind byte + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + } + return false; + } +}; + +// The RESERVED node-table id, the one id the language holds back +// (docs/SPEC-TABLES.md §3.1, §5). It rides in every unit, pointered or not, +// because every body has to know that a NESTED body claiming one is damaged. +static const uint64_t kTableNodeTableFieldId = 0xFFFFFFFFFFFFFFFFull; + +// TableWireForm is the FORM BYTE, and it is the whole header +// (docs/SPEC-TABLES.md §3). A reader that meets a byte it does not know +// refuses the wire by name and never reports damage. +const uint8_t kTableWireForm = 1; + +// TableOpen reads the form byte and the trailer, in that order, and hands back +// the ROOT BODY. It answers one of three verdicts, because five zero counters +// and a false flag are what a clean read prints too: +// +// TableOpenOk the form is known and the table read whole +// TableOpenRefused a FORM BYTE this reader does not carry: nothing is +// decoded, nothing is counted, and no damage is reported +// TableOpenDamaged a table that cannot be read whole — fewer than eight +// bytes, a count whose entries run past the front of the +// file, a count that leaves no room for the form byte, or +// ONE ID IN TWO ENTRIES. The whole wire is malformed, +// nothing is decoded, and one event is counted. +// TableOpenBodyStopped the form and the table were good and the ROOT BODY +// could not be walked to its own terminator. What it +// decoded before that is kept, as everywhere on this wire. +enum TableOpenVerdict { TableOpenOk, TableOpenRefused, TableOpenDamaged, TableOpenBodyStopped }; + +inline TableOpenVerdict TableOpen( const uint8_t * buffer, int64_t bytes, TableIdTable & table, int64_t & body_bytes ) +{ + if ( bytes < 1 ) { return TableOpenDamaged; } + if ( buffer[0] != kTableWireForm ) { return TableOpenRefused; } + if ( bytes < 9 ) { return TableOpenDamaged; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + uint64_t count = lo | ( hi << 32 ); + if ( count > (uint64_t) ( bytes / 8 ) ) { return TableOpenDamaged; } + const int64_t span = (int64_t) count * 8 + 8; + if ( span + 1 > bytes ) { return TableOpenDamaged; } + table.entries = buffer + bytes - span; + table.count = (int64_t) count; + // THE ENTRIES ARE DISTINCT: a table that carries one id twice is malformed + // for the whole wire, because no wire this schema writes carries a repeat + // and it would leave one more shape of table for a hostile writer to aim + // at (docs/SPEC-TABLES.md §3). + for ( int64_t i = 1; i < table.count; i++ ) + { + const uint64_t id = table.at( uint64_t( i ) + 1 ); + for ( int64_t j = 0; j < i; j++ ) + { + if ( table.at( uint64_t( j ) + 1 ) == id ) { return TableOpenDamaged; } + } + } + body_bytes = bytes - span - 1; + return TableOpenOk; +} + +// TableBodyExtent walks a body's framing to the zero reference that ends it, +// so a reader can tell a body that ENDED EARLY — leaving bytes no field claims +// — from one that is merely damaged. ANY BYTE BETWEEN THE ROOT'S TERMINATOR +// AND THE TABLE'S FIRST ENTRY IS MALFORMED, because no field claims it and the +// two ends of the file have met (docs/SPEC-TABLES.md §3). +inline bool TableBodyEndsEarly( const uint8_t * body, int64_t bytes, const TableIdTable & table ) +{ + TableReport ignored; + TableReader r( body, bytes, &ignored, &table ); + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { return false; } + if ( ref == 0 ) { return r.offset != bytes; } + if ( ref > (uint64_t) table.count ) { return false; } + if ( !r.has( 1 ) ) { return false; } + if ( !r.skip( r.get8() ) ) { return false; } + } +} + +// THE MESSAGE FORM (docs/SPEC-TABLES.md §3.3): a FILE carries its own id +// table and a MESSAGE STREAM announces one and then carries none. +// +// A form 2 wire is TWO PARTS, the form byte and the root body: the body ends +// at its own zero reference as it does in a file, there is no trailer, and the +// message's last byte is the body's terminator. Its references resolve against +// the CONNECTION's table, which is the unit's whole vocabulary in the order +// the compiler settled. +const uint8_t kTableWireMessageForm = 2; + +// The RESERVED build-version id, the second id the language holds back (§5, +// §11), beside the node table's. It is the announcement's one required field, +// and a reserved id in any body but the one whose transport it is, is +// malformed (§3.1). +static const uint64_t kTableBuildVersionFieldId = 0xFFFFFFFFFFFFFFFEull; + +// The reserved NODE-TABLE id's own slot in this unit's vocabulary (§3.3). A +// pointered message names the node table through it, exactly as every other +// field header names its id through a slot. +static const uint64_t kTableNodeTableFieldSlot = 38; + +// THE UNIT'S ANNOUNCEMENT, byte for byte: 61 entries and 508 bytes. It is an +// ordinary form 1 FILE — the form byte, a body carrying the BUILD VERSION +// under the reserved id at kind 9, and the trailer that IS the connection's +// table, slot 1 the reserved id and slots 2 and up the vocabulary under one +// numbering. +// +// The vocabulary is the unit's whole closure in the COOK PROJECTION's order +// (§20.2) — each record in the order the projection renders it and each +// record's fields in the order the projection renders them, then each enum's +// variants and each union's arms — followed by the tail the projection does +// not name: the reserved node-table id, the three blob type ids as bytes, +// string and wstring, and every table's own name id in the projection's sorted +// record order. The tail is UNCONDITIONAL, so an ordinary edit only ever grows +// it at its end and never moves a slot a generated field header carries as a +// literal. +static const int64_t kTableAnnounceBytes = 508; +static const uint8_t kTableAnnounce[ kTableAnnounceBytes ] = { + 0x01, 0x01, 0x09, 0xc7, 0x71, 0x45, 0xca, 0xda, 0x0e, 0x7c, 0x8d, 0x00, + 0xfe, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x30, 0xb1, 0x3a, 0xff, + 0x4a, 0xd9, 0xb1, 0x40, 0x20, 0xea, 0x4d, 0x40, 0x8e, 0xa7, 0x19, 0xaa, + 0x26, 0xa9, 0x02, 0x0c, 0x9b, 0x01, 0x48, 0x78, 0xe9, 0xea, 0x71, 0x6f, + 0x0f, 0x01, 0x82, 0xbf, 0x6f, 0x2c, 0x41, 0x4f, 0xbf, 0x84, 0x78, 0x3e, + 0xf3, 0xa4, 0x48, 0x44, 0x19, 0xab, 0xd7, 0x56, 0x05, 0x4a, 0xa3, 0x30, + 0x67, 0x55, 0x5b, 0x85, 0xc9, 0xe2, 0x4e, 0x30, 0x69, 0x6a, 0xb4, 0x81, + 0xfb, 0x67, 0x4d, 0x1a, 0xcf, 0x7b, 0x27, 0x21, 0x74, 0xa2, 0x79, 0x44, + 0x8e, 0xe2, 0xe5, 0xb1, 0x84, 0x76, 0xbc, 0x2e, 0xef, 0x83, 0x76, 0x1e, + 0xc5, 0x99, 0xf7, 0x82, 0x76, 0x4e, 0x0a, 0xd9, 0xa8, 0x2e, 0x86, 0x70, + 0x84, 0xed, 0xf2, 0x4a, 0xbb, 0xf0, 0x0c, 0x9b, 0xcc, 0xfb, 0x2d, 0x73, + 0x68, 0xb7, 0xf0, 0xae, 0x4c, 0x0c, 0xf6, 0x52, 0xbf, 0xe9, 0xd1, 0x2f, + 0x93, 0xcd, 0xda, 0xdb, 0x22, 0x72, 0x34, 0x7d, 0xf6, 0x0b, 0x72, 0x17, + 0x07, 0x17, 0x02, 0x86, 0x4c, 0xf5, 0x63, 0xaf, 0x54, 0x15, 0x02, 0x86, + 0x4c, 0xf4, 0x63, 0xaf, 0x3a, 0x70, 0x6e, 0x3e, 0x93, 0x43, 0xe5, 0x9d, + 0x3d, 0x62, 0xcb, 0x8f, 0xec, 0xfc, 0xf7, 0x39, 0x09, 0x06, 0x02, 0x86, + 0x4c, 0xeb, 0x63, 0xaf, 0x09, 0x4b, 0x4d, 0x57, 0xaa, 0x33, 0x47, 0xd2, + 0x31, 0x54, 0xaf, 0x1d, 0x19, 0x73, 0x50, 0x12, 0xb2, 0x0f, 0x40, 0x27, + 0x0b, 0x6b, 0x98, 0x01, 0x38, 0x81, 0x0a, 0xf1, 0x1f, 0x06, 0xa7, 0xa3, + 0x0f, 0x62, 0xad, 0x07, 0x77, 0x47, 0x82, 0x5f, 0x42, 0x4f, 0x4f, 0x30, + 0x0d, 0x39, 0x84, 0x1c, 0x86, 0x1b, 0x63, 0x8e, 0xba, 0xad, 0xbc, 0xc4, + 0xec, 0x10, 0x5b, 0x36, 0x19, 0x4a, 0xc9, 0x3d, 0xea, 0x0c, 0xe8, 0x30, + 0x94, 0xfd, 0xe4, 0x7c, 0xec, 0x22, 0x02, 0x86, 0x4c, 0xfc, 0x63, 0xaf, + 0x05, 0x28, 0x02, 0x86, 0x4c, 0xff, 0x63, 0xaf, 0x52, 0x26, 0x02, 0x86, + 0x4c, 0xfe, 0x63, 0xaf, 0xb1, 0x45, 0xc3, 0x44, 0x35, 0xab, 0xfe, 0x73, + 0xc0, 0x7f, 0xb3, 0x8a, 0xbe, 0x08, 0x63, 0x7f, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xe4, 0x4f, 0x1c, 0x4f, 0x47, 0xc0, 0x2e, 0x2f, + 0x58, 0xfc, 0xaf, 0xfa, 0xd8, 0xe0, 0x4b, 0x70, 0xc7, 0xd4, 0x7b, 0x26, + 0xb0, 0x9d, 0x29, 0x5f, 0xcc, 0x14, 0x15, 0x7f, 0xcb, 0xc2, 0x58, 0xd8, + 0x84, 0x9e, 0x3a, 0x55, 0x8b, 0x37, 0xe2, 0x06, 0x2a, 0x89, 0xf5, 0x12, + 0x09, 0xc1, 0xca, 0x0a, 0x44, 0xa2, 0x31, 0xc1, 0xad, 0xa7, 0xee, 0xee, + 0xe8, 0xcf, 0xbf, 0x43, 0x73, 0x18, 0x43, 0xd0, 0x42, 0xad, 0xf6, 0xf8, + 0x59, 0x86, 0x63, 0x91, 0xb7, 0xce, 0x00, 0x7c, 0xd1, 0xc5, 0x34, 0x20, + 0x06, 0x68, 0x47, 0x98, 0xd1, 0xa1, 0xcf, 0x52, 0x5f, 0x82, 0x58, 0xac, + 0x36, 0x15, 0x78, 0x5e, 0xb8, 0x8b, 0x59, 0x6f, 0xc9, 0xc6, 0x86, 0xbb, + 0xc3, 0x64, 0x89, 0x50, 0xd2, 0x8d, 0xa7, 0xf1, 0x80, 0xea, 0x3a, 0xb9, + 0xf1, 0x21, 0xf7, 0x41, 0x11, 0xed, 0xd9, 0xce, 0x96, 0x92, 0x43, 0x8a, + 0xfb, 0x06, 0xc9, 0xfe, 0x19, 0xe1, 0x13, 0xa0, 0xa7, 0x0a, 0xc7, 0x54, + 0x12, 0xd6, 0x40, 0xdc, 0x08, 0xf0, 0xf5, 0xc0, 0x24, 0x5f, 0xf8, 0x33, + 0xc8, 0xfb, 0x85, 0x9a, 0xaf, 0xe0, 0xc9, 0x0c, 0x91, 0x0a, 0x55, 0x60, + 0xf7, 0xa2, 0x07, 0xec, 0x8b, 0x6d, 0x02, 0x86, 0x43, 0xf3, 0xc2, 0x2e, + 0x87, 0x27, 0xcc, 0x86, 0xf0, 0xe0, 0x26, 0x8f, 0x3d, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, +}; + +// AnnounceMeasure is the announcement's byte count, which is a constant of the +// unit and not a walk. +inline int64_t AnnounceMeasure() { return kTableAnnounceBytes; } + +// Announce writes the announcement into the caller's buffer and answers the +// bytes written — exactly AnnounceMeasure's answer — or -1 when the buffer is +// too small. It allocates nothing and walks nothing. +inline int64_t Announce( uint8_t * buffer, int64_t capacity ) +{ + if ( buffer == NULL || capacity < kTableAnnounceBytes ) { return -1; } + memcpy( buffer, kTableAnnounce, (size_t) kTableAnnounceBytes ); + return kTableAnnounceBytes; +} + +// TableVocabulary is ONE DIRECTION of ONE CONNECTION's id table (§3.3): the +// entries an announcement carried, whole, under one numbering with slot 1 the +// reserved build-version id. +// +// A peer holds TWO of these for a connection, the one it writes with and the +// one it reads with, and neither is the other's. A restart opens a fresh +// connection with empty tables and nothing is cached across connections, so +// its whole life is one connection's. It BORROWS the announcement's bytes rather than +// copying them, so a receiver holds one table a direction and its memory is +// the bound below and nothing else. +struct TableVocabulary +{ + // THE CONFORMING DEFAULT BOUND (§3.3): 32 KiB a direction, eight times the + // 500-id unit that is already a large one. A connection's table is bounded + // by nothing the wire carries, so the receiver declares the maximum and an + // announcement above it is refused by name before an entry is touched. + static const int64_t kDefaultMaxEntries = 4096; + + TableIdTable table; + uint64_t build_version = 0; + bool announced = false; + int64_t max_entries = kDefaultMaxEntries; +}; + +// AnnounceRead reads an announcement into one direction's table (§3.3). +// +// THE BOUND IS CHECKED BEFORE ANYTHING IS ALLOCATED: the entry count is a +// fixed little-endian u64 at the end, so a receiver reads it, compares it and +// refuses without touching an entry. After that it is §3's ordinary FILE read, +// because the announcement IS a file, with EXACTLY ONE STRICT CHECK over its +// body: the reserved build-version field present, exactly once, under kind 9, +// eight bytes wide. Everything else is an ordinary field under §4's tolerance, +// so an unknown one is skipped and counted and the announcement can GAIN a +// field in a later minor without a lockstep redeploy. +// +// The FIRST announcement sets the table and it is the only one that can. A +// SECOND is refused by name: it does not replace the table, it does not amend +// it and it changes nothing. A refused announcement sets NO TABLE. +inline bool AnnounceRead( TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + if ( vocabulary.announced ) + { + to->refused = true; + to->reason = second_announcement; + return false; + } + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireForm ) + { + to->refused = true; + to->reason = buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + return false; + } + if ( bytes < 9 ) { to->malformed = true; return false; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + if ( ( lo | ( hi << 32 ) ) > (uint64_t) vocabulary.max_entries ) + { + to->refused = true; + to->reason = vocabulary_too_large; + return false; + } + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else { to->refused = true; to->reason = newer_form; } + return false; + } + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) { to->malformed = true; return false; } + // the body, under §4's tolerance and this form's one strict check + TableReader r( buffer + 1, body_bytes, to, &table ); + uint64_t version = 0; + int32_t seen = 0; + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { to->malformed = true; return false; } + if ( ref == 0 ) { break; } + if ( ref > (uint64_t) table.count || !r.has( 1 ) ) { to->malformed = true; return false; } + const uint64_t id = table.at( ref ); + const uint8_t kind = r.get8(); + if ( id != kTableBuildVersionFieldId ) + { + to->unknown++; + if ( !r.skip( kind ) ) { to->malformed = true; return false; } + continue; + } + if ( kind != 9 || !r.has( 8 ) ) { to->refused = true; to->reason = no_vocabulary; return false; } + version = r.get64(); + seen++; + } + if ( seen != 1 ) { to->refused = true; to->reason = no_vocabulary; return false; } + vocabulary.table = table; + vocabulary.build_version = version; + vocabulary.announced = true; + return true; +} + +inline float table_bits_to_float( uint32_t bits ) { float f; memcpy( &f, &bits, 4 ); return f; } +inline uint32_t table_float_to_bits( float f ) { uint32_t b; memcpy( &b, &f, 4 ); return b; } +inline double table_bits_to_double( uint64_t bits ) { double d; memcpy( &d, &bits, 8 ); return d; } +inline uint64_t table_double_to_bits( double d ) { uint64_t b; memcpy( &b, &d, 8 ); return b; } + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +#ifndef LISTDEMO_SCHEMA_TABLE_ARENA +#define LISTDEMO_SCHEMA_TABLE_ARENA + +namespace listdemo { + +// ---- variable-length tables: tuning constants (docs/SPEC-TABLES.md) ---- +// +// The segment size and the count multiply to exactly 2^32: the u32 reference +// is the arena's hard ceiling, and these constants saturate it rather than +// leaving address space unreachable. Slab handout costs one atomic per slab, +// so per-node allocation costs no synchronization at all. + +static const uint32_t kTableSegmentBits = 22; // 4 MiB segments +static const uint32_t kTableSegmentSize = 1u << kTableSegmentBits; +static const uint32_t kTableSegmentMask = kTableSegmentSize - 1u; +static const uint32_t kTableMaxSegments = 1u << ( 32 - kTableSegmentBits ); // 1024 -> 4 GiB +static const uint32_t kTableSlabBytes = 64u * 1024u; // one atomic per slab +static const uint32_t kTableAlign = 8; // every node starts 8-aligned +static const uint32_t kTableAllocFailed = 0xFFFFFFFFu; + +// ---- THE CALLER'S ALLOCATOR (docs/SPEC-TABLES.md §6.5) ---- +// +// Every allocation the variable-length runtime makes goes through one of +// these — the arena's segments, the pack walk's identity map, the numbering's +// entry array, the packed region, and the tool path's node directory. There is +// no other call to the C library on this path, so a counting allocator sees +// every byte and a game's own heap can own all of it. +// +// It is the shape TableBlockAllocator already has (§19.1): two function +// pointers and a context the caller carries. What it adds is a CONTRACT ON +// alloc — the bytes come back ZEROED. Lock copies whole nodes, PADDING +// INCLUDED, so anything left uninitialized reaches a packed region; the default +// pair reaches that through calloc, which costs nothing measurable because a +// fresh segment is untouched pages either way. +struct TableAllocator +{ + void * ( *alloc )( void * context, int64_t bytes ); // ZEROED bytes, NULL on failure + void ( *free )( void * context, void * pointer ); + void * context; +}; + +// The default pair, and it is the one every entry point takes when the caller +// names none. It calls schema_allocate / schema_release, so a program with its +// own C-library replacement can move the floor without writing a struct at all. +inline void * table_default_alloc( void * context, int64_t bytes ) { (void) context; return schema_allocate( bytes ); } +inline void table_default_free( void * context, void * pointer ) { (void) context; schema_release( pointer ); } + +inline TableAllocator TableDefaultAllocator() +{ + TableAllocator allocator; + allocator.alloc = table_default_alloc; + allocator.free = table_default_free; + allocator.context = NULL; + return allocator; +} + +// ---- TableRef: a relocatable reference (never a machine pointer) ---- +// +// Two encodings, one slot, and the FORM says which is in force: +// +// in the arena — the node's arena offset (segment index in the high bits) +// in a region — the SELF-RELATIVE byte delta from this slot's own address, +// so a deref is one add, needs no base pointer, and a whole +// region relocates by memcpy with zero fix-up +// +// 0 is null in both, and a slot can never name the node that contains it, so +// zero names nothing real in either form. +// +// A REGION DELTA HAS NO REQUIRED SIGN (§6.3). A region is packed depth-first, +// so a node's FIRST reference points forward; every LATER reference to that +// same node points BACK at the one body it already has, which is exactly what +// makes one node one node in a region. Sharing and a back-reference are the +// same fact, and nothing validates a reference by its sign. +// +// IT IS EIGHT BYTES, SIGNED, so ONE REGION REACHES EVERYTHING (§6.3, §7): a +// four-byte slot bounded a region at 2 GiB, and the scale a cook exists for is +// *"100mbs or many gigabytes of data in Assets.bin"*. +struct TableRef +{ + int64_t value = 0; + bool null() const { return value == 0; } +}; + +// TableSlot is what Alloc hands back: usable as the node pointer (write +// fields through it) AND as the reference to store in a pointer field. +template struct TableSlot +{ + T * ptr = NULL; + TableRef ref; + T * operator->() const { return ptr; } + T & operator*() const { return *ptr; } + operator T *() const { return ptr; } + operator TableRef() const { return ref; } + bool null() const { return ptr == NULL; } +}; + +inline uint32_t TableAlignUp( uint32_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( kTableAlign - 1 ); } +inline int64_t TableAlignUp64( int64_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( int64_t( kTableAlign ) - 1 ); } + +// ---- a BYTE BUFFER's node (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// A *bytes or *string slot is a TableRef like every pointer slot, and it names +// a BLOB NODE: this eight-byte header and then the bytes, at offset eight so +// the data is eight-aligned. A *string blob carries one more zero byte after +// its data, so a region hands back a C string with no copy. The node's extent +// is the header plus its bytes, rounded to the arena's alignment like every +// node's; on the wire it is a record whose body is the bytes (§3.1). +struct TableBlob +{ + uint32_t length; + uint32_t zero; +}; + +static const int64_t kTableBlobHeader = 8; // length (u32), then four zero bytes +static const int64_t kTableBlobMaxLength = 0xFFFFFFFF; // a record's length is a u32 (§3.1) + +// the node's storage: the header, the bytes, a string's terminator, rounded +// to the arena's alignment like every node +inline int64_t TableBlobStorage( int64_t length, bool terminated ) +{ + return TableAlignUp64( kTableBlobHeader + length + ( terminated ? 1 : 0 ) ); +} + +// What a read answers: a pointer INTO the region and the length, NULL and +// zero for a null slot. Off a locked region, a loaded one or an opened cook +// the pointer is one add from the slot, and nothing is copied. +struct TableBytesView +{ + const uint8_t * data; + int64_t length; +}; + +struct TableStringView +{ + const char * data; // zero-terminated + int64_t length; +}; + +// What AllocBytes and AllocString hand back: the bytes to write through, the +// length asked for, and the reference to store in the slot — the three +// answers TableSlot gives for a table node. +struct TableBytesSlot +{ + uint8_t * data = NULL; + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +struct TableStringSlot +{ + char * data = NULL; // room for length bytes and the terminator, already zero + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +// ---- the arena: segmented, slab-handed, lock-free by ownership ---- +// +// Allocation is thread-local inside a worker's slab — no atomics on the node +// path. A worker takes its next slab with ONE compare-exchange, and a new +// segment is published with one more. Nothing ever moves: a segment, once +// allocated, lives untouched until the arena is torn down, so a T* obtained +// from Alloc stays valid while other workers allocate, and an offset stays +// correct while the arena grows. +// +// The model this DELIBERATELY refuses: one buffer under a lock, grown by +// realloc. A realloc moves the buffer under workers mid-write; offsets fix +// identity but not the raw references already resolved from them, and the +// resulting corruption is invisible until much later. Segments never move, so +// that bug class cannot be written here. +// +// Slack: at most one slab tail per worker plus one slab per segment (a slab +// that will not fit is skipped rather than split), i.e. under 2% of a segment +// plus threads x 64 KiB. That is the price of never synchronizing per node. +struct TableArena +{ + std::atomic segments[ kTableMaxSegments ]; + std::atomic cursor; // (segment << kTableSegmentBits) | bytes handed out + bool locked = false; // MONOTONIC: Lock() is one-way, there is no unlock + // THE ARENA CARRIES ITS OWN, so everything downstream of a builder — + // segments, pack map, numbering, region, node directory — allocates through + // the one pair the caller named, with nothing to thread by hand. + TableAllocator allocator; +}; + +inline void TableArenaInit( TableArena & arena, TableAllocator allocator ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + arena.segments[i].store( NULL, std::memory_order_relaxed ); + } + arena.cursor.store( 0, std::memory_order_relaxed ); + arena.locked = false; + arena.allocator = allocator; +} + +inline void TableArenaShutdown( TableArena & arena ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + uint8_t * segment = arena.segments[i].exchange( NULL, std::memory_order_acq_rel ); + if ( segment != NULL ) { arena.allocator.free( arena.allocator.context, segment ); } + } + arena.cursor.store( 0, std::memory_order_relaxed ); +} + +// one L1 load plus an add: the segment table is 8 KiB and stays hot +inline uint8_t * TableArenaAt( const TableArena & arena, uint32_t offset ) +{ + return arena.segments[ offset >> kTableSegmentBits ].load( std::memory_order_relaxed ) + ( offset & kTableSegmentMask ); +} + +// TableArenaGrabSlab hands one worker its next private slab. Returns +// kTableAllocFailed when the arena's address space or the allocator is +// exhausted — a loud refusal, never a silent smaller slab. +inline uint32_t TableArenaGrabSlab( TableArena & arena ) +{ + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t segment = cursor >> kTableSegmentBits; + uint32_t used = cursor & kTableSegmentMask; + // strictly less: a slab is never split across segments, and the tail + // is the documented slack + if ( used + kTableSlabBytes < kTableSegmentSize ) + { + if ( arena.segments[segment].load( std::memory_order_acquire ) == NULL ) + { + // THE SEGMENT COMES BACK ZEROED, which is the allocator's + // contract and not an extra pass here: Lock copies whole nodes, + // PADDING INCLUDED, so anything uninitialized reaches a packed + // region. Value-initializing a node with placement new zeroes + // its MEMBERS and not its padding, so the zeroing has to happen + // at the segment or not at all. It costs nothing measurable: a + // fresh segment is untouched pages either way, and the default + // pair's calloc has the kernel hand them over zeroed. + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, (int64_t) kTableSegmentSize ); + if ( memory == NULL ) { return kTableAllocFailed; } + uint8_t * expected = NULL; + if ( !arena.segments[segment].compare_exchange_strong( expected, memory, std::memory_order_acq_rel ) ) + { + // another worker published this segment first + arena.allocator.free( arena.allocator.context, memory ); + } + } + if ( arena.cursor.compare_exchange_weak( cursor, cursor + kTableSlabBytes, std::memory_order_acq_rel ) ) + { + return ( segment << kTableSegmentBits ) | used; + } + continue; + } + uint32_t next_segment = segment + 1; + if ( next_segment >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + arena.cursor.compare_exchange_weak( cursor, next_segment << kTableSegmentBits, std::memory_order_acq_rel ); + } +} + +// TableArenaGrabSpan reserves a SPAN of the arena's address space for one node +// larger than a slab — a BYTE BUFFER of any size (docs/SPEC-TABLES.md §2.5) — +// and allocates it as one contiguous block. It takes whole segment indices +// from the cursor, starting at the index after the cursor's so nothing else +// is ever handed out inside the span, and publishes the block under the first +// of them; the indices the span covers past that one stay NULL, which is +// enough, because only a node's START is ever resolved through the segment +// table and a blob's bytes follow its header inside the one allocation. The +// unused tail of the segment the cursor was in is slack, like a slab tail. +// Returns kTableAllocFailed when the address space or the allocator is +// exhausted — a loud refusal, never a smaller blob. +inline uint32_t TableArenaGrabSpan( TableArena & arena, int64_t bytes ) +{ + if ( bytes <= 0 || bytes > ( (int64_t) kTableMaxSegments - 2 ) * (int64_t) kTableSegmentSize ) { return kTableAllocFailed; } + const uint32_t spanned = (uint32_t) ( ( bytes + kTableSegmentSize - 1 ) >> kTableSegmentBits ); + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t start = ( cursor >> kTableSegmentBits ) + 1; + if ( start + spanned >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + uint32_t next = ( start + spanned ) << kTableSegmentBits; + if ( !arena.cursor.compare_exchange_weak( cursor, next, std::memory_order_acq_rel ) ) { continue; } + // the span is this worker's now: nothing else can publish under its + // first index, so a plain store suffices, and the block comes back + // ZEROED like every segment — the blob's bytes and its tail are zeros + // until written + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, bytes ); + if ( memory == NULL ) { return kTableAllocFailed; } + arena.segments[start].store( memory, std::memory_order_release ); + return start << kTableSegmentBits; + } +} + +// ---- TableWorker: one thread's allocation front ---- +// +// The threading contract, stated plainly: +// * Alloc on YOUR OWN worker is safe concurrently with any other worker's. +// No locks, no atomics per node. +// * Writing fields of a node ANOTHER worker allocated is your own +// synchronization problem — this runtime does not arbitrate it. +// * Lock and Save are single-threaded: call them after the workers have +// joined. +struct TableWorker +{ + TableArena * arena = NULL; + uint32_t next = 0; + uint32_t end = 0; + + template TableSlot Alloc() + { + static_assert( alignof( T ) <= kTableAlign, "a table node's alignment must fit the arena's" ); + TableSlot slot; + if ( arena == NULL || arena->locked ) { return slot; } + uint32_t bytes = TableAlignUp( (uint32_t) sizeof( T ) ); + if ( bytes > kTableSlabBytes ) { return slot; } // a node larger than a slab: refused, never split + if ( end == 0 || next + bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return slot; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + uint32_t at = next; + next += bytes; + // A NODE IS BORN IN TWO HALVES: start its lifetime in the raw + // storage, then write the declared defaults ONE MEMBER AT A TIME. + // + // It is "T", not "T{}". Value-initialising the whole aggregate says + // the same thing and costs cl O(BYTES) TO COMPILE — it expands element + // by element in its front end — while both halves here cost + // O(declarations). The slab cap below refuses a large node at RUN + // TIME and bounds nothing at compile time: the cost is paid by + // whatever T a caller instantiates this with. + // Padding is not the difference: value-initialisation zeroes MEMBERS + // and not padding either way, which is why the segment is calloc'd. + // + // TableReset is an OVERLOAD SET, one per closure member, reached from + // this template by argument-dependent lookup on T's own namespace — + // Alloc is a template and cannot spell Reset. + // + // The reset is here because ONE DEFINITION SAYS WHAT THE DECLARED + // DEFAULTS ARE, and it is Reset. Default-initialisation lands on + // the same values today, because a member with a non-zero default + // carries a member initializer that says so — but that is the class + // definition agreeing with Reset, not the arena reading it, and #320's + // fix was itself a pass that MOVED initialisation between the two. + // The arena reads the definition. + slot.ptr = new ( TableArenaAt( *arena, at ) ) T; + TableReset( *slot.ptr ); + slot.ref.value = at; + return slot; + } + + // Alloc a BYTE BUFFER's node of exactly length bytes (docs/SPEC-TABLES.md + // §2.5): the blob header and its bytes, zeroed, in this thread's slab when + // it fits and in a span of the arena's own when it does not. NULL is the + // arena locked, a length below zero or past a record's u32, or the + // allocator refusing. The offset comes back for the reference. + TableBlob * AllocBlob( int64_t length, bool terminated, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( length < 0 || length > kTableBlobMaxLength ) { return NULL; } + const int64_t bytes = TableBlobStorage( length, terminated ); + if ( bytes > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, bytes ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + } + else + { + if ( end == 0 || next + (uint32_t) bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) bytes; + } + TableBlob * blob = (TableBlob *) TableArenaAt( *arena, at ); + blob->length = (uint32_t) length; // the bytes after it are the segment's zeros + blob->zero = 0; + return blob; + } + + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries + // no type id, takes no index and has no Reset, so it goes through the same + // slab and span the blob path uses rather than through Alloc. + uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( bytes <= 0 || align > (int64_t) kTableAlign ) { return NULL; } + const int64_t rounded = TableAlignUp64( bytes ); + if ( rounded > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, rounded ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + return TableArenaAt( *arena, at ); + } + if ( end == 0 || next + (uint32_t) rounded > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) rounded; + return TableArenaAt( *arena, at ); // the segment came back zeroed + } + // a *bytes node: the bytes to write through, and the reference to store + TableBytesSlot AllocBytes( int64_t length ) + { + TableBytesSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, false, at ); + if ( blob == NULL ) { return slot; } + slot.data = (uint8_t *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } + + // a *string node: room for length bytes and the zero byte after them + TableStringSlot AllocString( int64_t length ) + { + TableStringSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, true, at ); + if ( blob == NULL ) { return slot; } + slot.data = (char *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } +}; + +// ---- TablePackMap: the pack walk's identity map (docs/SPEC-TABLES.md §3.1, §6.2) ---- +// +// ONE ENTRY PER REACHABLE NODE, and that map IS identity: a node must know +// where it landed to be named a second time, so Lock packs a shared node ONCE +// and every later reference resolves to the one body it already has. That is +// the same first-visit numbering the wire uses, so the pack order and the node +// order are one order. +// +// COLOURING AN ENTRY WHILE ITS DESCENT IS OPEN COSTS ONE BIT, and it is what +// makes a data cycle free to refuse: a reference to an entry still open is a +// cycle, and Lock returns failure rather than recursing away. The ROOT's entry +// is open for the whole walk. +// +// The map is proportional to NODES, never to bytes, and it lives on the +// AUTHORING side, where §6.5 licenses allocation. Nothing on the reading path +// ever builds one. +struct TablePackEntry +{ + const void * key; // the node's address in the graph being packed + int64_t offset; // where that node landed in the region + uint8_t open; // its descent is still open: a reference here is a cycle +}; + +struct TablePackMap +{ + TablePackEntry * entries = NULL; + int64_t capacity = 0; // a power of two, or zero while empty + int64_t count = 0; + TableAllocator allocator; // the caller's, carried from the walk that built it +}; + +inline void TablePackMapInit( TablePackMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TablePackMapShutdown( TablePackMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TablePackMapInit( map, map.allocator ); +} + +// The two walks behind Lock re-derive the SAME map from the same graph — the +// numbering is never carried between them (§3.1) — so the second starts from +// an empty map and keeps the capacity the first paid for. +inline void TablePackMapReset( TablePackMap & map ) +{ + if ( map.entries != NULL ) { memset( map.entries, 0, (size_t) map.capacity * sizeof( TablePackEntry ) ); } + map.count = 0; +} + +// open addressing, linear probing, a multiply-shift hash over the address: a +// node key is a pointer and its low bits are alignment, so the low bits alone +// would collide on every node of one type +inline int64_t TablePackMapSlot( const TablePackMap & map, const void * key ) +{ + uint64_t hash = (uint64_t) (uintptr_t) key; + hash *= 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != NULL && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TablePackEntry * TablePackMapFind( TablePackMap & map, const void * key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TablePackEntry * entry = &map.entries[ TablePackMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +// QUADRUPLING, not doubling, and the reason is measured: growth rehashes every +// entry, and on a graph of 131,071 nodes the doubling schedule spent 45% of +// Lock in rehashing alone. Quadrupling from 1024 buys 1.35x on that graph and +// keeps the map NODE-proportional (§6.2) — under 128 bytes a node at its +// worst, right after a grow, and about 64 on average. +inline bool TablePackMapGrow( TablePackMap & map ) +{ + TablePackMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 1024; + grown.entries = (TablePackEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TablePackEntry ) ); + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == NULL ) { continue; } + grown.entries[ TablePackMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// REACH a node: one probe answers both questions the walk has. A true "taken" +// says this is a FIRST visit, and the entry is now the node's, coloured open +// at "offset"; otherwise the entry is the one the node already has, and its +// open bit says cycle or sharing. NULL is an allocation failure, and it is a +// refusal like any other: Lock fails rather than packing a graph it cannot +// track. +// +// It is one call and not a find followed by an insert because the walk asks +// this question twice per node — once to measure, once to pack — and every +// probe is a miss into a table larger than L2. +inline TablePackEntry * TablePackMapReach( TablePackMap & map, const void * key, int64_t offset, bool & taken, int64_t & slot ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TablePackMapGrow( map ) ) { return NULL; } + } + slot = TablePackMapSlot( map, key ); + TablePackEntry * entry = &map.entries[slot]; + taken = entry->key != key; // an empty slot is a first visit; the key is never NULL + if ( taken ) + { + entry->key = key; + entry->offset = offset; + entry->open = 1; + map.count++; + } + return entry; +} + +// The descent finished: the node keeps its entry — identity outlives the +// descent — and stops being a cycle. The "hint" is the slot Reach returned, and it +// is checked against the key rather than trusted, so a rehash between the two +// costs a second probe instead of correctness. +inline void TablePackMapClose( TablePackMap & map, const void * key, int64_t hint ) +{ + if ( hint >= 0 && hint < map.capacity && map.entries[hint].key == key ) + { + map.entries[hint].open = 0; + return; + } + TablePackEntry * entry = TablePackMapFind( map, key ); + if ( entry != NULL ) { entry->open = 0; } +} + +// ---- resolution contexts: which encoding a walk is reading ---- + +struct TableArenaCtx { const TableArena * arena; }; +struct TableRegionCtx {}; + +// ---- a BYTE BUFFER's resolution (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// The same two encodings a table pointer has, resolved the same way: a +// self-relative delta in a region — one add, no base — and an arena offset +// while the builder is mutable. The blob is reached through its header, and a +// view is the header plus eight and the header's first word. Nothing here +// allocates and nothing copies: off a locked region, a loaded one or an +// opened cook the view points INTO the region. +inline const TableBlob * TableBlobAt( const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableRegionCtx &, const TableRef & ref ) { return TableBlobAt( ref ); } +inline const TableBlob * TableBlobAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} + +inline TableBytesView TableBytesViewOf( const TableBlob * blob ) +{ + TableBytesView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const uint8_t *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} +inline TableStringView TableStringViewOf( const TableBlob * blob ) +{ + TableStringView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const char *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} + +// the const form's hot path: one add, no base +inline TableBytesView TableBytesAt( const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ref ) ); } +inline TableStringView TableStringAt( const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ref ) ); } +// and the context forms a walk uses: a region context, an arena context, or +// the arena itself while the builder is mutable +template inline TableBytesView TableBytesAt( const Ctx & ctx, const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ctx, ref ) ); } +template inline TableStringView TableStringAt( const Ctx & ctx, const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ctx, ref ) ); } + +// allocate a blob in the arena and point the slot at it; the slot holds the +// arena offset, as every slot does while the builder is mutable +inline uint8_t * TableBytesEmplace( TableWorker & worker, TableRef & slot, int64_t length ) +{ + TableBytesSlot allocated = worker.AllocBytes( length ); + slot = allocated.ref; + return allocated.data; +} +// the text is copied in when one is given; a NULL text leaves the zeros for +// the caller to fill +inline char * TableStringEmplace( TableWorker & worker, TableRef & slot, const char * text, int64_t length ) +{ + TableStringSlot allocated = worker.AllocString( length ); + slot = allocated.ref; + if ( allocated.data != NULL && text != NULL && length > 0 ) { memcpy( allocated.data, text, (size_t) length ); } + return allocated.data; +} + +// ---- the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table, and a +// pointer field rides as an INDEX into it under kind 17. The encoding is +// flat: no pointer edge is a nesting level, so a chain's length is not a depth, +// and two references to one node are one node. +// +// THE FIELD RIDES ONCE: an L with sixty-four bits of capability frames a +// numbering of any size, so the whole numbering is one contiguous payload and a +// save's node bodies have no aggregate ceiling. + +static const uint64_t kTableNodeIndexNull = 0; // absence and null are one value +static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts the table + +// The not-materialized sentinel (§6.3): a record whose type id this build could +// not name. Distinct from every real offset including the root's 0, so an index +// resolving through it yields NULL and can never fabricate the root. +static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; + +// ---- the numbering, on the SAVE side ---- +// +// One entry per reachable node in FIRST-VISIT order, so entry k is node index +// k + 2. The two thunks are what let one loop write a table of mixed types: the +// numbering walk knows each target's type STATICALLY at the site it numbers it, +// so it stores the instantiation there and the loop never asks what a node is. +struct TableNumbering; + +struct TableNodeEntry +{ + const void * node; + uint64_t type_id; + // the type id's MESSAGE-FORM SLOT (docs/SPEC-TABLES.md §3.3), stored where + // the numbering walk stores the id itself and for the same reason: the + // target's type is known STATICALLY at the site that numbers it, so a + // form 2 save reads the slot out of the entry instead of looking an id up. + // Every pointer target's type id is an entry of the announcement, which is + // what makes the slot a compile-time fact of a POINTERED message too. + uint64_t type_slot; + int64_t ( * measure )( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ); + bool ( * save )( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ); +}; + +struct TableNumbering +{ + TablePackMap seen; // node -> index; the ROOT is index 1, open for the whole walk + TableNodeEntry * entries = NULL; + int64_t count = 0; + int64_t capacity = 0; +}; + +// The numbering allocates through the map's pair rather than carrying a second +// copy of it: one numbering is one walk, and a walk has one allocator. +inline void TableNumberingInit( TableNumbering & n, TableAllocator allocator ) +{ + TablePackMapInit( n.seen, allocator ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +inline void TableNumberingShutdown( TableNumbering & n ) +{ + TableAllocator allocator = n.seen.allocator; + TablePackMapShutdown( n.seen ); + allocator.free( allocator.context, n.entries ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +// The index a numbered node was given, for the save that writes it into a +// pointer slot. False means the two walks disagree about the graph, which is a +// refusal and never a guess. +inline bool TableNumberingIndex( const TableNumbering & n, const void * node, uint64_t & index ) +{ + if ( n.seen.capacity == 0 ) { return false; } + const TablePackEntry & entry = n.seen.entries[ TablePackMapSlot( n.seen, node ) ]; + if ( entry.key != node ) { return false; } + index = (uint64_t) entry.offset; + return true; +} + +inline bool TableNumberingAppend( TableNumbering & n, const TableNodeEntry & entry ) +{ + if ( n.count == n.capacity ) + { + // GROW BY COPY, never by realloc: the allocator hook is a PAIR, and a + // game's heap is not required to have a resize primitive at all. The + // schedule quadruples, so the copying is amortized to a constant per + // entry and the growth is the same growth it always was. + int64_t capacity = n.capacity != 0 ? n.capacity * 4 : 256; + TableAllocator allocator = n.seen.allocator; + TableNodeEntry * grown = (TableNodeEntry *) allocator.alloc( allocator.context, capacity * (int64_t) sizeof( TableNodeEntry ) ); + if ( grown == NULL ) { return false; } + if ( n.entries != NULL ) + { + memcpy( grown, n.entries, (size_t) n.count * sizeof( TableNodeEntry ) ); + allocator.free( allocator.context, n.entries ); + } + n.entries = grown; + n.capacity = capacity; + } + n.entries[n.count++] = entry; + return true; +} + +// The thunks the numbering stores. Each resolves to the closure member's own +// MeasureBody / SaveBodyFields through an overload set in the member's DECLARING +// file, reached by argument-dependent lookup at instantiation — the same bridge +// the arena's TableReset uses, and the reason a numbering may span the files of +// one unit without any file naming another's members. +template +inline int64_t TableNodeMeasureThunk( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ) +{ + return TableNodeMeasure( *(const Ctx *) ctx, numbering, ids, *(const T *) node ); +} + +template +inline bool TableNodeSaveThunk( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ) +{ + return TableNodeSave( *(const Ctx *) ctx, numbering, w, ids, *(const T *) node ); +} + +// ---- a BYTE BUFFER's record (docs/SPEC-TABLES.md §2.5, §3.1) ---- +// +// A blob rides as a node record under one of two RESERVED type ids — the fold +// a table's name takes, over the keywords "bytes" and "string", which no table +// can be named — with the bytes as its body and nothing framed inside. These +// two thunks are what the numbering stores for a blob, as it stores a +// member's codec for a table: the length, and the bytes verbatim. +static const uint64_t kTableBytesTypeId = 0x2f2ec0474f1c4fe4ull; // fnv1a64( "bytes" ) +static const uint64_t kTableStringTypeId = 0x704be0d8faaffc58ull; // fnv1a64( "string" ) + +template +inline int64_t TableBlobMeasureThunk( const void *, const TableNumbering &, TableIds &, const void * node ) +{ + return (int64_t) ( (const TableBlob *) node )->length; +} + +template +inline bool TableBlobSaveThunk( const void *, const TableNumbering &, TableWriter & w, TableIds &, const void * node ) +{ + const TableBlob * blob = (const TableBlob *) node; + w.raw( (const void *) ( blob + 1 ), (int64_t) blob->length ); + return true; +} + +// TableNodeTableMeasure and TableNodeTableSave are the framing, and they are +// ONE fill rule written twice — measure derives it from the graph and save +// derives the same one, which is what makes measure == save hold across a +// pointer graph (§3.1). +// +// The field rides ONCE, under the reserved id, kind 12: the payload opens with +// the count and then carries the records back to back, each a type id +// REFERENCE, a length and a body. The reserved id is interned BEFORE the +// records, and a record's type id before its body, which is the first-use order +// the trailer is written in (§3). +template +inline int64_t TableNodeTablePayload( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + int64_t payload = TableLebBytes( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + payload += TableLebBytes( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return -1; } + payload += TableLebBytes( (uint64_t) body ) + body; + } + return payload; +} + +template +inline int64_t TableNodeTableMeasure( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return 0; } // a root that reaches no nodes writes none of them + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return -1; } + return TableLebBytes( ref ) + 1 + TableLebBytes( (uint64_t) payload ) + payload; +} + +template +inline bool TableNodeTableSave( const Ctx & ctx, TableWriter & w, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return true; } + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return false; } + w.putleb( ref ); + w.put8( 12 ); // kind 12 is the opaque byte payload: a reader that cannot name the id skips by L + w.putleb( (uint64_t) payload ); + w.putleb( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + w.putleb( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return false; } + w.putleb( (uint64_t) body ); + if ( !n.entries[k].save( (const void *) &ctx, n, w, ids, n.entries[k].node ) ) { return false; } + } + return true; +} + +// ---- the numbering, on the LOAD side: a region's NODE DIRECTORY (§6.3) ---- +// +// The wire's numbering made resident: one entry per numbered node, in index +// order, position i describing node index i + 1 — so position 0 is the ROOT at +// offset 0. It is ATTRIBUTION, and attribution is separable: nothing that reads +// a structure touches it, a deref is one add on a self-relative offset, and a +// caller may release it once Load returns. +struct TableNodeDirEntry +{ + uint64_t offset; + uint64_t type_id; +}; + +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; + +// TableNodeMap is what a pointer slot resolves through while a body decodes. +struct TableNodeMap +{ + uint8_t * base = NULL; + const TableNodeDirEntry * entries = NULL; + int64_t count = 0; // the ROOT's entry included, so it is records + 1 + bool good = false; // the node table read whole; a numbering that failed resolves nothing + // WHERE THE NODES LIVE, and therefore what a resolved slot holds: a region + // takes the SELF-RELATIVE delta so a deref is one add, and the tool's + // builder path takes the node's ARENA OFFSET (§6.3). + bool arena = false; + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. + TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; +}; + +// TableNodeResolve places one node index in a pointer slot, and every failure +// is one of §4's events with the pointer left null. The declared TARGET type id +// is checked at every index, the root's included: the root carries no record +// and therefore no wire type id, so the READER'S OWN root type is what the +// claim is checked against. +inline void TableNodeResolve( const TableNodeMap & map, TableRef & slot, uint64_t index, uint64_t target, TableReport * report ) +{ + slot.value = 0; + if ( index == kTableNodeIndexNull || !map.good ) { return; } + if ( index - 1 >= (uint64_t) map.count ) + { + report->malformed = true; // an index above node_count + 1 + return; + } + const TableNodeDirEntry & entry = map.entries[index - 1]; + if ( entry.offset == kTableNodeAbsent ) + { + // a node whose type id this build could not name KEEPS ITS INDEX, and + // every pointer naming it reads null. The unknown was counted once, at + // the node, not once per pointer. + return; + } + if ( entry.type_id != target ) + { + report->kind_mismatch++; + return; + } + slot.value = map.arena ? (int64_t) entry.offset + : (int64_t) ( ( map.base + entry.offset ) - (const uint8_t *) &slot ); +} + +// ---- the record SCAN, and it is the whole of load's bound (§3.1) ---- +// +// Reading follows no reference. The scan walks the root body's top-level fields, +// finds the ONE under the reserved id, and reads records out of its payload in +// order — the field rides once, so nothing is copied to make a body contiguous +// and the generated body decoder never learns the transport exists. +struct TableNodeScan +{ + TableReader fields; // over the ROOT body, skipping past everything else + const uint8_t * payload; // the node-table field's payload + int64_t payload_size; + int64_t payload_offset; + bool opened; // the root body has been walked for the field + uint64_t declared; + int64_t records; + bool present; // the root body carries a node table at all + bool malformed; + const TableIdTable * ids; +}; + +inline TableNodeScan TableNodeScanBegin( const uint8_t * body, int64_t size, TableReport * report, const TableIdTable * ids ) +{ + TableNodeScan s = { TableReader( body, size, report, ids ), NULL, 0, 0, false, 0, 0, false, false, ids }; + return s; +} + +// find the node-table field, or answer false when the root body has none. A +// body carrying an id more than once is legal input and THE LAST OCCURRENCE +// WINS (docs/SPEC-TABLES.md §3), so the walk runs to the terminator and keeps +// the last rather than stopping at the first. +inline bool TableNodeScanOpen( TableNodeScan & s ) +{ + if ( s.opened ) { return false; } + s.opened = true; + for ( ;; ) + { + uint64_t ref = 0; + if ( !s.fields.getleb( ref ) ) { break; } + if ( ref == 0 ) { break; } // the terminator + if ( s.ids == NULL || ref > (uint64_t) s.ids->count ) { break; } + const uint64_t id = s.ids->at( ref ); + if ( !s.fields.has( 1 ) ) { break; } + const uint8_t kind = s.fields.get8(); + if ( id == kTableNodeTableFieldId ) + { + s.present = true; + if ( kind != 12 ) { s.malformed = true; return false; } + uint64_t length = 0; + if ( !s.fields.getleb( length ) || !s.fields.room( length ) ) { s.malformed = true; return false; } + s.payload = s.fields.buffer + s.fields.offset; + s.payload_size = (int64_t) length; + s.fields.offset += (int64_t) length; + continue; + } + if ( !s.fields.skip( kind ) ) { break; } + } + if ( s.payload == NULL ) { return false; } + TableReader head( s.payload, s.payload_size, s.fields.report, s.ids ); + if ( !head.getleb( s.declared ) ) { s.malformed = true; return false; } + s.payload_offset = head.offset; + return true; +} + +// the next record, or false at the end of the table — s.malformed says whether +// the end was the end or the framing giving out +inline bool TableNodeScanNext( TableNodeScan & s, uint64_t & type_id, const uint8_t * & body, int64_t & length ) +{ + if ( !s.opened && !TableNodeScanOpen( s ) ) { return false; } + if ( s.payload == NULL || s.payload_offset >= s.payload_size ) { return false; } + TableReader rec( s.payload, s.payload_size, s.fields.report, s.ids ); + rec.offset = s.payload_offset; + uint64_t ref = 0; + if ( !rec.getleb( ref ) || ref == 0 || s.ids == NULL || ref > (uint64_t) s.ids->count ) + { + s.malformed = true; // a type id reference of 0, or one past the table + return false; + } + type_id = s.ids->at( ref ); + uint64_t declared_length = 0; + if ( !rec.getleb( declared_length ) ) + { + s.malformed = true; // a record whose length is damaged + return false; + } + if ( declared_length > (uint64_t) ( s.payload_size - rec.offset ) ) + { + s.malformed = true; // a record whose length runs past its field + return false; + } + body = s.payload + rec.offset; + length = (int64_t) declared_length; + s.payload_offset = rec.offset + length; + s.records++; + return true; +} + +// The record scan is AUTHORITATIVE: node_count is data from the wire, and a +// count that disagrees with the scan is malformed. Nothing is sized from it +// before the scan has confirmed it. +inline bool TableNodeScanWhole( TableNodeScan & s ) +{ + if ( s.malformed ) { return false; } + if ( !s.present ) { return true; } // no node table at all is not a broken one + return s.declared == (uint64_t) s.records; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_ARENA + +#ifndef LISTDEMO_SCHEMA_TABLE_EXTENT +#define LISTDEMO_SCHEMA_TABLE_EXTENT + +namespace listdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_EXTENT + +#ifndef LISTDEMO_SCHEMA_TABLE_MAP +#define LISTDEMO_SCHEMA_TABLE_MAP + +namespace listdemo { + +// ---- a MAP: a sorted entry array, and the lookup over it (§2.8) ---- +// +// On the wire, in a region and in a cook a map is an array of one generated +// ENTRY table held in ascending key order. What this adds is Find — a binary +// search over that array where it lies — and a builder that inserts, replaces +// and erases by key. Nothing here is stored: a region and a cook carry the +// array and the count, and not one byte about a hash or a probe. + +// entries carved from ONE call to the allocator pair; a new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableMapSegmentEntries = 32; + +// TableDeclRef names a type in an unevaluated context and is never defined — +// what 's declval is for, without the include the generated corpus +// refuses to pay for (the iterator_traits note, §13.9). +template T & TableDeclRef(); + +// THE ORDER IS TOTAL, AND IT IS THE SAME IN NINE LANGUAGES (§2.8). Integers +// compare by VALUE, signed for the signed kinds and unsigned for the unsigned. +// Strings compare by BYTES, unsigned, a shorter string that is a prefix of a +// longer one first: memcmp over the common length, then the lengths. Never a +// locale, never a code point, never a case fold. +inline int TableKeyOrder( uint64_t a, uint64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( int64_t a, int64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( const char * a, int32_t a_length, const char * b, int32_t b_length ) +{ + const int32_t common = a_length < b_length ? a_length : b_length; + if ( common > 0 ) + { + const int order = memcmp( (const void *) a, (const void *) b, (size_t) common ); + if ( order != 0 ) { return order < 0 ? -1 : 1; } + } + return a_length < b_length ? -1 : ( a_length > b_length ? 1 : 0 ); +} + +// the length of a NUL-terminated key at a call site, bounded by the storage it +// has to fit: a key one byte longer than the bound is refused, never truncated +inline int32_t TableKeyLength( const char * key, int32_t bound ) +{ + if ( key == NULL ) { return 0; } + for ( int32_t i = 0; i <= bound; i++ ) { if ( key[i] == 0 ) { return i; } } + return bound + 1; // longer than the bound: the caller refuses it +} + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.8, §7.2) ---- +// +// An int64 self-relative reference to the entry array and an int32 count, then +// padding to eight. The reference is a TableRef like a pointer's: in the arena +// it names the builder's HEAD, in a region it is the delta from the slot to +// the first entry, and 0 is the empty map in both. +template struct TableMap +{ + TableRef entries; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Entry * Entries() const + { + return entries.value != 0 ? (const Entry *) ( (const uint8_t *) &entries + entries.value ) : NULL; + } + int32_t size() const { return count; } + + // FIND: floor( log2 n ) + 1 key compares, in place, no allocation. NULL + // when absent, and on a map[K]*T the RESOLVED pointer, which is what a + // pointer field's accessor answers. + template const Entry * FindEntry( Key key ) const + { + const Entry * base = Entries(); + int32_t low = 0, high = count; + while ( low < high ) + { + const int32_t mid = low + ( high - low ) / 2; + const int order = TableEntryOrder( base[mid], key ); + if ( order == 0 ) { return base + mid; } + if ( order < 0 ) { low = mid + 1; } else { high = mid; } + } + return NULL; + } + // the return type is DEDUCED, so it is worked out when a call site + // instantiates Find and not when the holder's record declares the slot — + // which is what lets the entry's own overloads be declared after it + template auto Find( Key key ) const + { + return TableEntryFound( FindEntry( key ) ); + } + + // ---- iteration: ASCENDING key order, the key beside the value ---- + // + // A proxy BY VALUE, the keyed array's shape (§2.4): for ( auto [ key, + // value ] : map ). It carries no iterator_traits, for the reason + // TableKeyed's does not (§13.9). + struct ConstEntry + { + decltype( TableEntryKey( TableDeclRef() ) ) key; + decltype( TableEntryFound( (const Entry *) NULL ) ) value; + }; + + struct ConstIterator + { + const Entry * at; + ConstEntry operator*() const { return ConstEntry{ TableEntryKey( *at ), TableEntryFound( at ) }; } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Entries() }; } + ConstIterator end() const { return ConstIterator{ Entries() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.8, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first entry is inserted. Each +// segment is a fixed number of entries carved from one call to the allocator +// pair. An entry's address is stable for the arena's life, so a value handed +// back by an insert stays valid while other entries arrive. +struct TableMapHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an insert appends into + int32_t live; + int32_t dead; +}; + +template struct TableMapSegment +{ + TableRef next; + int32_t used; // entries carved from this segment + int32_t padding; + uint32_t dead[ ( kTableMapSegmentEntries + 31 ) / 32 ]; // Erase marks one bit, never the entry + Entry entries[ kTableMapSegmentEntries ]; +}; + +inline bool TableMapSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// ---- the ORDERED CURSOR the four writing walks read (§2.8) ---- +// +// Measure, Save, Lock and Cook each write a map's entries in ascending key +// order with no key twice, deriving the order from the builder's entries as +// each walk derives the numbering (§3.1). Nothing passes between them, so +// measure == save over a map is a real check on two sorts agreeing. +// +// A REGION is already sorted, so its cursor is the array in place and +// allocates nothing. The BUILDER's is the sort: an array of entry pointers +// allocated through the pair and released before the walk returns, because +// sorting the segments themselves would move entries whose addresses a caller +// holds. +template struct TableMapCursor +{ + const Entry * const * order = NULL; // the builder's form: sorted pointers + const Entry * entries = NULL; // the region's form: the array in place + int32_t count = 0; + TableAllocator allocator; + bool ok = false; + const Entry * operator[]( int32_t index ) const + { + return order != NULL ? order[index] : entries + index; + } +}; + +// heapsort: O( n log n ) once per map, no recursion, no allocation past the +// pointer array the caller already paid for +template inline void TableMapSort( const Entry ** order, int32_t count ) +{ + for ( int32_t start = count / 2 - 1; start >= 0; start-- ) + { + int32_t root = start; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= count ) { break; } + if ( child + 1 < count && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * swap = order[root]; order[root] = order[child]; order[child] = swap; + root = child; + } + } + for ( int32_t end = count - 1; end > 0; end-- ) + { + const Entry * swap = order[0]; order[0] = order[end]; order[end] = swap; + int32_t root = 0; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= end ) { break; } + if ( child + 1 < end && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * hold = order[root]; order[root] = order[child]; order[child] = hold; + root = child; + } + } +} + +// the REGION form: the array is already sorted, so the cursor is the array +template +inline TableMapCursor TableMapOrder( const TableRegionCtx &, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.entries = map.Entries(); + cursor.count = map.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: gather the LIVE entries out of the segment chain in +// insertion order, then sort. A dead entry costs nothing on any wire (§2.8). +template +inline TableMapCursor TableMapOrder( const TableArena & arena, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.allocator = arena.allocator; + cursor.count = map.count; + if ( map.entries.value == 0 || map.count <= 0 ) { cursor.ok = map.count == 0; cursor.count = 0; return cursor; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + if ( head->live != map.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + const Entry ** order = (const Entry **) arena.allocator.alloc( arena.allocator.context, (int64_t) map.count * (int64_t) sizeof( const Entry * ) ); + if ( order == NULL ) { return cursor; } + int32_t at = 0; + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 && at < map.count ) + { + const TableMapSegment * segment = (const TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used && at < map.count; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + order[at++] = segment->entries + i; + } + segment_ref = segment->next; + } + if ( at != map.count ) + { + arena.allocator.free( arena.allocator.context, order ); + return cursor; + } + TableMapSort( order, map.count ); + cursor.order = order; + cursor.ok = true; + return cursor; +} + +template +inline TableMapCursor TableMapOrder( const TableArenaCtx & ctx, const TableMap & map ) +{ + return TableMapOrder( *ctx.arena, map ); +} + +template inline void TableMapRelease( TableMapCursor & cursor ) +{ + if ( cursor.order != NULL ) { cursor.allocator.free( cursor.allocator.context, (void *) cursor.order ); } + cursor.order = NULL; +} + +// ---- the builder's five (§2.8) ---- +// +// Insert APPENDS after one LINEAR SCAN of the live entries for the key it may +// replace, Find is that same scan, and Erase is the scan and one bit. The +// builder builds NO INDEX, and that is a rule: the sort happens once, at Lock, +// Save or Cook, and every lookup that matters runs over the sorted region. + +// the head, allocated when the first entry is inserted +template +inline TableMapHead * TableMapReach( TableWorker & worker, TableMap & map ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( map.entries.value != 0 ) { return (TableMapHead *) TableArenaAt( *worker.arena, (uint32_t) map.entries.value ); } + uint32_t at = 0; + TableMapHead * head = (TableMapHead *) worker.AllocRaw( (int64_t) sizeof( TableMapHead ), (int64_t) alignof( TableMapHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + map.entries.value = (int64_t) at; + return head; +} + +// one entry's storage, appended: the current segment when it has room, a new +// one carved from one call to the pair when it does not +template +inline Entry * TableMapAppend( TableWorker & worker, TableMapHead * head, TableMap & map ) +{ + TableMapSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableMapSegmentEntries ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableMapSegment *) worker.AllocRaw( (int64_t) sizeof( TableMapSegment ), (int64_t) alignof( TableMapSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableMapSegment * previous = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Entry * entry = segment->entries + segment->used; + segment->used++; + head->live++; + map.count++; + return entry; +} + +// the LINEAR SCAN: the live entries in insertion order, O( n ) key compares +template +inline Entry * TableMapScan( const TableArena & arena, const TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) == 0 ) { return segment->entries + i; } + } + segment_ref = segment->next; + } + return NULL; +} + +// ERASE marks the entry DEAD, one bit in the segment's slot and not in the +// entry table, and decrements the live count. Its storage is reclaimed at +// RESET and never reused mid-build, because reusing a slot would make "an +// entry's address is stable" false for exactly one case. +template +inline bool TableMapErase( TableArena & arena, TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return false; } + TableMapHead * head = (TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) != 0 ) { continue; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + map.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INSERTION order, live entries only (§2.8) ---- +template struct TableMapEach +{ + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableMapSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableMapSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + auto operator*() const { return TableEntryEach( segment->entries + index ); } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableMapEach TableMapEachOf( const TableArena & arena, const TableMap & map ) +{ + TableMapEach each = { &arena, TableRef() }; + if ( map.entries.value != 0 ) + { + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + each.first = head->first; + } + return each; +} + +// ---- the LOAD side: where a decoded entry lands (§2.8) ---- +// +// THE READER TRUSTS NOTHING and spends one compare per entry. Every load path +// applies the same rules and produces one report (§4), so the region load of +// §6.5 and LoadBuilder never disagree about a wire. These two shapes are what +// makes that true with one generated decoder: a REGION carves the entry array +// out of the holder node's own extent, and the TOOL's path appends into the +// builder's arena, and the decoder above them cannot tell which it has. + +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. + +// TableMapFill is one map field being decoded: where the next entry lands, and +// the entry that last LANDED, which is what the ascending check compares +// against. +template struct TableMapFill +{ + TableMap * map = NULL; + Entry * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; +}; + +template +inline TableMapFill TableMapFillBegin( const TableNodeMap & nodes, TableMap & map, uint32_t n ) +{ + TableMapFill fill; + fill.map = ↦ + map.entries.value = 0; + map.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Entry ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Entry ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Entry *) base; + fill.capacity = (int32_t) n; + map.entries.value = (int64_t) ( base - (const uint8_t *) &map.entries ); + fill.ok = true; + return fill; +} + +// the entry that last LANDED — NULL before the first +template inline Entry * TableMapFillLast( TableMapFill & fill ) +{ + if ( fill.map->count <= 0 ) { return NULL; } + if ( fill.array != NULL ) { return fill.array + ( fill.map->count - 1 ); } + return TableMapLive( *fill.worker->arena, *fill.map, fill.map->count - 1 ); +} + +// the next slot, at the entry type's declared defaults +template inline Entry * TableMapFillNext( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + if ( fill.map->count >= fill.capacity ) { return NULL; } + Entry * entry = fill.array + fill.map->count; + TableReset( *entry ); + fill.map->count++; + return entry; + } + TableMapHead * head = TableMapReach( *fill.worker, *fill.map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( *fill.worker, head, *fill.map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// A MAP WITH HALF ITS KEYS IS NOT A MAP (§2.8): at the first entry whose key +// kind disagrees with the reader's declaration the map resets to EMPTY, one +// kind_mismatch is counted for the map, and its remaining bytes are skipped. +template inline void TableMapFillReset( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + fill.map->entries.value = 0; + fill.map->count = 0; + return; + } + if ( fill.map->entries.value != 0 ) + { + TableMapHead * head = (TableMapHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.map->entries.value ); + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + } + fill.map->count = 0; +} + +// an EMPTY map's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableMapFillEnd( TableMapFill & fill ) +{ + if ( fill.array != NULL && fill.map->count == 0 ) { fill.map->entries.value = 0; } +} + +// the k-th LIVE entry of a builder map, in insertion order — what the tool +// path's ascending check compares against +template +inline Entry * TableMapLive( const TableArena & arena, const TableMap & map, int32_t index ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + int32_t at = 0; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( at == index ) { return segment->entries + i; } + at++; + } + segment_ref = segment->next; + } + return NULL; +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.8, §6.5) ---- +// +// LoadMeasure's term for a map is N x sizeof( Entry ) rounded to +// alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this +// reads no field: it walks the map's own header and, where an entry's value +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect, +// and a refusal carries its reason (§6.5). +// A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its +// own L and the body's terminator, and under this form's variable lengths that +// footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a +// map's L can carry, and therefore what a LoadMeasure may be asked for. +static const int64_t kTableMapEntryFloor = 2; + +inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry + at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); + at += (int64_t) n * entry_size; + if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// ---- the TEXT form's placement (docs/SPEC-TABLES.md §2.8, §16) ---- +// +// The text is a plain JSON object keyed by the KEY, and the generic walk fills +// it through the ENTRY'S OWN descriptor — so all it needs from here is one +// entry at one key, handed back at its defaults. It is the builder's Insert +// with the ENTRY returned rather than its value, because the walk writes the +// value through a field row and not through a typed pointer. +template +inline Entry * TableMapPlace( TableWorker & worker, TableMap & map, Key key ) +{ + if ( worker.arena == NULL ) { return NULL; } + Entry * found = TableMapScan( *worker.arena, map, key ); + if ( found != NULL ) + { + TableResetMapValue( *found ); // a repeated key is LAST-WINS, whole + return found; + } + TableMapHead * head = TableMapReach( worker, map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( worker, head, map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// ---- the OPTIONAL RUNTIME INDEX (§2.8) ---- +// +// Open addressing with LINEAR PROBING over the sorted array, built AT LOAD for +// a map large enough that log n compares over a cold array cost more than one +// hash and a probe. IT IS NEVER STORED: the caller measures it, owns its +// storage, builds it in one pass and releases it whenever. +// +// ITS HASH AND ITS LOAD FACTOR ARE NOT A CROSS-PORT CONTRACT, and that is a +// rule. What a port is held to is the CONTRACT of the lookup: the same value +// the sorted array's Find returns for the same key, and no allocation past the +// storage the caller handed in. +struct TableMapIndex +{ + int32_t * slots = NULL; // entry indices, +1; 0 is an empty slot + int32_t capacity = 0; + bool good = false; +}; + +// this runtime's own, and no port reproduces it: fnv1a64 over the key's bytes +inline uint64_t TableMapHash( const void * bytes, int32_t length ) +{ + uint64_t hash = 0xCBF29CE484222325ull; + const uint8_t * at = (const uint8_t *) bytes; + for ( int32_t i = 0; i < length; i++ ) { hash ^= (uint64_t) at[i]; hash *= 0x100000001B3ull; } + return hash; +} +inline uint64_t TableMapHash( uint64_t key ) { return TableMapHash( (const void *) &key, (int32_t) sizeof( key ) ); } + +// this runtime's own load factor, and no port reproduces it either: the next +// power of two at or above twice the count, so a probe run stays short +inline int32_t TableMapIndexSlots( int32_t count ) +{ + int32_t slots = 8; + while ( slots < count * 2 ) { slots *= 2; } + return slots; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_MAP + +#ifndef LISTDEMO_SCHEMA_TABLE_LIST +#define LISTDEMO_SCHEMA_TABLE_LIST + +namespace listdemo { + +// ---- an UNBOUNDED ARRAY: a counted array whose count the data decides (§2.9) ---- +// +// On the wire, in a region and in a cook a list is the kind 14 body a [..N]T +// writes, its elements by-value records inside the holder's node extent. What +// this adds is the slot, a builder that appends into segments that never +// move, and a const surface that indexes and iterates in place. There is no +// sort, no key and no lookup: the order is INSERTION order, and it is +// identity the way position is identity in a fixed array. + +// elements carved from ONE call to the allocator pair. A new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableListSegmentElements = 32; + +// THE ELEMENT STORAGE: T itself, and a TableRef slot for a []*T, whose +// elements are references exactly as a pointer field's slot is (§2.1) +template struct TableListStorage { typedef T Element; }; +template struct TableListStorage { typedef TableRef Element; }; + +// WHAT THE CONST FORM ANSWERS: the element by reference, and on a []*T the +// RESOLVED pointer, one add on the self-relative delta, NULL for a null slot, +// exactly as At answers it (§6.2, §6.3) +template struct TableListConst +{ + typedef const T & Result; + static Result At( const T * element ) { return *element; } +}; +template struct TableListConst +{ + typedef const T * Result; + static Result At( const TableRef * element ) + { + return element->value != 0 ? (const T *) ( (const uint8_t *) element + element->value ) : NULL; + } +}; + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.9, §7.2) ---- +// +// An int64 self-relative reference to the element array and an int32 count, +// then padding to eight. The reference is a TableRef like a pointer's: in the +// arena it names the builder's HEAD, in a region it is the delta from the slot +// to the first element, and 0 is the empty list in both. It is the map's slot +// exactly, because it is the same two facts. +template struct TableList +{ + typedef typename TableListStorage::Element Element; + + TableRef elements; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Element * Elements() const + { + return elements.value != 0 ? (const Element *) ( (const uint8_t *) &elements + elements.value ) : NULL; + } + int32_t size() const { return count; } + + // INDEXING IS BOUNDS-CHECKED IN EVERY BUILD (§2.4, §2.9): the extent is a + // number that CAME FROM A FILE, so an index past it is not a mistake a + // release build gets to make cheaply. There is no undefined-behavior path + // here in any configuration. The assert carries the message where a + // debugger can read it and NDEBUG removes that. The fatal is what stands + // after it. Both go through the hooks: define schema_assert and + // schema_fatal and this refusal lands in your own handler. + void RefuseIndex( int32_t index ) const + { + if ( (uint32_t) index >= (uint32_t) count ) + { + schema_assert( false && "an unbounded array is indexed inside its count, which came from a file" ); + schema_fatal(); + } + } + typename TableListConst::Result operator[]( int32_t index ) const + { + RefuseIndex( index ); + return TableListConst::At( Elements() + index ); + } + + // ---- iteration: INDEX order, the element and no key ---- + // + // It carries no iterator_traits, for the reason TableKeyed's does not + // (§13.9). + struct ConstIterator + { + const Element * at; + typename TableListConst::Result operator*() const { return TableListConst::At( at ); } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Elements() }; } + ConstIterator end() const { return ConstIterator{ Elements() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.9, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first element is added. Each +// segment is a fixed number of elements carved from one call to the allocator +// pair. An element's address is stable for the arena's life, so a T * handed +// back by Add stays valid while other elements arrive. +struct TableListHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an Add appends into + int32_t live; + int32_t dead; +}; + +template struct TableListSegment +{ + TableRef next; + int32_t used; // elements carved from this segment + int32_t padding; + uint32_t dead[ ( kTableListSegmentElements + 31 ) / 32 ]; // Erase marks one bit, never the element + Element elements[ kTableListSegmentElements ]; +}; + +inline bool TableListSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// the head, allocated when the first element is added +template +inline TableListHead * TableListReach( TableWorker & worker, TableList & list ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( list.elements.value != 0 ) { return (TableListHead *) TableArenaAt( *worker.arena, (uint32_t) list.elements.value ); } + uint32_t at = 0; + TableListHead * head = (TableListHead *) worker.AllocRaw( (int64_t) sizeof( TableListHead ), (int64_t) alignof( TableListHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + list.elements.value = (int64_t) at; + return head; +} + +// one element's storage, appended: the current segment when it has room, a +// new one carved from one call to the pair when it does not. NULL means NOT +// ADDED: an arena that cannot carve another segment, or a count at the int32 +// cap (§2.2, §2.9). +template +inline typename TableList::Element * TableListAppend( TableWorker & worker, TableListHead * head, TableList & list ) +{ + typedef typename TableList::Element Element; + if ( list.count >= INT32_MAX ) { return NULL; } // the int32 storage cap + TableListSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableListSegmentElements ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableListSegment *) worker.AllocRaw( (int64_t) sizeof( TableListSegment ), (int64_t) alignof( TableListSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableListSegment * previous = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Element * element = segment->elements + segment->used; + segment->used++; + head->live++; + list.count++; + return element; +} + +// ADD, whole: the head, the append, and the element at its declared defaults +// (§2.9). The text form's placement is this same call, because a list has no +// key to place under (§16). +template +inline typename TableList::Element * TableListPlace( TableWorker & worker, TableList & list ) +{ + typedef typename TableList::Element Element; + TableListHead * head = TableListReach( worker, list ); + if ( head == NULL ) { return NULL; } + Element * element = TableListAppend( worker, head, list ); + if ( element == NULL ) { return NULL; } + new ( element ) Element(); // value-init: the declared defaults, and null for a slot + return element; +} + +// ERASE, ADDRESSED BY THE POINTER (§2.9): the element Add handed back is the +// handle, because a list has no key and the address is the one thing the +// builder promises never moves (§6.4). It marks the element DEAD, one bit in +// the segment's slot and not in the element storage, and decrements the live +// count. False when the pointer is not this list's. Its storage is reclaimed +// at RESET and never reused mid-build, the map's rule for the map's reason. +template +inline bool TableListErase( TableArena & arena, TableList & list, const typename TableList::Element * element ) +{ + typedef typename TableList::Element Element; + if ( list.elements.value == 0 || element == NULL ) { return false; } + TableListHead * head = (TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableListSegment * segment = (TableListSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + if ( element >= segment->elements && element < segment->elements + segment->used ) + { + const int32_t i = (int32_t) ( element - segment->elements ); + if ( TableListSegmentDead( segment->dead, i ) ) { return false; } // already erased + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + list.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INDEX order, live elements only (§2.9) ---- +template struct TableListEach +{ + typedef typename TableList::Element Element; + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableListSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableListSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + Element * operator*() const { return segment->elements + index; } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableListEach TableListEachOf( const TableArena & arena, const TableList & list ) +{ + TableListEach each = { &arena, TableRef() }; + if ( list.elements.value != 0 ) + { + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + each.first = head->first; + } + return each; +} + +// ---- the INDEX-ORDER CURSOR the four writing walks read (§2.9) ---- +// +// Measure, Save, Lock and Cook each visit a list's live elements in the order +// they were added, and they allocate nothing to do it: a region's cursor is +// the array in place, and the builder's walks the segment chain. Indexing the +// builder's form is SEQUENTIAL by construction, every walk steps i, i + 1, +// i + 2, so the cursor remembers where the last access landed and moves one +// live slot per step. An access behind the memo restarts from the first +// segment, which no walk here does. +template struct TableListCursor +{ + const Element * elements = NULL; // the region's form: the array in place + const TableArena * arena = NULL; // the builder's form: the segments + TableRef first; + int32_t count = 0; + bool ok = false; + // the memo: the segment and slot the last access landed on, and the live + // index that slot holds + mutable const TableListSegment * segment = NULL; + mutable int32_t within = -1; + mutable int32_t logical = -1; + + const Element * At( int32_t index ) const + { + if ( elements != NULL ) { return elements + index; } + if ( segment == NULL || index < logical ) + { + segment = first.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL; + within = -1; + logical = -1; + } + while ( logical < index ) + { + for ( ;; ) + { + within++; + while ( segment != NULL && within >= segment->used ) + { + segment = segment->next.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + within = 0; + } + if ( segment == NULL ) { return NULL; } // the slot and the head disagree + if ( !TableListSegmentDead( segment->dead, within ) ) { break; } + } + logical++; + } + return segment->elements + within; + } + const Element & operator[]( int32_t index ) const { return *At( index ); } +}; + +// the REGION form: the array is the cursor +template +inline TableListCursor::Element> TableListElements( const TableRegionCtx &, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.elements = list.Elements(); + cursor.count = list.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: the live elements out of the segment chain, in the +// order they were added. A dead element costs nothing on any wire (§2.9). +template +inline TableListCursor::Element> TableListElements( const TableArena & arena, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.arena = &arena; + cursor.count = list.count; + if ( list.elements.value == 0 || list.count <= 0 ) { cursor.ok = list.count == 0; cursor.count = 0; return cursor; } + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + if ( head->live != list.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + cursor.first = head->first; + cursor.ok = true; + return cursor; +} + +template +inline TableListCursor::Element> TableListElements( const TableArenaCtx & ctx, const TableList & list ) +{ + return TableListElements( *ctx.arena, list ); +} + +// ---- the LOAD side: where a decoded element lands (§2.9) ---- +// +// The same two shapes the map's fill takes, because the decoder above them +// cannot tell which it has: a REGION carves the element array out of the +// holder node's own extent, PRE-ORDER, and the TOOL's path appends into the +// builder's arena. +template struct TableListFill +{ + typedef typename TableList::Element Element; + TableList * list = NULL; + Element * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; + bool refused = false; // a count above the int32 cap on the tool's path: LoadBuilder answers NULL +}; + +template +inline TableListFill TableListFillBegin( const TableNodeMap & nodes, TableList & list, uint64_t n ) +{ + typedef typename TableList::Element Element; + TableListFill fill; + fill.list = &list; + list.elements.value = 0; + list.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( n > (uint64_t) INT32_MAX ) + { + // A COUNT ABOVE THE int32 STORAGE CAP (§2.2, §2.9): into a region it was + // refused by LoadMeasure before this ran, and into a builder it is the + // refusal LoadBuilder answers NULL for, moving no counter + fill.refused = nodes.carve->worker != NULL; + return fill; + } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Element ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Element ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Element *) base; + fill.capacity = (int32_t) n; + list.elements.value = (int64_t) ( base - (const uint8_t *) &list.elements ); + fill.ok = true; + return fill; +} + +// the next slot, at the element's declared defaults. NULL when the arena +// could not carve, which the decoder reports as framing damage +template inline typename TableList::Element * TableListFillNext( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count >= fill.capacity ) { return NULL; } + Element * element = fill.array + fill.list->count; + new ( element ) Element(); + fill.list->count++; + return element; + } + return TableListPlace( *fill.worker, *fill.list ); +} + +// A SLOT WHOSE ELEMENT NEVER LANDED is given back (§2.9, §4): the array keeps +// what it decoded, and an element whose own framing gave out before one byte +// of it decoded was not decoded. The region's form uncounts it, and the builder's +// marks it dead, which is what the storage rule allows mid-build. +template inline void TableListFillDrop( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count > 0 ) { fill.list->count--; } + return; + } + if ( fill.list->elements.value == 0 ) { return; } + TableListHead * head = (TableListHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.list->elements.value ); + if ( head->last.value == 0 ) { return; } + TableListSegment * segment = (TableListSegment *) TableArenaAt( *fill.worker->arena, (uint32_t) head->last.value ); + if ( segment->used <= 0 ) { return; } + const int32_t i = segment->used - 1; + if ( TableListSegmentDead( segment->dead, i ) ) { return; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + fill.list->count--; +} + +// an EMPTY list's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableListFillEnd( TableListFill & fill ) +{ + if ( fill.array != NULL && fill.list->count == 0 ) { fill.list->elements.value = 0; } +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.9, §6.5) ---- +// +// N x sizeof( T ) rounded to alignof( T ), AT EVERY DEPTH. N is framing and +// not a value, so this reads no field: it walks the list's own header and, +// where a table element holds a list or a map of its own, the elements' +// headers under it. Every -1 carries its REASON (§6.5): the int32 cap first, +// because a count past it cannot fit any body, and then the body's own L. +inline bool TableListWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t elem_size, int64_t elem_align, uint8_t elem_kind, int64_t elem_floor, + TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != elem_kind ) { return true; } // another element kind: §4's ordinary kind mismatch, the field reads empty + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; } // an N the list's L cannot carry + at = ( at + elem_align - 1 ) & ~( elem_align - 1 ); + at += (int64_t) n * elem_size; + if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_LIST + +#ifndef LISTDEMO_SCHEMA_BUILD_VERSION +#define LISTDEMO_SCHEMA_BUILD_VERSION + +namespace listdemo { + +// THE BUILD VERSION (docs/SPEC-TABLES.md §20): one digest over every fact the bytes +// this build produces depend on — the type wire's protocol id, every record's +// layout as the compiler's own C ABI model computes it, and the facts that +// decide what a load PUTS in those slots. It is the number a cook's header +// carries and the number Open compares, and the number a block's prologue +// carries and BlockOpen compares: a build version answers "which build?" and +// not "which form?", and what separates the two forms is their MAGIC. +// +// There are TWO ids in the design and they are not interchangeable: the +// PROTOCOL ID is the type wire's and nothing else, and the BUILD VERSION is +// what everything cooked or blocked is keyed by. A table edit moves this and +// never the protocol id; a type edit moves both. +static const uint64_t BuildVersion = 0x8d7c0edaca4571c7ull; + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_BUILD_VERSION + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK +#define LISTDEMO_SCHEMA_TABLE_COOK + +namespace listdemo { + +// ---- the cooked form (docs/SPEC-TABLES.md §7) ---- +// +// A cooked file is a HEADER, a DATA part and an ATTRIBUTION part, in that +// order. Every word of the header is a u64 written in the byte order the cook +// was produced in, and the header is 64 bytes: +// +// 0 magic 0x4b4f4f434d484353, read BYTEWISE before anything else +// 8 build_version the unit's id (docs/SPEC-TABLES.md §20) +// 16 byte_order 1 little, 2 big — the order that WROTE the file +// 24 data_length the region's bytes, rounded up to alignment +// 32 attribution_length the directory's bytes, or 0 +// 40 alignment the region's alignment, never below eight +// 48 reserved zero +// 56 reserved zero +// +// The DATA part is Lock's region written verbatim (§7.2) — the root at its +// base — and it is what a runtime points at. The ATTRIBUTION part is the node +// directory (§6.3), and NOTHING THAT READS THE STRUCTURE TOUCHES IT: it is +// written beside the data for schema cook-check, so a build that ships no +// tooling need not carry it at all. +static const int64_t kTableCookHeaderBytes = 64; + +// THE MAGIC'S VALUE, and a consumer written from the page needs the constant +// rather than a description of one. It is "SCHMCOOK" read as ASCII in the byte +// order a little-endian store produces — the same shape the block form's +// SCHMABLK takes, so a hex dump of a little-endian cook is legible and the two +// accelerators sit in one vocabulary. +// +// IT IS STORED IN THE PRODUCER'S ORDER, which is what makes it the byte-order +// check as well as the form check: a consumer reads back this build's +// constant, or that constant byte-reversed — which identifies a cook of the +// OTHER order — or something that is not a cook. All three answers but the +// first refuse, and a cook and a BLOCK are separated here too, because a +// form's identity belongs in its magic rather than in a second digest. +static const uint64_t TableCookMagic = 0x4b4f4f434d484353ull; + +// THIS BUILD's byte order, as the header's own word carries it. The magic is +// what REFUSES a foreign order; this word is what RECORDS which order wrote +// the file, so a refusal names the order rather than inferring it and a tool +// dumping a cook reads the fact. A file whose magic matched and whose order +// word did not is corrupt, and there is no reading that recovers it. +// +// The BUILD VERSION cannot do either job: §20.1 digests byteorder as a +// GENERATION input, little for every target schema generates for today, so +// two builds of one schema for two orders emit the same id. +#if defined( __BYTE_ORDER__ ) && defined( __ORDER_BIG_ENDIAN__ ) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__ +static const uint64_t TableCookByteOrder = 2; // big +#else +static const uint64_t TableCookByteOrder = 1; // little +#endif + +// The greatest region alignment a cooked file may name. The DATA part begins +// at align_up( 64, alignment ), which is 64 for every unit this language can +// declare — the largest alignment it has is sixteen — so a word past this cap +// describes a file no build of this schema wrote (docs/SPEC-TABLES.md §7.1). +static const uint64_t TableCookMaxAlign = 64; + +// The header read, BYTEWISE. memcpy is the portable spelling of "these eight +// bytes, in this machine's order"; every compiler this repo builds under folds +// it to one load, and it is the only read in the whole of Open that is not a +// comparison. +inline uint64_t table_cook_read64( const uint8_t * p ) +{ + uint64_t v; + memcpy( &v, p, sizeof( v ) ); + return v; +} + +// TableCookOpen: THE WHOLE CHECK, in one place, because §7 states the +// enumeration once and every generated Open is that one enumeration plus +// its own root's two layout facts. +// +// THE CHECK, in order: the magic read bytewise, the byte order it establishes, +// the build version against this build's own, both RESERVED words zero, the +// region alignment the header names, the two part lengths against the length +// the caller passed — a truncated file and a file with trailing bytes are the +// same refusal — the root's own storage inside the data part, and the +// alignment of the base. +// +// AND THAT IS ALL OF IT. On a match the bytes ARE what this build wrote, in +// this build's layout and this build's byte order, so there is nothing to +// validate and nothing to fix up: the caller gets the root. Nothing per node +// happens here, which is what makes open O(1) in the file's size; a walk of +// any shape would forfeit that, and validating an untrusted file is schema +// cook-check's job and a person's decision (§7.4). +// +// EVERY NUMBER BELOW COMES OUT OF THE FILE, so the arithmetic is unsigned and +// each term is BOUNDED BEFORE IT IS ADDED: a forged length near 2^64 must +// refuse, and an addition that wrapped would be the defect the comparison +// after it was supposed to catch. Nothing past length is read on any path, +// including every refusing one. +inline const uint8_t * TableCookOpen( const void * bytes, uint64_t length, uint64_t root_size, uint64_t root_align ) +{ + if ( bytes == NULL ) { return NULL; } + if ( length < (uint64_t) kTableCookHeaderBytes ) { return NULL; } + const uint8_t * raw = (const uint8_t *) bytes; + // the MAGIC, bytewise and first: it is what establishes the byte order + // every other header word is read in, so nothing else may be read before + // it. A byte-reversed constant is a cook of the other order and refuses + // here, which is why the order never reaches a fix-up pass. + if ( table_cook_read64( raw ) != TableCookMagic ) { return NULL; } + if ( table_cook_read64( raw + 16 ) != TableCookByteOrder ) { return NULL; } + if ( table_cook_read64( raw + 8 ) != BuildVersion ) { return NULL; } + // the RESERVED words: a non-zero one means a writer used a form this build + // does not understand, and Open refuses rather than ignoring it. + if ( table_cook_read64( raw + 48 ) != 0 ) { return NULL; } + if ( table_cook_read64( raw + 56 ) != 0 ) { return NULL; } + const uint64_t data_length = table_cook_read64( raw + 24 ); + const uint64_t attribution_length = table_cook_read64( raw + 32 ); + const uint64_t alignment = table_cook_read64( raw + 40 ); + // THE ALIGNMENT WORD IS DATA, and it is the one header field the rest of + // the check does arithmetic WITH rather than only comparison against. A + // region's alignment is a power of two, never below eight (the floor that + // puts the attribution part on an eight-byte boundary without a second + // padding rule) and never past the cap above; a word that is none of those + // rounds nothing and aligns nothing, so it is refused before it is used. + if ( alignment < 8 || alignment > TableCookMaxAlign ) { return NULL; } + if ( ( alignment & ( alignment - 1 ) ) != 0 ) { return NULL; } + // and it must be an alignment THE ROOT CAN SIT AT, since the root is at + // the region's base: both are powers of two, so "at least the root's" + // is one division. + if ( ( alignment % root_align ) != 0 ) { return NULL; } + // The DATA part begins at align_up( 64, alignment ). It is DERIVED and not + // a header field, because a fact a reader computes is a fact two writers + // cannot disagree about. + const uint64_t data_offset = ( (uint64_t) kTableCookHeaderBytes + alignment - 1 ) & ~( alignment - 1 ); + if ( length < data_offset ) { return NULL; } + // the two part lengths against the length the caller passed. The whole + // file is data_offset + data_length + attribution_length, and a length + // that is not EXACTLY that refuses — truncation and trailing bytes are one + // refusal, and both terms are subtracted rather than added so no sum can + // carry. + if ( data_length > length - data_offset ) { return NULL; } + if ( attribution_length != length - data_offset - data_length ) { return NULL; } + // the ROOT sits at the region's base, so the region has to hold it: a + // shorter data part describes a root partly outside the file, which is the + // one way a match-and-point reader could hand back storage it never + // received. + if ( data_length < root_size ) { return NULL; } + const uint8_t * base = raw + data_offset; + // the alignment of the BASE. The header pads the data part to the region's + // alignment, so a base an allocator or mmap gave you is already aligned — + // mmap gives page alignment for free — and a base that is not is a caller's + // buffer this form cannot be read out of. + if ( ( (uintptr_t) base % (uintptr_t) alignment ) != 0 ) { return NULL; } + return base; +} + +// ---- the cooked form, the WRITE side (docs/SPEC-TABLES.md §7.6) ---- +// +// THE BYTE ORDER IS THE TARGET'S, NOT THE HOST'S. A cook is produced in the +// byte order of the build that will read it (§7), so the fixing happens here — +// offline, once, on the writing side — and never at Open. Passing +// TableByteOrder::Big on a little-endian machine produces a big-endian build's +// file, and nothing about the writing host reaches the bytes. +enum class TableByteOrder +{ + Little = 1, // the header's byte_order word, and the order every scalar is written in + Big = 2, +}; + +// One store, width as an argument. Every call site passes a literal width, so +// the loop folds to a store (and a byte swap on the foreign order); a name per +// width would claim four §11 names to save nothing. +inline void table_cook_put( uint8_t * at, uint64_t value, int32_t width, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * i ) ); } + } + else + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * ( width - 1 - i ) ) ); } + } +} + +// A 128-bit store as two lanes: sixteen bytes, the low lane first in the +// little order and the high lane first — each lane big-endian — in the big +// order, exactly as a u64 is one lane of eight (docs/SPEC-TABLES.md §7.2). +inline void table_cook_put128( uint8_t * at, uint64_t lo, uint64_t hi, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) { table_cook_put( at, lo, 8, order ); table_cook_put( at + 8, hi, 8, order ); } + else { table_cook_put( at, hi, 8, order ); table_cook_put( at + 8, lo, 8, order ); } +} + +// A buffer piece: the USED bytes and nothing else. The tail is already zero — +// the whole extent was zeroed before any field was written — so this copies the +// used prefix and leaves the rest, which is what makes a string's unused tail a +// consequence of one memset rather than a rule per buffer. A used length past +// the buffer, or below zero, is a value no reader could have produced and it is +// clamped rather than trusted: this writes inside the caller's buffer on every +// input. +inline void table_cook_bytes( uint8_t * at, const void * source, int64_t used, int64_t capacity ) +{ + if ( used <= 0 ) { return; } + const int64_t n = used < capacity ? used : capacity; + memcpy( at, source, (size_t) n ); +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE +#define LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// ---- the cooked form's WRITE side for a POINTERED root (docs/SPEC-TABLES.md §7.6) ---- +// +// A pointered root's cook is the region of §7.2: every node the numbering +// reached (§3.1), once, at its own type's alignment, in index order, the root +// at offset zero. This is that region while it is being laid out and written — +// the tool's own Layout and Write, in one struct. +// +// The OFFSETS are one per node, the root's zero at position 0 and node index k +// at position k - 1, which is the directory's own order (§6.3); they are the +// one allocation the write makes beyond the numbering, and they go through the +// same pair. A measure needs no offsets and leaves the pointer NULL. +struct TableCookRegion +{ + const TableNumbering * numbering = NULL; // node -> index, from the walk that placed it + int64_t * offsets = NULL; // index - 1 -> the node's region offset; NULL while measuring + int64_t count = 0; // nodes, the root included + int64_t bytes = 0; // the data part's length, rounded to align + int64_t align = 0; // the region's alignment: the nodes' greatest, never below eight + uint8_t * base = NULL; // where the data part is being written; NULL while measuring +}; + +// A reference slot: the SELF-RELATIVE delta from the slot's own address to the +// node's start (§6.3), and zero for null. The node is found by the address the +// numbering keyed it under, which is the same address the walk resolved through +// the same context — so a reference the numbering does not carry is a slot the +// walk never reached (a counted array's slot past its count, an absent +// optional's value) holding a node the region will not hold, and it is refused +// rather than written as a delta to nowhere. +inline bool table_cook_ref( const TableCookRegion & region, uint8_t * at, const void * pointee, TableByteOrder order ) +{ + if ( pointee == NULL ) { table_cook_put( at, 0, 8, order ); return true; } + uint64_t index = 0; + if ( !TableNumberingIndex( *region.numbering, pointee, index ) ) { return false; } + if ( index == 0 || index > (uint64_t) region.count ) { return false; } + const int64_t delta = region.offsets[index - 1] - (int64_t) ( at - region.base ); + table_cook_put( at, (uint64_t) delta, 8, order ); + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// table Photo — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Photo { + uint32_t width = 0; + uint32_t height = 0; +}; + +// table Album — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Album { + TableList photos; // Photo: the element array, empty until an Add + TableRef cover; // *Photo — null until assigned +}; + +// ---- prefill: the declared defaults, in place (docs/SPEC-TABLES.md) ---- + +inline void PhotoReset( Photo & value ); +inline void AlbumReset( Album & value ); + +inline void PhotoReset( Photo & value ) +{ + value.width = 0; + value.height = 0; +} + +inline void AlbumReset( Album & value ) +{ + value.photos.elements.value = 0; // Photo: empty + value.photos.count = 0; + value.photos.padding = 0; + value.cover.value = 0; // *Photo — null +} + +// ---- the arena's reset hook (docs/SPEC-TABLES.md §6) ---- +// +// TableWorker::Alloc is a template and cannot name a member's Reset, so +// the arena reaches it through this overload set by argument-dependent +// lookup. It is how a node born in raw arena storage comes to hold the +// declared defaults without value-initialising the whole aggregate. + +inline void TableReset( Photo & value ) { PhotoReset( value ); } +inline void TableReset( Album & value ) { AlbumReset( value ); } + +// ---- pointer targets: allocation and resolution (docs/SPEC-TABLES.md §2) ---- +// +// A reference resolves differently in the two forms, and the CONTEXT says +// which: in the arena it is an offset; in a region it is a self-relative +// delta, so the const deref below is one add and needs no base pointer. + +// Photo is a pointer target. +inline const Photo * PhotoAt( const TableRef & ref ) // the const form's hot path: one add, no base +{ + return ref.value != 0 ? (const Photo *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline Photo * PhotoAt( TableRef & ref ) +{ + return ref.value != 0 ? (Photo *) ( (uint8_t *) &ref + ref.value ) : NULL; +} +inline const Photo * PhotoAt( const TableRegionCtx &, const TableRef & ref ) { return PhotoAt( ref ); } +inline const Photo * PhotoAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const Photo *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +// while the builder is mutable, resolve against the arena itself +inline Photo * PhotoAt( TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (Photo *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} +inline const Photo * PhotoAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const Photo *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} +// allocate one Photo in the arena; the slot holds the arena offset +inline Photo * PhotoEmplace( TableWorker & worker, TableRef & slot ) +{ + TableSlot allocated = worker.Alloc(); + slot = allocated.ref; + return allocated.ptr; +} + +// ---- codecs: measure/save/load per closure member ---- + +inline int64_t PhotoMeasureBody( TableIds & ids, const Photo & value ); +LISTDEMO_TABLE_INLINE bool PhotoSaveBody( TableWriter & w, TableIds & ids, const Photo & value ); +LISTDEMO_TABLE_INLINE bool PhotoLoadBody( TableReader & r, Photo & value ); +template inline int64_t AlbumMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Album & value ); +template inline bool AlbumSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Album & value ); +template inline bool AlbumSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Album & value ); +inline bool AlbumLoadBody( TableReader & r, const TableNodeMap & nodes, Album & value ); + +// ---- pointer-graph walkers: number (measure/save), pack (Lock) ---- + +template inline bool PhotoNumber( const Ctx & ctx, TableNumbering & numbering, const Photo & value ); +template inline int64_t PhotoPackMeasure( const Ctx & ctx, TablePackMap & seen, const Photo & value ); +template inline bool PhotoPack( const Ctx & ctx, TablePackMap & seen, const Photo & src, Photo & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool AlbumNumber( const Ctx & ctx, TableNumbering & numbering, const Album & value ); +template inline int64_t AlbumPackMeasure( const Ctx & ctx, TablePackMap & seen, const Album & value ); +template inline bool AlbumPack( const Ctx & ctx, TablePackMap & seen, const Album & src, Album & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +// ---- the numbering's bridge to each member's codec (docs/SPEC-TABLES.md §3.1) ---- + +template inline int64_t TableNodeMeasure( const Ctx &, const TableNumbering &, TableIds & ids, const Photo & value ) { return PhotoMeasureBody( ids, value ); } +template inline bool TableNodeSave( const Ctx &, const TableNumbering &, TableWriter & w, TableIds & ids, const Photo & value ) { return PhotoSaveBody( w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Album & value ) { return AlbumMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Album & value ) { return AlbumSaveBody( ctx, numbering, w, ids, value ); } + +inline int64_t PhotoMeasureBody( TableIds & ids, const Photo & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.width != 0 ) { bytes += TableLebBytes( ids.ref( 0xdbdacd932fd1e9bfull, 17 ) ) + 1 + 4; } // width + if ( value.height != 0 ) { bytes += TableLebBytes( ids.ref( 0x17720bf67d347222ull, 18 ) ) + 1 + 4; } // height + return bytes; +} + +inline int64_t PhotoMeasure( const Photo & value ) +{ + TableIds ids; + const int64_t body = PhotoMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool PhotoSaveBody( TableWriter & w, TableIds & ids, const Photo & value ) +{ + if ( value.width != 0 ) + { + w.putleb( ids.ref( 0xdbdacd932fd1e9bfull, 17 ) ); w.put8( 8 ); // width + w.put32( uint32_t( value.width ) ); + } + if ( value.height != 0 ) + { + w.putleb( ids.ref( 0x17720bf67d347222ull, 18 ) ); w.put8( 8 ); // height + w.put32( uint32_t( value.height ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t PhotoSave( const Photo & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !PhotoSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == PhotoMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool PhotoLoadBody( TableReader & r, Photo & value ) +{ + PhotoReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xdbdacd932fd1e9bfull: // width + { + if ( kind != 8 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + uint32_t decoded_v = uint32_t( r.get32( ) ); + value.width = decoded_v; + break; + } + case 0x17720bf67d347222ull: // height + { + if ( kind != 8 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + uint32_t decoded_v = uint32_t( r.get32( ) ); + value.height = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict PhotoLoadVerdict( Photo & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + PhotoReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + PhotoReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !PhotoLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool PhotoLoad( Photo & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return PhotoLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t PhotoMeasureMessage( const Photo & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = PhotoMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t PhotoSaveMessage( const Photo & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !PhotoSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == PhotoMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool PhotoLoadMessage( Photo & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + PhotoReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return PhotoLoadBody( r, value ); +} + +template +inline int64_t AlbumMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Album & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // photos: a kind 14 array of kind 17 elements, INDEX order (§2.9) + TableListCursor cursor_photos = TableListElements( ctx, value.photos ); + if ( !cursor_photos.ok ) { return -1; } // the slot and the head disagree + if ( cursor_photos.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_photos = ids.ref( 0x40b1d94aff3ab130ull, 2 ); + int64_t body_photos = 0; + body_photos += 1 + TableLebBytes( (uint64_t) ( cursor_photos.count ) ); // the element kind byte and the count + for ( int32_t elem_i_photos = 0; elem_i_photos < cursor_photos.count; elem_i_photos++ ) + { + { + const Photo * slot_pointee_photos = PhotoAt( ctx, cursor_photos[elem_i_photos] ); + uint64_t slot_index_photos = 0; + if ( slot_pointee_photos != NULL && !TableNumberingIndex( numbering, (const void *) slot_pointee_photos, slot_index_photos ) ) { return -1; } + body_photos += TableLebBytes( slot_index_photos ); + } + } + bytes += TableLebBytes( ref_photos ) + 1 + TableLebBytes( (uint64_t) ( body_photos ) ) + ( body_photos ); + } + } + { + const Photo * pointee_cover = PhotoAt( ctx, value.cover ); // *Photo + // A POINTER RIDES AS A NODE INDEX (docs/SPEC-TABLES.md §3.1): the + // header and the index and nothing below it, because the pointee's + // body is in the node table and not here. NULL IS ELIDED — absence + // and null are one value — and a non-null pointer ALWAYS rides, even + // when its node's body is entirely default. + if ( pointee_cover != NULL ) + { + uint64_t index_cover = 0; + if ( !TableNumberingIndex( numbering, (const void *) pointee_cover, index_cover ) ) { return -1; } + bytes += TableLebBytes( ids.ref( 0xaa19a78e404dea20ull, 3 ) ) + 1 + TableLebBytes( index_cover ); + } + } + return bytes; +} + +template +inline bool AlbumSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Album & value ) +{ + { + TableListCursor cursor_photos = TableListElements( ctx, value.photos ); // photos + if ( !cursor_photos.ok ) { return false; } + if ( cursor_photos.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_photos = ids.ref( 0x40b1d94aff3ab130ull, 2 ); + int64_t body_photos = 0; + body_photos += 1 + TableLebBytes( (uint64_t) ( cursor_photos.count ) ); // the element kind byte and the count + for ( int32_t elem_i_photos = 0; elem_i_photos < cursor_photos.count; elem_i_photos++ ) + { + { + const Photo * slot_pointee_photos = PhotoAt( ctx, cursor_photos[elem_i_photos] ); + uint64_t slot_index_photos = 0; + if ( slot_pointee_photos != NULL && !TableNumberingIndex( numbering, (const void *) slot_pointee_photos, slot_index_photos ) ) { return false; } + body_photos += TableLebBytes( slot_index_photos ); + } + } + w.putleb( ref_photos ); w.put8( 14 ); w.putleb( (uint64_t) body_photos ); // photos + w.put8( 17 ); w.putleb( (uint64_t) ( cursor_photos.count ) ); + for ( int32_t elem_i_photos = 0; elem_i_photos < cursor_photos.count; elem_i_photos++ ) + { + { + const Photo * slot_pointee_photos = PhotoAt( ctx, cursor_photos[elem_i_photos] ); + uint64_t slot_index_photos = 0; + if ( slot_pointee_photos != NULL && !TableNumberingIndex( numbering, (const void *) slot_pointee_photos, slot_index_photos ) ) { return false; } + w.putleb( slot_index_photos ); + } + } + } + } + { + const Photo * pointee_cover = PhotoAt( ctx, value.cover ); // *Photo + if ( pointee_cover != NULL ) + { + uint64_t index_cover = 0; + if ( !TableNumberingIndex( numbering, (const void *) pointee_cover, index_cover ) ) { return false; } + w.putleb( ids.ref( 0xaa19a78e404dea20ull, 3 ) ); w.put8( 17 ); // cover — a NODE INDEX into the flat node table + w.putleb( index_cover ); + } + } + return !w.overflow; +} + +template +inline bool AlbumSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Album & value ) +{ + if ( !AlbumSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool AlbumLoadBody( TableReader & r, const TableNodeMap & nodes, Album & value ) +{ + AlbumReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x40b1d94aff3ab130ull: // photos + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 17 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.photos, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + TableRef * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + { + uint64_t node_index_photos = 0; + if ( !sub.getleb( node_index_photos ) ) { r.report->malformed = true; break; } + TableNodeResolve( nodes, ( *slot ), node_index_photos, 0xf1a78dd2508964c3ull, r.report ); // *Photo + } + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xaa19a78e404dea20ull: // cover + { + if ( kind != 17 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + // A POINTER FIELD'S PAYLOAD IS A NUMBER (docs/SPEC-TABLES.md §3.1): it is + // bounds-checked and resolved through the numbering, never FOLLOWED, so + // there is no traversal here and therefore no traversal bound. + { + uint64_t node_index = 0; + if ( !r.getleb( node_index ) ) { r.report->malformed = true; return false; } + TableNodeResolve( nodes, value.cover, node_index, 0xf1a78dd2508964c3ull, r.report ); // *Photo + } + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// PhotoWireExtent: the extent Photo's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool PhotoWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + (void) body; (void) length; (void) at; (void) ids; (void) reason; // no list or map below this record + return true; +} + +// PhotoExtentAt: the node extent Photo's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as PhotoExtentPack advances it (§2.8, §2.9). +template +inline bool PhotoExtentAt( const Ctx & ctx, const Photo & value, int64_t & at ) +{ + (void) ctx; (void) value; (void) at; // no list or map below this record + return true; +} + +// PhotoExtentPack: carve Photo's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset PhotoExtentAt advances (§2.8, §2.9). +template +inline bool PhotoExtentPack( const Ctx & ctx, const Photo & src, Photo & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no list or map below this record + return true; +} + +// AlbumWireExtent: the extent Album's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool AlbumWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x40b1d94aff3ab130ull && field_kind == 14 ) // photos: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( TableRef ), (int64_t) alignof( TableRef ), 17, 1, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// AlbumExtentAt: the node extent Album's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as AlbumExtentPack advances it (§2.8, §2.9). +template +inline bool AlbumExtentAt( const Ctx & ctx, const Album & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.photos ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( TableRef ) - 1 ) & ~( (int64_t) alignof( TableRef ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( TableRef ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t AlbumExtent( const Ctx & ctx, const Album & value ) +{ + int64_t at = 0; + if ( !AlbumExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// AlbumExtentPack: carve Album's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset AlbumExtentAt advances (§2.8, §2.9). +template +inline bool AlbumExtentPack( const Ctx & ctx, const Album & src, Album & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.photos ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( TableRef ) - 1 ) & ~( (int64_t) alignof( TableRef ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( TableRef ); + if ( at + bytes > capacity ) { return false; } + TableRef * placed = (TableRef *) ( extent + at ); + at += bytes; + dst.photos.count = cursor.count; + dst.photos.padding = 0; + dst.photos.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.photos.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( TableRef ) ); // trivially copyable, by construction + } + } + return true; +} + +// ---- Album.photos: the builder's three (§2.9) ---- + +// ADD: the element is appended and handed back to fill. On a []*T that is +// the SLOT at null, which PhotoEmplace fills as it fills any pointer slot, +// and a second slot may hold the same reference: two slots, one node. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline TableRef * AlbumPhotosAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool AlbumPhotosErase( TableArena & arena, TableList & list, const TableRef * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach AlbumPhotosEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// PhotoNumber: number everything Photo POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool PhotoNumber( const Ctx & ctx, TableNumbering & numbering, const Photo & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// PhotoPackMeasure: the packed region bytes of everything Photo POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t PhotoPackMeasure( const Ctx & ctx, TablePackMap & seen, const Photo & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// PhotoPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool PhotoPackEdges( const Ctx & ctx, TablePackMap & seen, const Photo & src, Photo & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool PhotoPack( const Ctx & ctx, TablePackMap & seen, const Photo & src, Photo & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Photo ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Photo ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !PhotoExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return PhotoPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool PhotoPackEdges( const Ctx & ctx, TablePackMap & seen, const Photo & src, Photo & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// AlbumNumber: number everything Album POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool AlbumNumber( const Ctx & ctx, TableNumbering & numbering, const Album & value ) +{ + { // photos: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_photos = TableListElements( ctx, value.photos ); + if ( !cursor_photos.ok ) { return false; } + for ( int32_t i = 0; i < cursor_photos.count; i++ ) + { + { + const Photo * pointee = PhotoAt( ctx, cursor_photos[i] ); // photos + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( numbering.seen, (const void *) pointee, + (int64_t) ( numbering.count + 2 ), taken, slot ); // its index, if this is its first visit + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + } + else + { + TableNodeEntry node; + node.node = (const void *) pointee; + node.type_id = 0xf1a78dd2508964c3ull; // fnv1a64( "Photo" ) + node.type_slot = 52; // its slot in the unit's vocabulary (§3.3) + node.measure = &TableNodeMeasureThunk; + node.save = &TableNodeSaveThunk; + if ( !TableNumberingAppend( numbering, node ) ) { return false; } + if ( !PhotoNumber( ctx, numbering, *pointee ) ) { return false; } + TablePackMapClose( numbering.seen, (const void *) pointee, slot ); + } + } + } + } + } + { + const Photo * pointee = PhotoAt( ctx, value.cover ); // cover + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( numbering.seen, (const void *) pointee, + (int64_t) ( numbering.count + 2 ), taken, slot ); // its index, if this is its first visit + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + } + else + { + TableNodeEntry node; + node.node = (const void *) pointee; + node.type_id = 0xf1a78dd2508964c3ull; // fnv1a64( "Photo" ) + node.type_slot = 52; // its slot in the unit's vocabulary (§3.3) + node.measure = &TableNodeMeasureThunk; + node.save = &TableNodeSaveThunk; + if ( !TableNumberingAppend( numbering, node ) ) { return false; } + if ( !PhotoNumber( ctx, numbering, *pointee ) ) { return false; } + TablePackMapClose( numbering.seen, (const void *) pointee, slot ); + } + } + } + return true; +} + +// AlbumPackMeasure: the packed region bytes of everything Album POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t AlbumPackMeasure( const Ctx & ctx, TablePackMap & seen, const Album & value ) +{ + int64_t bytes = 0; + { // photos: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_photos = TableListElements( ctx, value.photos ); + if ( !cursor_photos.ok ) { return -1; } + for ( int32_t i = 0; i < cursor_photos.count; i++ ) + { + { + const Photo * pointee = PhotoAt( ctx, cursor_photos[i] ); // photos + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, 0, taken, slot ); + if ( entry == NULL ) { return -1; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return -1; } // a data cycle + } + else + { + int64_t inner = PhotoPackMeasure( ctx, seen, *pointee ); + if ( inner < 0 ) { return -1; } + TablePackMapClose( seen, (const void *) pointee, slot ); + bytes += TableAlignUp64( (int64_t) sizeof( Photo ) ) + inner; + } + } + } + } + } + { + const Photo * pointee = PhotoAt( ctx, value.cover ); // cover + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, 0, taken, slot ); + if ( entry == NULL ) { return -1; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return -1; } // a data cycle + } + else + { + int64_t inner = PhotoPackMeasure( ctx, seen, *pointee ); + if ( inner < 0 ) { return -1; } + TablePackMapClose( seen, (const void *) pointee, slot ); + bytes += TableAlignUp64( (int64_t) sizeof( Photo ) ) + inner; + } + } + } + return bytes; +} + +// AlbumPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool AlbumPackEdges( const Ctx & ctx, TablePackMap & seen, const Album & src, Album & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool AlbumPack( const Ctx & ctx, TablePackMap & seen, const Album & src, Album & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Album ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Album ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !AlbumExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return AlbumPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool AlbumPackEdges( const Ctx & ctx, TablePackMap & seen, const Album & src, Album & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + { // photos: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_photos = TableListElements( ctx, src.photos ); + if ( !cursor_photos.ok ) { return false; } + TableRef * placed_photos = (TableRef *) ( dst.photos.elements.value != 0 ? ( (uint8_t *) &dst.photos.elements + dst.photos.elements.value ) : NULL ); + for ( int32_t i = 0; i < cursor_photos.count; i++ ) + { + { + placed_photos[i].value = 0; // photos + const Photo * pointee = PhotoAt( ctx, cursor_photos[i] ); + if ( pointee != NULL ) + { + int64_t at = TableAlignUp64( used ); // where it WOULD land, if this is its first visit + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, at, taken, slot ); + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + placed_photos[i].value = (int64_t) ( ( base + entry->offset ) - (const uint8_t *) &placed_photos[i] ); // the one body it already has + } + else + { + if ( at + (int64_t) sizeof( Photo ) > capacity ) { return false; } + used = at + TableAlignUp64( (int64_t) sizeof( Photo ) ); + Photo * child = new ( base + at ) Photo; // lifetime only: the Pack below memcpy's the whole node over it + placed_photos[i].value = (int64_t) ( ( base + at ) - (const uint8_t *) &placed_photos[i] ); + if ( !PhotoPack( ctx, seen, *pointee, *child, base, capacity, used ) ) { return false; } + TablePackMapClose( seen, (const void *) pointee, slot ); + } + } + } + } + } + { + dst.cover.value = 0; // cover + const Photo * pointee = PhotoAt( ctx, src.cover ); + if ( pointee != NULL ) + { + int64_t at = TableAlignUp64( used ); // where it WOULD land, if this is its first visit + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, at, taken, slot ); + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + dst.cover.value = (int64_t) ( ( base + entry->offset ) - (const uint8_t *) &dst.cover ); // the one body it already has + } + else + { + if ( at + (int64_t) sizeof( Photo ) > capacity ) { return false; } + used = at + TableAlignUp64( (int64_t) sizeof( Photo ) ); + Photo * child = new ( base + at ) Photo; // lifetime only: the Pack below memcpy's the whole node over it + dst.cover.value = (int64_t) ( ( base + at ) - (const uint8_t *) &dst.cover ); + if ( !PhotoPack( ctx, seen, *pointee, *child, base, capacity, used ) ) { return false; } + TablePackMapClose( seen, (const void *) pointee, slot ); + } + } + } + return true; +} + +// ---- Album: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: AlbumBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Album is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct AlbumBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + AlbumBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~AlbumBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + AlbumBuilder( const AlbumBuilder & ) = delete; + AlbumBuilder & operator=( const AlbumBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Album * GetRoot() { return arena.locked ? NULL : (Album *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Album * AsConst() const { return (const Album *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool AlbumBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Album & root = *(const Album *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = AlbumPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = AlbumExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + Album * destination = new ( packed ) Album; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !AlbumPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Album on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// AlbumNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t AlbumNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + case 0xf1a78dd2508964c3ull: return TableAlignUp64( (int64_t) sizeof( Photo ) ); // Photo + default: break; + } + return -1; +} + +// AlbumNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void AlbumNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + switch ( type_id ) + { + case 0xf1a78dd2508964c3ull: { Photo * node = new ( at ) Photo; PhotoReset( *node ); break; } // Photo + default: break; + } +} + +// AlbumNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t AlbumNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + case 0xf1a78dd2508964c3ull: return TableAlignUp64( (int64_t) sizeof( Photo ) ); // Photo + default: break; + } + return 0; +} + +// AlbumNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t AlbumNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + switch ( type_id ) + { + case 0xf1a78dd2508964c3ull: return (uint32_t) worker.Alloc().ref.value; // Photo + default: break; + } + return 0; +} + +// AlbumNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void AlbumNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = AlbumNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? AlbumNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + case 0xf1a78dd2508964c3ull: PhotoLoadBody( r, *(Photo *) at ); break; // Photo + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool AlbumNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Album & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return AlbumNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t AlbumMeasureWire( const Ctx & ctx, const Album & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( AlbumNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = AlbumMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t AlbumSaveWire( const Ctx & ctx, const Album & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !AlbumNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = AlbumSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == AlbumMeasure( root ) +} + +inline int64_t AlbumMeasure( const Album * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return AlbumMeasureWire( ctx, *root, allocator ); +} + +inline int64_t AlbumSave( const Album * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return AlbumSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t AlbumMeasure( const AlbumBuilder & builder ) +{ + if ( builder.region != NULL ) { return AlbumMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return AlbumMeasureWire( ctx, *(const Album *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t AlbumSave( const AlbumBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return AlbumSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return AlbumSaveWire( ctx, *(const Album *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t AlbumMeasureMessage( const Album * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return AlbumMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t AlbumSaveMessage( const Album * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return AlbumSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t AlbumMeasureMessage( const AlbumBuilder & builder ) +{ + if ( builder.region != NULL ) { return AlbumMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return AlbumMeasureWire( ctx, *(const Album *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t AlbumSaveMessage( const AlbumBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return AlbumSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return AlbumSaveWire( ctx, *(const Album *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// AlbumLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t AlbumLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !AlbumWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = AlbumNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// AlbumLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Album * AlbumLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Album ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !AlbumWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = AlbumNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xd858c2cb7f1514ccull; + Album * root = new ( region ) Album; // lifetime only: LoadBody's first act is AlbumReset + AlbumReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = AlbumNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + AlbumNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + AlbumNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Album ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + AlbumLoadBody( r, nodes, *root ); + return root; +} + +// AlbumLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t AlbumLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !AlbumWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = AlbumNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// AlbumLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Album * AlbumLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Album ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !AlbumWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = AlbumNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xd858c2cb7f1514ccull; + Album * root = new ( region ) Album; // lifetime only: LoadBody's first act is AlbumReset + AlbumReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = AlbumNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + AlbumNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + AlbumNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Album ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + AlbumLoadBody( r, nodes, *root ); + return root; +} + +// AlbumLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool AlbumLoadBuilder( AlbumBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Album * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0xd858c2cb7f1514ccull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = AlbumNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + AlbumNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = AlbumLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- the cooked form: point at a cook (docs/SPEC-TABLES.md §7) ---- + +// PhotoOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Photo IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Photo * PhotoOpen( const void * bytes, uint64_t length ) +{ + return (const Photo *) TableCookOpen( bytes, length, (uint64_t) sizeof( Photo ), (uint64_t) alignof( Photo ) ); +} + +// AlbumOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH AlbumAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Album * AlbumOpen( const void * bytes, uint64_t length ) +{ + return (const Album *) TableCookOpen( bytes, length, (uint64_t) sizeof( Album ), (uint64_t) alignof( Album ) ); +} + +// ---- the cooked form: WRITE a cook (docs/SPEC-TABLES.md §7.6) ---- +// +// The bytes are `schema cook`'s, and the tool stays the reference: the two +// writers are held to one file, byte for byte, in both byte orders. A cook is +// content-addressed by (asset hash, build version), so two writers of one +// instance produce ONE artifact or the pair means nothing. + +inline void PhotoCookBody( uint8_t * at, const Photo & value, TableByteOrder order ); +template inline bool AlbumCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Album & value, TableByteOrder order ); + +inline void PhotoCookBody( uint8_t * at, const Photo & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.width, 4, order ); + table_cook_put( at + 4, (uint64_t) value.height, 4, order ); +} + +template inline bool AlbumCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Album & value, TableByteOrder order ) +{ + table_cook_put( at + 0, 0, 8, order ); // photos: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + if ( !table_cook_ref( region, at + 16, (const void *) PhotoAt( ctx, value.cover ), order ) ) { return false; } // cover + return true; +} + +template inline bool PhotoCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Photo & value, TableByteOrder order ); +template inline bool AlbumCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Album & value, TableByteOrder order ); + +// PhotoCookExtent: Photo's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool PhotoCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Photo & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// AlbumCookExtent: Album's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool AlbumCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Album & value, TableByteOrder order ) +{ + { // photos: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.photos ); + if ( !cursor.ok ) { return false; } + at = ( at + 7 ) & ~(int64_t) 7; // at alignof( TableRef ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 8; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + if ( !table_cook_ref( region, array + i * 8, (const void *) PhotoAt( ctx, cursor[i] ), order ) ) { return false; } + } + } + return true; +} + +// PhotoCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool PhotoCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Photo & value, TableByteOrder order ) +{ + PhotoCookBody( at, value, order ); + int64_t extent_at = 0; + return PhotoCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// AlbumCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool AlbumCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Album & value, TableByteOrder order ) +{ + if ( !AlbumCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return AlbumCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// PhotoCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Photo IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t PhotoCookMeasure( const Photo & value ) +{ + (void) value; + return 88; // 64 header + 8 data + 16 attribution +} + +// PhotoCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract PhotoMeasure/PhotoSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool PhotoCook( const Photo & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) PhotoCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 8, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + PhotoCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 72, 0, 8, order ); + table_cook_put( raw + 80, 0xf1a78dd2508964c3ull, 8, order ); + return true; +} + +// AlbumCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool AlbumCookLayout( const Ctx & ctx, const Album & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = AlbumExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + case 0xf1a78dd2508964c3ull: size = 8; node_align = 4; break; // Photo + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// AlbumCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t AlbumCookMeasureFrom( const Ctx & ctx, const Album & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( AlbumNumberFrom( ctx, numbering, root ) && AlbumCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// AlbumCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool AlbumCookFrom( const Ctx & ctx, const Album & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = AlbumNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && AlbumCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = AlbumCookNode( ctx, region, region.base, root, order ); + for ( int64_t k = 0; ok && k < numbering.count; k++ ) + { + uint8_t * at = region.base + region.offsets[k + 1]; + const void * node = numbering.entries[k].node; + switch ( numbering.entries[k].type_id ) + { + case 0xf1a78dd2508964c3ull: ok = PhotoCookNode( ctx, region, at, *(const Photo *) node, order ); break; // Photo + default: ok = false; break; + } + } + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0xd858c2cb7f1514ccull, 8, order ); // the root: fnv1a64( "Album" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// AlbumCookMeasure / AlbumCook over a REGION root — a locked builder's AsConst, a +// region AlbumLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t AlbumCookMeasure( const Album * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return AlbumCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool AlbumCook( const Album * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return AlbumCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t AlbumCookMeasure( const AlbumBuilder & builder ) +{ + if ( builder.region != NULL ) { return AlbumCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return AlbumCookMeasureFrom( ctx, *(const Album *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool AlbumCook( const AlbumBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return AlbumCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return AlbumCookFrom( ctx, *(const Album *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ---- relocatability, enforced: the wire is a pure length-prefixed +// stream AND the decoded storage is pointer-free — every closure type +// must stay trivially copyable and standard-layout, so instances can be +// memcpy'd, mmap'd, shared across processes, and walked through +// descriptor offsets. A failure here means a pointer, virtual or +// non-trivial member crept into generated storage. +// +// They ask the COMPILER ITSELF, which is what every C++ standard library +// answers the same two questions with — and it costs this header no +// include at all. +// A pointer FIELD is a TableRef — eight bytes and no address — so the +// property holds in BOTH forms: a fixed-size table is one relocatable +// struct, and a packed region is one relocatable block whose references +// are self-relative and therefore survive a plain memcpy. +static_assert( __is_trivially_copyable( Photo ), "Photo must stay relocatable" ); +static_assert( __is_standard_layout( Photo ), "Photo must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Album ), "Album must stay relocatable" ); +static_assert( __is_standard_layout( Album ), "Album must stay standard-layout for offsetof" ); + +// ---- the cook's layout contract (docs/SPEC-TABLES.md §20.3) ---- +// +// The compiler derived every number below from the declaration and folded it +// into the BUILD VERSION; these asserts are this compiler saying whether it +// agrees. The model is not self-evidently right — on 32-bit System V +// alignof(uint64_t) is 4, not 8 — which is precisely why it is asserted +// rather than assumed. +static_assert( sizeof( Photo ) == 8, "Photo's sizeof moved: the build version was taken over 8, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Photo ) == 4, "Photo's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Photo, width ) == 0, "Photo's field width moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Photo, height ) == 4, "Photo's field height moved: the build version was taken over offset 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Album ) == 24, "Album's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Album ) == 8, "Album's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Album, photos ) == 0, "Album's field photos moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Album, cover ) == 16, "Album's field cover moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); + +static_assert( alignof( TableRef ) <= kTableAlign, "Album.photos: an unbounded array's element alignment must fit the arena's" ); + +// ---- reflection descriptors (tables only, docs/SPEC-TABLES.md) ---- + +inline const TableTypeInfo * PhotoTableType(); +inline const TableTypeInfo * AlbumTableType(); +// The descriptors are CONSTANT-INITIALISED data, and a field's target is +// the ADDRESS of another descriptor. These declarations are what let a +// self- or mutually-referential graph — Node naming itself through *Node — +// be expressed as constant data instead of a lazy link, which could not +// have been written race-free OR recursion-safe. The whole reflection +// surface is therefore immutable: read it from any thread, any time. +extern const TableTypeInfo PhotoTableInfo; +extern const TableTypeInfo AlbumTableInfo; + +inline const TableFieldInfo PhotoTableFields[] = { + { "width", "width", "uint32", 0xdbdacd932fd1e9bfull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Photo, width ), (uint32_t) sizeof( Photo::width ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "height", "height", "uint32", 0x17720bf67d347222ull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Photo, height ), (uint32_t) sizeof( Photo::height ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo PhotoTableInfo = { "Photo", (uint32_t) sizeof( Photo ), 2, PhotoTableFields, +[]( void * p ) { PhotoReset( *(Photo *) p ); }, false }; +inline const TableTypeInfo * PhotoTableType() { return &PhotoTableInfo; } + +inline const TableFieldInfo AlbumTableFields[] = { + { "photos", "photos", "Photo", 0x40b1d94aff3ab130ull, 17, true, true, []( const void * slot ) -> const void * { return (const void *) PhotoAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) PhotoEmplace( worker, *(TableRef *) slot ); }, true, false, 0, (uint32_t) offsetof( Album, photos ), (uint32_t) sizeof( TableRef ), (uint32_t) offsetof( Album, photos.count ), 0xffffffffu, &PhotoTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "cover", "cover", "Photo", 0xaa19a78e404dea20ull, 17, false, true, []( const void * slot ) -> const void * { return (const void *) PhotoAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) PhotoEmplace( worker, *(TableRef *) slot ); }, false, false, 0, (uint32_t) offsetof( Album, cover ), (uint32_t) sizeof( TableRef ), 0xffffffffu, 0xffffffffu, &PhotoTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo AlbumTableInfo = { "Album", (uint32_t) sizeof( Album ), 2, AlbumTableFields, +[]( void * p ) { AlbumReset( *(Album *) p ); }, true }; +inline const TableTypeInfo * AlbumTableType() { return &AlbumTableInfo; } + +// ---- the text form (docs/SPEC-TABLES.md §16) ---- + +// Photo in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// SharedTable.cpp; link it to use them. +bool PhotoFromJson( Photo & value, const char * text, int64_t bytes, TableReport * report ); +int64_t PhotoToJsonMeasure( const Photo & value ); +int64_t PhotoToJson( const Photo & value, char * buffer, int64_t capacity ); + +// Album in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in SharedTable.cpp; link it to use them. +bool AlbumFromJson( AlbumBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t AlbumToJsonMeasure( const Album * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t AlbumToJson( const Album * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +} // namespace listdemo From d17a3d72689881fee1895e9e7db29f475e03fc8e Mon Sep 17 00:00:00 2001 From: Glenn Fiedler Date: Sat, 5 Sep 2026 08:30:12 -0700 Subject: [PATCH 5/7] =?UTF-8?q?tables:=20re-pin=20the=20C++=20table=20gold?= =?UTF-8?q?ens=20under=20the=20list=20adapters=20and=20the=20=C2=A78.1=20c?= =?UTF-8?q?olumns?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every list-free unit gains the JSON walk's three list adapter stubs, as it carries the map's, and the map units take the descriptor columns §8.1 now names, the TableRefuseReason enum and the extent carve. Co-Authored-By: Claude Fable 5.1 --- testdata/golden/tables/blobs/AssetsTable.cpp | 52 +- testdata/golden/tables/block/PaddedTable.cpp | 52 +- testdata/golden/tables/block/RenderTable.cpp | 52 +- .../golden/tables/blockhome/DataTable.cpp | 52 +- .../golden/tables/blockhome/FrameTable.cpp | 52 +- .../golden/tables/examples/GuardedTable.cpp | 52 +- .../golden/tables/examples/KeyedTable.cpp | 52 +- .../golden/tables/examples/NestedTable.cpp | 52 +- testdata/golden/tables/examples/PackTable.cpp | 52 +- .../golden/tables/examples/RangesTable.cpp | 52 +- .../golden/tables/examples/TablesTable.cpp | 52 +- testdata/golden/tables/examples/WideTable.cpp | 52 +- testdata/golden/tables/maps/DepthTable.cpp | 88 ++- testdata/golden/tables/maps/DepthTable.h | 488 ++++++++------- testdata/golden/tables/maps/FleetTable.cpp | 88 ++- testdata/golden/tables/maps/FleetTable.h | 580 ++++++++++-------- testdata/golden/tables/maps/RowsTable.cpp | 88 ++- testdata/golden/tables/maps/RowsTable.h | 469 +++++++------- .../golden/tables/messages/MessagesTable.cpp | 52 +- .../golden/tables/pointers/GraphTable.cpp | 52 +- .../golden/tables/pointers/MarksTable.cpp | 52 +- .../golden/tables/pointers/PartsTable.cpp | 52 +- .../golden/tables/scalars/ScalarsTable.cpp | 52 +- testdata/golden/tables/stream/StreamTable.cpp | 52 +- 24 files changed, 1856 insertions(+), 881 deletions(-) diff --git a/testdata/golden/tables/blobs/AssetsTable.cpp b/testdata/golden/tables/blobs/AssetsTable.cpp index e1620d36e..86b2bb020 100644 --- a/testdata/golden/tables/blobs/AssetsTable.cpp +++ b/testdata/golden/tables/blobs/AssetsTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2750,6 +2767,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace blobdemo #endif // BLOBDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/block/PaddedTable.cpp b/testdata/golden/tables/block/PaddedTable.cpp index 37fc3c912..de8b50825 100644 --- a/testdata/golden/tables/block/PaddedTable.cpp +++ b/testdata/golden/tables/block/PaddedTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace blockdemo #endif // BLOCKDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/block/RenderTable.cpp b/testdata/golden/tables/block/RenderTable.cpp index 72a6a2f46..63c5f85e2 100644 --- a/testdata/golden/tables/block/RenderTable.cpp +++ b/testdata/golden/tables/block/RenderTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace blockdemo #endif // BLOCKDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/blockhome/DataTable.cpp b/testdata/golden/tables/blockhome/DataTable.cpp index 889b5b8c8..fc9089196 100644 --- a/testdata/golden/tables/blockhome/DataTable.cpp +++ b/testdata/golden/tables/blockhome/DataTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace blockhome #endif // BLOCKHOME_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/blockhome/FrameTable.cpp b/testdata/golden/tables/blockhome/FrameTable.cpp index 42e9edab9..66de719df 100644 --- a/testdata/golden/tables/blockhome/FrameTable.cpp +++ b/testdata/golden/tables/blockhome/FrameTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace blockhome #endif // BLOCKHOME_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/GuardedTable.cpp b/testdata/golden/tables/examples/GuardedTable.cpp index 5e0e1dd5d..ad26d6936 100644 --- a/testdata/golden/tables/examples/GuardedTable.cpp +++ b/testdata/golden/tables/examples/GuardedTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/KeyedTable.cpp b/testdata/golden/tables/examples/KeyedTable.cpp index 119f639c0..767dababf 100644 --- a/testdata/golden/tables/examples/KeyedTable.cpp +++ b/testdata/golden/tables/examples/KeyedTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/NestedTable.cpp b/testdata/golden/tables/examples/NestedTable.cpp index 837d9ddf5..e5f0acf7d 100644 --- a/testdata/golden/tables/examples/NestedTable.cpp +++ b/testdata/golden/tables/examples/NestedTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/PackTable.cpp b/testdata/golden/tables/examples/PackTable.cpp index 51aefb058..d33ec128a 100644 --- a/testdata/golden/tables/examples/PackTable.cpp +++ b/testdata/golden/tables/examples/PackTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/RangesTable.cpp b/testdata/golden/tables/examples/RangesTable.cpp index ef14d69c6..1e4129090 100644 --- a/testdata/golden/tables/examples/RangesTable.cpp +++ b/testdata/golden/tables/examples/RangesTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/TablesTable.cpp b/testdata/golden/tables/examples/TablesTable.cpp index eee81f0e6..970077ae9 100644 --- a/testdata/golden/tables/examples/TablesTable.cpp +++ b/testdata/golden/tables/examples/TablesTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/WideTable.cpp b/testdata/golden/tables/examples/WideTable.cpp index 6c69959c8..ca0e06912 100644 --- a/testdata/golden/tables/examples/WideTable.cpp +++ b/testdata/golden/tables/examples/WideTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/maps/DepthTable.cpp b/testdata/golden/tables/maps/DepthTable.cpp index 703a26c0b..4265104a5 100644 --- a/testdata/golden/tables/maps/DepthTable.cpp +++ b/testdata/golden/tables/maps/DepthTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2731,14 +2748,33 @@ inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * inf // ---- json graph walk: end ---- +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + // ---- json map walk: begin ---- -inline bool TableJsonIsMap( const TableFieldInfo * f ) { return f->entry != NULL; } +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} // the entry's two rows: fields[0] IS the key and fields[1] IS the value, which // is what makes a user's own table of pairs the same bytes (§2.8) -inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->entry->fields[0]; } -inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->entry->fields[1]; } +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } @@ -2798,16 +2834,17 @@ inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const // A region holds them in that order already, so this is the array in place. inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) { - const int32_t count = f->map_count( slot ); + const int32_t count = TableJsonExtentCount( slot ); if ( count == 0 ) { out.raw( "{}", 2 ); return true; } const TableFieldInfo * key = TableJsonMapKeyField( f ); const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); out.put( '{' ); for ( int32_t i = 0; i < count; i++ ) { if ( i > 0 ) { out.put( ',' ); } out.line( depth + 1 ); - const void * entry = f->map_at( slot, i ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); TableJsonWriteMapKey( out, entry, key ); out.raw( ": ", 2 ); if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } @@ -2896,15 +2933,15 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf } if ( !fits ) { in.report->kind_mismatch++; place = false; } } - const int32_t before = f->map_count( (const void *) slot ); - void * entry = place ? f->map_insert( *graph->worker, slot, token, token_length, key_value ) : NULL; + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; if ( place && entry == NULL ) { // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the // wire's rule, because a clamped key is a merged entry (§2.8). in.report->clamped++; } - else if ( entry != NULL && f->map_count( (const void *) slot ) == before ) + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) { in.report->duplicate++; // last-wins, the object rule inside the map } @@ -2955,6 +2992,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf // ---- json map walk: end ---- +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace mapdemo #endif // MAPDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/maps/DepthTable.h b/testdata/golden/tables/maps/DepthTable.h index 78653da82..347219edb 100644 --- a/testdata/golden/tables/maps/DepthTable.h +++ b/testdata/golden/tables/maps/DepthTable.h @@ -110,6 +110,18 @@ struct TableReport TableMessageReason reason = newer_form; }; + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; // ---- reflection (tables only, docs/SPEC-TABLES.md) ---- // // Static field descriptors for every type in the table closure: name, wire @@ -225,18 +237,13 @@ struct TableFieldInfo // a function pointer at compile time; the arms themselves are a static // inside it). NULL for every other kind. const TableUnionInfo * (*arms)(); - // a MAP (docs/SPEC-TABLES.md §2.8): the generated ENTRY's descriptor — - // fields[0] is the key and fields[1] the value — and the three the ONE - // text walk cannot spell for itself, because TableMap is a type - // it has no name for. NULL on every field that is not a map. - const TableTypeInfo * entry; - int32_t ( * map_count )( const void * slot ); - const void * ( * map_at )( const void * slot, int32_t index ); - // place one entry BY KEY and hand back the entry, at its defaults: a - // string key comes in as the bytes and the length, an integer key as - // the value, and NULL is NOT INSERTED — a key past the bound, or an - // arena that could not carve another segment. - void * ( * map_insert )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded }; @@ -1292,8 +1299,8 @@ struct TableWorker return blob; } - // RAW, ZEROED storage of the bytes asked for, at the alignment asked for — a MAP's builder head and its - // entry segments (docs/SPEC-TABLES.md §2.8). It is not a node: it carries + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries // no type id, takes no index and has no Reset, so it goes through the same // slab and span the blob path uses rather than through Alloc. uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) @@ -1577,12 +1584,6 @@ static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts th // resolving through it yields NULL and can never fabricate the root. static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; -// What a node's storage answers when the FRAMING ITSELF is refused rather than -// merely unnameable: a map whose N cannot fit in its L (docs/SPEC-TABLES.md -// §2.8). An unnameable type id commands no storage and keeps its index; this -// one makes the whole measure answer -1 (§7.6). -static const int64_t kTableNodeRefused = -2; - // ---- the numbering, on the SAVE side ---- // // One entry per reachable node in FIRST-VISIT order, so entry k is node index @@ -1780,9 +1781,9 @@ struct TableNodeDirEntry uint64_t type_id; }; -// a map's extent cursor, defined with the map runtime (docs/SPEC-TABLES.md -// §2.8); the node map names it only through a pointer. -struct TableMapCarve; +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; // TableNodeMap is what a pointer slot resolves through while a body decodes. struct TableNodeMap @@ -1795,17 +1796,21 @@ struct TableNodeMap // takes the SELF-RELATIVE delta so a deref is one add, and the tool's // builder path takes the node's ARENA OFFSET (§6.3). bool arena = false; - // WHERE A MAP'S ENTRIES LAND while this node's body decodes - // (docs/SPEC-TABLES.md §2.8): the node's own extent on the region path - // and the builder's arena on the tool's. It is MUTABLE because the - // cursor belongs to ONE node's decode and the dispatch that owns that - // node holds the map by const reference, exactly as it did before maps - // existed — the decoder's signature does not move for a construct it - // may not carry. - mutable TableMapCarve * carve = NULL; - // and the TOOL's path's allocation front, set once: there a map's - // entries are the builder's arena's rather than a node's extent. + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; }; // TableNodeResolve places one node index in a pointer slot, and every failure @@ -1949,6 +1954,90 @@ inline bool TableNodeScanWhole( TableNodeScan & s ) #endif // MAPDEMO_SCHEMA_TABLE_ARENA +#ifndef MAPDEMO_SCHEMA_TABLE_EXTENT +#define MAPDEMO_SCHEMA_TABLE_EXTENT + +namespace mapdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace mapdemo + +#endif // MAPDEMO_SCHEMA_TABLE_EXTENT + #ifndef MAPDEMO_SCHEMA_TABLE_MAP #define MAPDEMO_SCHEMA_TABLE_MAP @@ -2379,12 +2468,6 @@ inline TableMapEach TableMapEachOf( const TableArena & arena, const Table return each; } -// AN UNREACHED SLOT MUST HOLD NO MAP WITH ENTRIES IN IT (§2.8, §7.6). An empty -// map takes no bytes, so a record whose extent measures ZERO is a record whose -// every by-value map is empty; a measure that REFUSED answers non-zero here -// too, and refusing on it is the same answer one level up. -inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } - // ---- the LOAD side: where a decoded entry lands (§2.8) ---- // // THE READER TRUSTS NOTHING and spends one compare per entry. Every load path @@ -2394,16 +2477,9 @@ inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } // out of the holder node's own extent, and the TOOL's path appends into the // builder's arena, and the decoder above them cannot tell which it has. -// TableMapCarve is a node's extent cursor, PRE-ORDER: a map's whole entry -// array first, then, entry by entry in key order, the arrays of any map an -// entry's value holds by value. The cursor is the node map's, because the -// generated decoder is threaded with that and not with a region. -struct TableMapCarve -{ - uint8_t * at = NULL; // the region path: the node's extent, unspent - int64_t left = 0; - TableWorker * worker = NULL; // the TOOL's path: entries come from the arena -}; +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. // TableMapFill is one map field being decoded: where the next entry lands, and // the entry that last LANDED, which is what the ascending check compares @@ -2528,10 +2604,9 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // LoadMeasure's term for a map is N x sizeof( Entry ) rounded to // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value -// holds a map of its own, the entries' headers under it. The caller owns the -// allocation precisely so it can refuse a number it did not expect. -typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ); - +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect, +// and a refusal carries its reason (§6.5). // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2539,8 +2614,8 @@ typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, i static const int64_t kTableMapEntryFloor = 2; inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, - int64_t entry_size, int64_t entry_align, TableMapWireExtentFn inner, - const TableIdTable * ids ) + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; TableReader r( body, length, &scratch, ids ); @@ -2549,7 +2624,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } const int64_t rest = length - r.offset; - if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { return false; } // an N the map's L cannot carry + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); at += (int64_t) n * entry_size; if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term @@ -2557,49 +2632,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & { uint64_t elem = 0; if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// the same framing walk over an ARRAY OF TABLES that is not a map: its -// elements' own maps are part of this node's extent too -inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each -// length-prefixed element (docs/SPEC-TABLES.md §3.2) -inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t key = 0; - if ( !r.getleb( key ) ) { return true; } - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } r.offset += (int64_t) elem; } return true; @@ -3067,7 +3100,7 @@ inline void SquadRosterEntryReset( SquadRosterEntry & value ) inline void SquadReset( Squad & value ) { - value.roster.entries.value = 0; // map[uint8]Item — empty + value.roster.entries.value = 0; // map[uint8]Item: empty value.roster.count = 0; value.roster.padding = 0; } @@ -3960,10 +3993,10 @@ inline bool DepthLoadBody( TableReader & r, const TableNodeMap & nodes, Depth & } } -// SquadWireExtent: the extent Squad's maps command, from the FRAMING alone. +// SquadWireExtent: the extent Squad's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool SquadWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool SquadWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; // the scan's framing damage is the LOAD's to report TableReader r( body, length, &scratch, ids ); @@ -3982,17 +4015,17 @@ inline bool SquadWireExtent( const uint8_t * body, int64_t length, int64_t & at, if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( SquadRosterEntry ), (int64_t) alignof( SquadRosterEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( SquadRosterEntry ), (int64_t) alignof( SquadRosterEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( !r.skip( field_kind ) ) { return true; } } } -// SquadMapExtentAt: the node extent Squad's maps take, PRE-ORDER, advancing the -// running offset exactly as SquadMapPack advances it (docs/SPEC-TABLES.md §2.8). +// SquadExtentAt: the node extent Squad's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as SquadExtentPack advances it (§2.8, §2.9). template -inline bool SquadMapExtentAt( const Ctx & ctx, const Squad & value, int64_t & at ) +inline bool SquadExtentAt( const Ctx & ctx, const Squad & value, int64_t & at ) { { TableMapCursor cursor = TableMapOrder( ctx, value.roster ); @@ -4007,18 +4040,18 @@ inline bool SquadMapExtentAt( const Ctx & ctx, const Squad & value, int64_t & at // the whole extent of one node, from a fresh offset: what a pack reserves // for it beside the record's own storage. template -inline int64_t SquadMapExtent( const Ctx & ctx, const Squad & value ) +inline int64_t SquadExtent( const Ctx & ctx, const Squad & value ) { int64_t at = 0; - if ( !SquadMapExtentAt( ctx, value, at ) ) { return -1; } + if ( !SquadExtentAt( ctx, value, at ) ) { return -1; } return at; } -// SquadMapPack: carve Squad's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset SquadMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// SquadExtentPack: carve Squad's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset SquadExtentAt advances (§2.8, §2.9). template -inline bool SquadMapPack( const Ctx & ctx, const Squad & src, Squad & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool SquadExtentPack( const Ctx & ctx, const Squad & src, Squad & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { { TableMapCursor cursor = TableMapOrder( ctx, src.roster ); @@ -4040,10 +4073,10 @@ inline bool SquadMapPack( const Ctx & ctx, const Squad & src, Squad & dst, uint8 return true; } -// DepthWireExtent: the extent Depth's maps command, from the FRAMING alone. +// DepthWireExtent: the extent Depth's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool DepthWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool DepthWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; // the scan's framing damage is the LOAD's to report TableReader r( body, length, &scratch, ids ); @@ -4056,34 +4089,34 @@ inline bool DepthWireExtent( const uint8_t * body, int64_t length, int64_t & at, const uint64_t field_id = ids->at( field_ref ); if ( !r.has( 1 ) ) { return true; } uint8_t field_kind = r.get8(); - if ( field_id == 0x1a08aa1921ca5cafull && field_kind == 13 ) // one: a nesting that holds a map + if ( field_id == 0x1a08aa1921ca5cafull && field_kind == 13 ) // one: a nesting that holds a list or a map { uint64_t nested_len = 0; if ( !r.getleb( nested_len ) || !r.room( nested_len ) ) { return true; } const uint8_t * nested_body = r.buffer + r.offset; r.offset += (int64_t) nested_len; - if ( !SquadWireExtent( nested_body, (int64_t) nested_len, at, ids ) ) { return false; } + if ( !SquadWireExtent( nested_body, (int64_t) nested_len, at, ids, reason ) ) { return false; } continue; } - if ( field_id == 0x1f6459a2cea1fc02ull && field_kind == 14 ) // many: a nesting that holds a map + if ( field_id == 0x1f6459a2cea1fc02ull && field_kind == 14 ) // many: a nesting that holds a list or a map { uint64_t nested_len = 0; if ( !r.getleb( nested_len ) || !r.room( nested_len ) ) { return true; } const uint8_t * nested_body = r.buffer + r.offset; r.offset += (int64_t) nested_len; - if ( !TableWireExtentElements( nested_body, (int64_t) nested_len, at, &SquadWireExtent, ids ) ) { return false; } + if ( !TableWireExtentElements( nested_body, (int64_t) nested_len, at, &SquadWireExtent, ids, reason ) ) { return false; } continue; } - if ( field_id == 0x70551ff29550f15dull && field_kind == 16 ) // keyed: a nesting that holds a map + if ( field_id == 0x70551ff29550f15dull && field_kind == 16 ) // keyed: a nesting that holds a list or a map { uint64_t nested_len = 0; if ( !r.getleb( nested_len ) || !r.room( nested_len ) ) { return true; } const uint8_t * nested_body = r.buffer + r.offset; r.offset += (int64_t) nested_len; - if ( !TableWireExtentKeyed( nested_body, (int64_t) nested_len, at, &SquadWireExtent, ids ) ) { return false; } + if ( !TableWireExtentKeyed( nested_body, (int64_t) nested_len, at, &SquadWireExtent, ids, reason ) ) { return false; } continue; } - if ( field_id == 0xe756c0190570ccb5ull && field_kind == 15 ) // arm: a union arm that holds a map + if ( field_id == 0xe756c0190570ccb5ull && field_kind == 15 ) // arm: a union arm that holds a list or a map { uint64_t arm_ref = 0; if ( !r.getleb( arm_ref ) ) { return true; } @@ -4098,7 +4131,7 @@ inline bool DepthWireExtent( const uint8_t * body, int64_t length, int64_t & at, r.offset += (int64_t) arm_len; switch ( arm_id ) { - case 0xd5c2bb95d63e6331ull: if ( !SquadWireExtent( arm_body, (int64_t) arm_len, at, ids ) ) { return false; } break; // squad + case 0xd5c2bb95d63e6331ull: if ( !SquadWireExtent( arm_body, (int64_t) arm_len, at, ids, reason ) ) { return false; } break; // squad default: break; // an arm this reader cannot name reads None } continue; @@ -4107,31 +4140,31 @@ inline bool DepthWireExtent( const uint8_t * body, int64_t length, int64_t & at, } } -// DepthMapExtentAt: the node extent Depth's maps take, PRE-ORDER, advancing the -// running offset exactly as DepthMapPack advances it (docs/SPEC-TABLES.md §2.8). +// DepthExtentAt: the node extent Depth's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as DepthExtentPack advances it (§2.8, §2.9). template -inline bool DepthMapExtentAt( const Ctx & ctx, const Depth & value, int64_t & at ) +inline bool DepthExtentAt( const Ctx & ctx, const Depth & value, int64_t & at ) { { // one (nested by value) - if ( !SquadMapExtentAt( ctx, value.one, at ) ) { return false; } + if ( !SquadExtentAt( ctx, value.one, at ) ) { return false; } } for ( int32_t i = 0; i < value.many_count && i < 3; i++ ) // many { - if ( !SquadMapExtentAt( ctx, value.many[i], at ) ) { return false; } + if ( !SquadExtentAt( ctx, value.many[i], at ) ) { return false; } } for ( int32_t i = value.many_count; i < 3; i++ ) // many: the slots the walk does not reach (§7.6) { - if ( !TableMapUnreachedEmpty( SquadMapExtent( ctx, value.many[i] ) ) ) { return false; } + if ( !TableExtentUnreachedEmpty( SquadExtent( ctx, value.many[i] ) ) ) { return false; } } for ( int32_t i = 0; i < 2; i++ ) // keyed { - if ( !SquadMapExtentAt( ctx, value.keyed.slots[i], at ) ) { return false; } + if ( !SquadExtentAt( ctx, value.keyed.slots[i], at ) ) { return false; } } switch ( value.arm.type ) // arm: the set arm is the edge { case ForceType::Squad: { - if ( !SquadMapExtentAt( ctx, value.arm.squad, at ) ) { return false; } + if ( !SquadExtentAt( ctx, value.arm.squad, at ) ) { return false; } break; } default: break; @@ -4142,39 +4175,39 @@ inline bool DepthMapExtentAt( const Ctx & ctx, const Depth & value, int64_t & at // the whole extent of one node, from a fresh offset: what a pack reserves // for it beside the record's own storage. template -inline int64_t DepthMapExtent( const Ctx & ctx, const Depth & value ) +inline int64_t DepthExtent( const Ctx & ctx, const Depth & value ) { int64_t at = 0; - if ( !DepthMapExtentAt( ctx, value, at ) ) { return -1; } + if ( !DepthExtentAt( ctx, value, at ) ) { return -1; } return at; } -// DepthMapPack: carve Depth's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset DepthMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// DepthExtentPack: carve Depth's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset DepthExtentAt advances (§2.8, §2.9). template -inline bool DepthMapPack( const Ctx & ctx, const Depth & src, Depth & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool DepthExtentPack( const Ctx & ctx, const Depth & src, Depth & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { { // one (nested by value) - if ( !SquadMapPack( ctx, src.one, dst.one, extent, at, capacity ) ) { return false; } + if ( !SquadExtentPack( ctx, src.one, dst.one, extent, at, capacity ) ) { return false; } } for ( int32_t i = 0; i < src.many_count && i < 3; i++ ) // many { - if ( !SquadMapPack( ctx, src.many[i], dst.many[i], extent, at, capacity ) ) { return false; } + if ( !SquadExtentPack( ctx, src.many[i], dst.many[i], extent, at, capacity ) ) { return false; } } for ( int32_t i = src.many_count; i < 3; i++ ) // many: the slots the walk does not reach (§7.6) { - if ( !TableMapUnreachedEmpty( SquadMapExtent( ctx, src.many[i] ) ) ) { return false; } + if ( !TableExtentUnreachedEmpty( SquadExtent( ctx, src.many[i] ) ) ) { return false; } } for ( int32_t i = 0; i < 2; i++ ) // keyed { - if ( !SquadMapPack( ctx, src.keyed.slots[i], dst.keyed.slots[i], extent, at, capacity ) ) { return false; } + if ( !SquadExtentPack( ctx, src.keyed.slots[i], dst.keyed.slots[i], extent, at, capacity ) ) { return false; } } switch ( src.arm.type ) // arm: the set arm is the edge { case ForceType::Squad: { - if ( !SquadMapPack( ctx, src.arm.squad, dst.arm.squad, extent, at, capacity ) ) { return false; } + if ( !SquadExtentPack( ctx, src.arm.squad, dst.arm.squad, extent, at, capacity ) ) { return false; } break; } default: break; @@ -4323,7 +4356,7 @@ inline bool SquadPack( const Ctx & ctx, TablePackMap & seen, const Squad & src, int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Squad ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !SquadMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !SquadExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return SquadPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -4422,7 +4455,7 @@ inline bool DepthPack( const Ctx & ctx, TablePackMap & seen, const Depth & src, int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Depth ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !DepthMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !DepthExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return DepthPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -4533,7 +4566,7 @@ inline bool SquadBuilder::Lock() below = SquadPackMeasure( ctx, seen, root ); } if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it - int64_t root_extent = SquadMapExtent( ctx, root ); + int64_t root_extent = SquadExtent( ctx, root ); if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ) + below; // the AUTHORING path may allocate (§6.5), and it does so through the @@ -4631,10 +4664,11 @@ inline uint32_t SquadNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t // already owns. inline void SquadNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) { - // the node's own EXTENT, where its maps' entry arrays are carved from, - // PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8). The tool's - // path carries a worker instead: there the entries are the arena's. - TableMapCarve carve; + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; carve.worker = nodes.worker; if ( carve.worker == NULL ) { @@ -4783,7 +4817,7 @@ inline int64_t SquadSaveMessage( const SquadBuilder & builder, uint8_t * buffer, // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t SquadLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t SquadLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; TableIdTable ids_table; @@ -4796,8 +4830,9 @@ inline int64_t SquadLoadMeasure( const uint8_t * wire_file, int64_t wire_file_by const uint8_t * const wire = wire_file + 1; const int64_t wire_bytes = body_bytes; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -4807,7 +4842,7 @@ inline int64_t SquadLoadMeasure( const uint8_t * wire_file, int64_t wire_file_by { records++; int64_t storage = SquadNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -4851,8 +4886,9 @@ inline const Squad * SquadLoad( uint8_t * region, int64_t region_bytes, const ui int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); int64_t records = 0; { @@ -4935,7 +4971,7 @@ inline const Squad * SquadLoad( uint8_t * region, int64_t region_bytes, const ui // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Squad ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -4951,7 +4987,7 @@ inline const Squad * SquadLoad( uint8_t * region, int64_t region_bytes, const ui // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t SquadLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t SquadLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } @@ -4959,8 +4995,9 @@ inline int64_t SquadLoadMeasure( const TableVocabulary & vocabulary, const uint8 const uint8_t * const wire = message + 1; const int64_t wire_bytes = message_bytes - 1; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -4970,7 +5007,7 @@ inline int64_t SquadLoadMeasure( const TableVocabulary & vocabulary, const uint8 { records++; int64_t storage = SquadNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -5005,8 +5042,9 @@ inline const Squad * SquadLoadMessage( uint8_t * region, int64_t region_bytes, c int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); int64_t records = 0; { @@ -5089,7 +5127,7 @@ inline const Squad * SquadLoadMessage( uint8_t * region, int64_t region_bytes, c // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Squad ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -5144,7 +5182,7 @@ inline bool SquadLoadBuilder( SquadBuilder & builder, const uint8_t * wire_file, nodes.entries = directory; nodes.count = records + 1; nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here - nodes.worker = &builder.main; // and a map's entries are the arena's, not a node extent's (§2.8) + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) { TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); int64_t k = 0; @@ -5183,10 +5221,14 @@ inline bool SquadLoadBuilder( SquadBuilder & builder, const uint8_t * wire_file, } TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.worker = &builder.main; nodes.carve = &root_carve; bool ok = SquadLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; allocator.free( allocator.context, directory ); return ok; } @@ -5272,7 +5314,7 @@ inline bool DepthBuilder::Lock() below = DepthPackMeasure( ctx, seen, root ); } if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it - int64_t root_extent = DepthMapExtent( ctx, root ); + int64_t root_extent = DepthExtent( ctx, root ); if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Depth ) ) + root_extent ) + below; // the AUTHORING path may allocate (§6.5), and it does so through the @@ -5370,10 +5412,11 @@ inline uint32_t DepthNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t // already owns. inline void DepthNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) { - // the node's own EXTENT, where its maps' entry arrays are carved from, - // PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8). The tool's - // path carries a worker instead: there the entries are the arena's. - TableMapCarve carve; + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; carve.worker = nodes.worker; if ( carve.worker == NULL ) { @@ -5522,7 +5565,7 @@ inline int64_t DepthSaveMessage( const DepthBuilder & builder, uint8_t * buffer, // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t DepthLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t DepthLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; TableIdTable ids_table; @@ -5535,8 +5578,9 @@ inline int64_t DepthLoadMeasure( const uint8_t * wire_file, int64_t wire_file_by const uint8_t * const wire = wire_file + 1; const int64_t wire_bytes = body_bytes; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Depth ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -5546,7 +5590,7 @@ inline int64_t DepthLoadMeasure( const uint8_t * wire_file, int64_t wire_file_by { records++; int64_t storage = DepthNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -5590,8 +5634,9 @@ inline const Depth * DepthLoad( uint8_t * region, int64_t region_bytes, const ui int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Depth ) ) + root_extent ); int64_t records = 0; { @@ -5674,7 +5719,7 @@ inline const Depth * DepthLoad( uint8_t * region, int64_t region_bytes, const ui // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Depth ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -5690,7 +5735,7 @@ inline const Depth * DepthLoad( uint8_t * region, int64_t region_bytes, const ui // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t DepthLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t DepthLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } @@ -5698,8 +5743,9 @@ inline int64_t DepthLoadMeasure( const TableVocabulary & vocabulary, const uint8 const uint8_t * const wire = message + 1; const int64_t wire_bytes = message_bytes - 1; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Depth ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -5709,7 +5755,7 @@ inline int64_t DepthLoadMeasure( const TableVocabulary & vocabulary, const uint8 { records++; int64_t storage = DepthNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -5744,8 +5790,9 @@ inline const Depth * DepthLoadMessage( uint8_t * region, int64_t region_bytes, c int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Depth ) ) + root_extent ); int64_t records = 0; { @@ -5828,7 +5875,7 @@ inline const Depth * DepthLoadMessage( uint8_t * region, int64_t region_bytes, c // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Depth ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -5883,7 +5930,7 @@ inline bool DepthLoadBuilder( DepthBuilder & builder, const uint8_t * wire_file, nodes.entries = directory; nodes.count = records + 1; nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here - nodes.worker = &builder.main; // and a map's entries are the arena's, not a node extent's (§2.8) + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) { TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); int64_t k = 0; @@ -5922,10 +5969,14 @@ inline bool DepthLoadBuilder( DepthBuilder & builder, const uint8_t * wire_file, } TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.worker = &builder.main; nodes.carve = &root_carve; bool ok = DepthLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; allocator.free( allocator.context, directory ); return ok; } @@ -6003,9 +6054,9 @@ inline void SquadRosterEntryCookBody( uint8_t * at, const SquadRosterEntry & val template inline bool SquadCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Squad & value, TableByteOrder order ) { - (void) ctx; (void) region; // no reference below this node: the class was decided by a pointer elsewhere in its closure + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure (void) value; - table_cook_put( at + 0, 0, 8, order ); // roster: the entry array's delta, filled by the extent writer + table_cook_put( at + 0, 0, 8, order ); // roster: the array's delta, filled by the extent writer table_cook_put( at + 8, 0, 4, order ); // and its count return true; } @@ -6041,23 +6092,25 @@ template inline bool DepthCookBody( const Ctx & ctx, const TableC return true; } -template inline bool SquadRosterEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const SquadRosterEntry & value, TableByteOrder order ); -template inline bool SquadCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Squad & value, TableByteOrder order ); -template inline bool DepthCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Depth & value, TableByteOrder order ); +template inline bool SquadRosterEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const SquadRosterEntry & value, TableByteOrder order ); +template inline bool SquadCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Squad & value, TableByteOrder order ); +template inline bool DepthCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Depth & value, TableByteOrder order ); -// SquadRosterEntryCookMaps: SquadRosterEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool SquadRosterEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const SquadRosterEntry & value, TableByteOrder order ) +// SquadRosterEntryCookExtent: SquadRosterEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SquadRosterEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const SquadRosterEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// SquadCookMaps: Squad's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool SquadCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Squad & value, TableByteOrder order ) +// SquadCookExtent: Squad's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SquadCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Squad & value, TableByteOrder order ) { - (void) region; // a map's entries carry their own references through their own bodies + (void) region; // a table element's and an entry's references resolve through their own bodies { // roster TableMapCursor cursor = TableMapOrder( ctx, value.roster ); if ( !cursor.ok ) { return false; } @@ -6076,45 +6129,46 @@ template inline bool SquadCookMaps( const Ctx & ctx, const TableC return true; } -// DepthCookMaps: Depth's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool DepthCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Depth & value, TableByteOrder order ) +// DepthCookExtent: Depth's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool DepthCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Depth & value, TableByteOrder order ) { - (void) region; // a map's entries carry their own references through their own bodies - if ( !SquadCookMaps( ctx, region, extent, at, record + 0, value.one, order ) ) { return false; } // one + (void) region; // a table element's and an entry's references resolve through their own bodies + if ( !SquadCookExtent( ctx, region, extent, at, record + 0, value.one, order ) ) { return false; } // one for ( int32_t i = 0; i < ( value.many_count < 3 ? value.many_count : 3 ); i++ ) // many { - if ( !SquadCookMaps( ctx, region, extent, at, record + 16 + i * 16, value.many[i], order ) ) { return false; } + if ( !SquadCookExtent( ctx, region, extent, at, record + 16 + i * 16, value.many[i], order ) ) { return false; } } for ( int32_t i = 0; i < 2; i++ ) // keyed { - if ( !SquadCookMaps( ctx, region, extent, at, record + 72 + i * 16, value.keyed.slots[i], order ) ) { return false; } + if ( !SquadCookExtent( ctx, region, extent, at, record + 72 + i * 16, value.keyed.slots[i], order ) ) { return false; } } return true; } -// SquadRosterEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// SquadRosterEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool SquadRosterEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const SquadRosterEntry & value, TableByteOrder order ) { SquadRosterEntryCookBody( at, value, order ); int64_t extent_at = 0; - return SquadRosterEntryCookMaps( ctx, region, at + 8, extent_at, at, value, order ); + return SquadRosterEntryCookExtent( ctx, region, at + 8, extent_at, at, value, order ); } -// SquadCookNode: one node — the record, then the extent its maps take (§2.8). +// SquadCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool SquadCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Squad & value, TableByteOrder order ) { if ( !SquadCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return SquadCookMaps( ctx, region, at + 16, extent_at, at, value, order ); + return SquadCookExtent( ctx, region, at + 16, extent_at, at, value, order ); } -// DepthCookNode: one node — the record, then the extent its maps take (§2.8). +// DepthCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool DepthCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Depth & value, TableByteOrder order ) { if ( !DepthCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return DepthCookMaps( ctx, region, at + 136, extent_at, at, value, order ); + return DepthCookExtent( ctx, region, at + 136, extent_at, at, value, order ); } // SquadCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one @@ -6124,15 +6178,15 @@ template inline bool DepthCookNode( const Ctx & ctx, const TableC // eight. The offsets go into the region's table when it has one, and are only // summed when it does not (a measure). A type id the numbering carries that // this root cannot name is the two walks disagreeing, and it is refused. -// A NODE'S SIZE DEPENDS ON ITS VALUE where a map rides in its extent +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent // (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context -// the numbering walked and reads the same maps that walk read. +// the numbering walked and reads the same arrays that walk read. template inline bool SquadCookLayout( const Ctx & ctx, const Squad & root, const TableNumbering & numbering, TableCookRegion & region ) { region.numbering = &numbering; region.count = numbering.count + 1; - const int64_t root_extent = SquadMapExtent( ctx, root ); + const int64_t root_extent = SquadExtent( ctx, root ); if ( root_extent < 0 ) { return false; } int64_t offset = 16 + root_extent; // the root at zero, its extent behind it int64_t align = 8; @@ -6287,15 +6341,15 @@ inline bool SquadCook( const SquadBuilder & builder, void * out, uint64_t capaci // eight. The offsets go into the region's table when it has one, and are only // summed when it does not (a measure). A type id the numbering carries that // this root cannot name is the two walks disagreeing, and it is refused. -// A NODE'S SIZE DEPENDS ON ITS VALUE where a map rides in its extent +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent // (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context -// the numbering walked and reads the same maps that walk read. +// the numbering walked and reads the same arrays that walk read. template inline bool DepthCookLayout( const Ctx & ctx, const Depth & root, const TableNumbering & numbering, TableCookRegion & region ) { region.numbering = &numbering; region.count = numbering.count + 1; - const int64_t root_extent = DepthMapExtent( ctx, root ); + const int64_t root_extent = DepthExtent( ctx, root ); if ( root_extent < 0 ) { return false; } int64_t offset = 136 + root_extent; // the root at zero, its extent behind it int64_t align = 8; @@ -6502,24 +6556,24 @@ extern const TableTypeInfo SquadTableInfo; extern const TableTypeInfo DepthTableInfo; inline const TableFieldInfo SquadRosterEntryTableFields[] = { - { "key", "key", "uint8", 0x3dc94a19365b10ecull, 6, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( SquadRosterEntry, key ), (uint32_t) sizeof( SquadRosterEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( SquadRosterEntry, value ), (uint32_t) sizeof( SquadRosterEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "uint8", 0x3dc94a19365b10ecull, 6, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( SquadRosterEntry, key ), (uint32_t) sizeof( SquadRosterEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( SquadRosterEntry, value ), (uint32_t) sizeof( SquadRosterEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo SquadRosterEntryTableInfo = { "SquadRosterEntry", (uint32_t) sizeof( SquadRosterEntry ), 2, SquadRosterEntryTableFields, +[]( void * p ) { SquadRosterEntryReset( *(SquadRosterEntry *) p ); }, false }; inline const TableTypeInfo * SquadRosterEntryTableType() { return &SquadRosterEntryTableInfo; } inline const TableFieldInfo SquadTableFields[] = { - { "roster", "roster", "map[uint8]Item", 0x1c84390d304f4f42ull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Squad, roster ), (uint32_t) sizeof( Squad::roster ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &SquadRosterEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { SquadRosterEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint8_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint8_t) key_value ); } return (void *) placed; }, "" }, + { "roster", "roster", "map[uint8]Item", 0x1c84390d304f4f42ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Squad, roster ), (uint32_t) sizeof( SquadRosterEntry ), (uint32_t) offsetof( Squad, roster.count ), 0xffffffffu, &SquadRosterEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { SquadRosterEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint8_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint8_t) key_value ); } return (void *) placed; }, "" }, }; inline const TableTypeInfo SquadTableInfo = { "Squad", (uint32_t) sizeof( Squad ), 1, SquadTableFields, +[]( void * p ) { SquadReset( *(Squad *) p ); }, true }; inline const TableTypeInfo * SquadTableType() { return &SquadTableInfo; } inline const TableFieldInfo DepthTableFields[] = { - { "one", "one", "Squad", 0x1a08aa1921ca5cafull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Depth, one ), (uint32_t) sizeof( Depth::one ), 0xffffffffu, 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "many", "many", "Squad", 0x1f6459a2cea1fc02ull, 13, true, false, NULL, NULL, true, false, 3, (uint32_t) offsetof( Depth, many ), (uint32_t) sizeof( Depth::many[0] ), (uint32_t) offsetof( Depth, many_count ), 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "keyed", "keyed", "Squad", 0x70551ff29550f15dull, 13, true, false, NULL, NULL, false, false, (int32_t) Slot::Max, (uint32_t) offsetof( Depth, keyed ), (uint32_t) sizeof( Depth::keyed.slots[0] ), 0xffffffffu, 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, "Slot", +[]( uint64_t v ) { return EnumName( Slot( v ) ); }, +[]( uint64_t v ) -> uint64_t { uint64_t id = 0; TableEnumId( Slot( v ), id ); return id; }, NULL, NULL, NULL, NULL, NULL, "" }, - { "arm", "arm", "Force", 0xe756c0190570ccb5ull, 15, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Depth, arm ), (uint32_t) sizeof( Depth::arm ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, 2, +[]( uint64_t v ) -> const char * { switch ( v ) { case 0: return "None"; case 1: return "squad"; case 2: return "plain"; default: return "???"; } }, +[]( uint64_t v ) -> uint64_t { switch ( v ) { case 0: return 0; case 1: return 0xd5c2bb95d63e6331ull; case 2: return 0xfd4d194e1652b207ull; default: return 0; } }, NULL, NULL, NULL, +[]() -> const TableUnionInfo * { static const TableFieldInfo arm_fields_Force[] = { { "plain", "plain", "int32", 0xfd4d194e1652b207ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Force, plain ), (uint32_t) sizeof( Force::plain ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; static const TableUnionArmInfo arms[] = { { 0, NULL, NULL, 0 }, { (uint32_t) offsetof( Force, squad ), &SquadTableInfo, NULL, 16 }, { (uint32_t) offsetof( Force, plain ), NULL, &arm_fields_Force[0], 4 }, }; static const TableUnionInfo info = { (uint32_t) offsetof( Force, type ), (uint32_t) sizeof( Force::type ), arms }; return &info; }, NULL, NULL, NULL, NULL, "" }, - { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Depth, after ), (uint32_t) sizeof( Depth::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "one", "one", "Squad", 0x1a08aa1921ca5cafull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Depth, one ), (uint32_t) sizeof( Depth::one ), 0xffffffffu, 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "many", "many", "Squad", 0x1f6459a2cea1fc02ull, 13, true, false, NULL, NULL, true, false, 3, (uint32_t) offsetof( Depth, many ), (uint32_t) sizeof( Depth::many[0] ), (uint32_t) offsetof( Depth, many_count ), 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "keyed", "keyed", "Squad", 0x70551ff29550f15dull, 13, true, false, NULL, NULL, false, false, (int32_t) Slot::Max, (uint32_t) offsetof( Depth, keyed ), (uint32_t) sizeof( Depth::keyed.slots[0] ), 0xffffffffu, 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, "Slot", +[]( uint64_t v ) { return EnumName( Slot( v ) ); }, +[]( uint64_t v ) -> uint64_t { uint64_t id = 0; TableEnumId( Slot( v ), id ); return id; }, NULL, NULL, "" }, + { "arm", "arm", "Force", 0xe756c0190570ccb5ull, 15, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Depth, arm ), (uint32_t) sizeof( Depth::arm ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, 2, +[]( uint64_t v ) -> const char * { switch ( v ) { case 0: return "None"; case 1: return "squad"; case 2: return "plain"; default: return "???"; } }, +[]( uint64_t v ) -> uint64_t { switch ( v ) { case 0: return 0; case 1: return 0xd5c2bb95d63e6331ull; case 2: return 0xfd4d194e1652b207ull; default: return 0; } }, NULL, NULL, NULL, +[]() -> const TableUnionInfo * { static const TableFieldInfo arm_fields_Force[] = { { "plain", "plain", "int32", 0xfd4d194e1652b207ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Force, plain ), (uint32_t) sizeof( Force::plain ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; static const TableUnionArmInfo arms[] = { { 0, NULL, NULL, 0 }, { (uint32_t) offsetof( Force, squad ), &SquadTableInfo, NULL, 16 }, { (uint32_t) offsetof( Force, plain ), NULL, &arm_fields_Force[0], 4 }, }; static const TableUnionInfo info = { (uint32_t) offsetof( Force, type ), (uint32_t) sizeof( Force::type ), arms }; return &info; }, NULL, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Depth, after ), (uint32_t) sizeof( Depth::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo DepthTableInfo = { "Depth", (uint32_t) sizeof( Depth ), 5, DepthTableFields, +[]( void * p ) { DepthReset( *(Depth *) p ); }, true }; inline const TableTypeInfo * DepthTableType() { return &DepthTableInfo; } diff --git a/testdata/golden/tables/maps/FleetTable.cpp b/testdata/golden/tables/maps/FleetTable.cpp index 8fea2bfe8..6e774adde 100644 --- a/testdata/golden/tables/maps/FleetTable.cpp +++ b/testdata/golden/tables/maps/FleetTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2731,14 +2748,33 @@ inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * inf // ---- json graph walk: end ---- +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + // ---- json map walk: begin ---- -inline bool TableJsonIsMap( const TableFieldInfo * f ) { return f->entry != NULL; } +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} // the entry's two rows: fields[0] IS the key and fields[1] IS the value, which // is what makes a user's own table of pairs the same bytes (§2.8) -inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->entry->fields[0]; } -inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->entry->fields[1]; } +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } @@ -2798,16 +2834,17 @@ inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const // A region holds them in that order already, so this is the array in place. inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) { - const int32_t count = f->map_count( slot ); + const int32_t count = TableJsonExtentCount( slot ); if ( count == 0 ) { out.raw( "{}", 2 ); return true; } const TableFieldInfo * key = TableJsonMapKeyField( f ); const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); out.put( '{' ); for ( int32_t i = 0; i < count; i++ ) { if ( i > 0 ) { out.put( ',' ); } out.line( depth + 1 ); - const void * entry = f->map_at( slot, i ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); TableJsonWriteMapKey( out, entry, key ); out.raw( ": ", 2 ); if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } @@ -2896,15 +2933,15 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf } if ( !fits ) { in.report->kind_mismatch++; place = false; } } - const int32_t before = f->map_count( (const void *) slot ); - void * entry = place ? f->map_insert( *graph->worker, slot, token, token_length, key_value ) : NULL; + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; if ( place && entry == NULL ) { // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the // wire's rule, because a clamped key is a merged entry (§2.8). in.report->clamped++; } - else if ( entry != NULL && f->map_count( (const void *) slot ) == before ) + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) { in.report->duplicate++; // last-wins, the object rule inside the map } @@ -2955,6 +2992,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf // ---- json map walk: end ---- +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace mapdemo #endif // MAPDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/maps/FleetTable.h b/testdata/golden/tables/maps/FleetTable.h index 3ef7f2542..02df5cfd9 100644 --- a/testdata/golden/tables/maps/FleetTable.h +++ b/testdata/golden/tables/maps/FleetTable.h @@ -109,6 +109,18 @@ struct TableReport TableMessageReason reason = newer_form; }; + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; // ---- reflection (tables only, docs/SPEC-TABLES.md) ---- // // Static field descriptors for every type in the table closure: name, wire @@ -224,18 +236,13 @@ struct TableFieldInfo // a function pointer at compile time; the arms themselves are a static // inside it). NULL for every other kind. const TableUnionInfo * (*arms)(); - // a MAP (docs/SPEC-TABLES.md §2.8): the generated ENTRY's descriptor — - // fields[0] is the key and fields[1] the value — and the three the ONE - // text walk cannot spell for itself, because TableMap is a type - // it has no name for. NULL on every field that is not a map. - const TableTypeInfo * entry; - int32_t ( * map_count )( const void * slot ); - const void * ( * map_at )( const void * slot, int32_t index ); - // place one entry BY KEY and hand back the entry, at its defaults: a - // string key comes in as the bytes and the length, an integer key as - // the value, and NULL is NOT INSERTED — a key past the bound, or an - // arena that could not carve another segment. - void * ( * map_insert )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded }; @@ -1291,8 +1298,8 @@ struct TableWorker return blob; } - // RAW, ZEROED storage of the bytes asked for, at the alignment asked for — a MAP's builder head and its - // entry segments (docs/SPEC-TABLES.md §2.8). It is not a node: it carries + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries // no type id, takes no index and has no Reset, so it goes through the same // slab and span the blob path uses rather than through Alloc. uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) @@ -1576,12 +1583,6 @@ static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts th // resolving through it yields NULL and can never fabricate the root. static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; -// What a node's storage answers when the FRAMING ITSELF is refused rather than -// merely unnameable: a map whose N cannot fit in its L (docs/SPEC-TABLES.md -// §2.8). An unnameable type id commands no storage and keeps its index; this -// one makes the whole measure answer -1 (§7.6). -static const int64_t kTableNodeRefused = -2; - // ---- the numbering, on the SAVE side ---- // // One entry per reachable node in FIRST-VISIT order, so entry k is node index @@ -1779,9 +1780,9 @@ struct TableNodeDirEntry uint64_t type_id; }; -// a map's extent cursor, defined with the map runtime (docs/SPEC-TABLES.md -// §2.8); the node map names it only through a pointer. -struct TableMapCarve; +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; // TableNodeMap is what a pointer slot resolves through while a body decodes. struct TableNodeMap @@ -1794,17 +1795,21 @@ struct TableNodeMap // takes the SELF-RELATIVE delta so a deref is one add, and the tool's // builder path takes the node's ARENA OFFSET (§6.3). bool arena = false; - // WHERE A MAP'S ENTRIES LAND while this node's body decodes - // (docs/SPEC-TABLES.md §2.8): the node's own extent on the region path - // and the builder's arena on the tool's. It is MUTABLE because the - // cursor belongs to ONE node's decode and the dispatch that owns that - // node holds the map by const reference, exactly as it did before maps - // existed — the decoder's signature does not move for a construct it - // may not carry. - mutable TableMapCarve * carve = NULL; - // and the TOOL's path's allocation front, set once: there a map's - // entries are the builder's arena's rather than a node's extent. + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; }; // TableNodeResolve places one node index in a pointer slot, and every failure @@ -1948,6 +1953,90 @@ inline bool TableNodeScanWhole( TableNodeScan & s ) #endif // MAPDEMO_SCHEMA_TABLE_ARENA +#ifndef MAPDEMO_SCHEMA_TABLE_EXTENT +#define MAPDEMO_SCHEMA_TABLE_EXTENT + +namespace mapdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace mapdemo + +#endif // MAPDEMO_SCHEMA_TABLE_EXTENT + #ifndef MAPDEMO_SCHEMA_TABLE_MAP #define MAPDEMO_SCHEMA_TABLE_MAP @@ -2378,12 +2467,6 @@ inline TableMapEach TableMapEachOf( const TableArena & arena, const Table return each; } -// AN UNREACHED SLOT MUST HOLD NO MAP WITH ENTRIES IN IT (§2.8, §7.6). An empty -// map takes no bytes, so a record whose extent measures ZERO is a record whose -// every by-value map is empty; a measure that REFUSED answers non-zero here -// too, and refusing on it is the same answer one level up. -inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } - // ---- the LOAD side: where a decoded entry lands (§2.8) ---- // // THE READER TRUSTS NOTHING and spends one compare per entry. Every load path @@ -2393,16 +2476,9 @@ inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } // out of the holder node's own extent, and the TOOL's path appends into the // builder's arena, and the decoder above them cannot tell which it has. -// TableMapCarve is a node's extent cursor, PRE-ORDER: a map's whole entry -// array first, then, entry by entry in key order, the arrays of any map an -// entry's value holds by value. The cursor is the node map's, because the -// generated decoder is threaded with that and not with a region. -struct TableMapCarve -{ - uint8_t * at = NULL; // the region path: the node's extent, unspent - int64_t left = 0; - TableWorker * worker = NULL; // the TOOL's path: entries come from the arena -}; +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. // TableMapFill is one map field being decoded: where the next entry lands, and // the entry that last LANDED, which is what the ascending check compares @@ -2527,10 +2603,9 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // LoadMeasure's term for a map is N x sizeof( Entry ) rounded to // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value -// holds a map of its own, the entries' headers under it. The caller owns the -// allocation precisely so it can refuse a number it did not expect. -typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ); - +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect, +// and a refusal carries its reason (§6.5). // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2538,8 +2613,8 @@ typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, i static const int64_t kTableMapEntryFloor = 2; inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, - int64_t entry_size, int64_t entry_align, TableMapWireExtentFn inner, - const TableIdTable * ids ) + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; TableReader r( body, length, &scratch, ids ); @@ -2548,7 +2623,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } const int64_t rest = length - r.offset; - if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { return false; } // an N the map's L cannot carry + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); at += (int64_t) n * entry_size; if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term @@ -2556,49 +2631,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & { uint64_t elem = 0; if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// the same framing walk over an ARRAY OF TABLES that is not a map: its -// elements' own maps are part of this node's extent too -inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each -// length-prefixed element (docs/SPEC-TABLES.md §3.2) -inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t key = 0; - if ( !r.getleb( key ) ) { return true; } - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } r.offset += (int64_t) elem; } return true; @@ -3057,7 +3090,7 @@ inline void FleetLoadoutsEntryReset( FleetLoadoutsEntry & value ) { memset( value.key, 0, sizeof( value.key ) ); value.key_length = 0; - value.value.entries.value = 0; // map[uint8]Item — empty + value.value.entries.value = 0; // map[uint8]Item: empty value.value.count = 0; value.value.padding = 0; } @@ -3070,17 +3103,17 @@ inline void FleetTiersEntryReset( FleetTiersEntry & value ) inline void FleetReset( Fleet & value ) { - value.ships.entries.value = 0; // map[string(32)]ShipConfig — empty + value.ships.entries.value = 0; // map[string(32)]ShipConfig: empty value.ships.count = 0; value.ships.padding = 0; - value.by_id.entries.value = 0; // map[uint32]*ShipConfig — empty + value.by_id.entries.value = 0; // map[uint32]*ShipConfig: empty value.by_id.count = 0; value.by_id.padding = 0; value.flagship.value = 0; // *ShipConfig — null - value.loadouts.entries.value = 0; // map[string(16)]map[uint8]Item — empty + value.loadouts.entries.value = 0; // map[string(16)]map[uint8]Item: empty value.loadouts.count = 0; value.loadouts.padding = 0; - value.tiers.entries.value = 0; // map[int16]Item — empty + value.tiers.entries.value = 0; // map[int16]Item: empty value.tiers.count = 0; value.tiers.padding = 0; } @@ -3436,7 +3469,7 @@ struct FleetLoadoutsEntryEach { const char * key; decltype( TableEntryValue( (Fl inline FleetLoadoutsEntryEach TableEntryEach( FleetLoadoutsEntry * entry ) { return FleetLoadoutsEntryEach{ TableEntryKey( *entry ), TableEntryValue( entry ) }; } inline void TableResetMapValue( FleetLoadoutsEntry & value ) { - value.value.entries.value = 0; // map[uint8]Item — empty + value.value.entries.value = 0; // map[uint8]Item: empty value.value.count = 0; value.value.padding = 0; } @@ -5249,66 +5282,66 @@ inline bool FleetLoadBody( TableReader & r, const TableNodeMap & nodes, Fleet & } } -// ShipConfigWireExtent: the extent ShipConfig's maps command, from the FRAMING alone. +// ShipConfigWireExtent: the extent ShipConfig's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool ShipConfigWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool ShipConfigWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { - (void) body; (void) length; (void) at; (void) ids; // no map below this record + (void) body; (void) length; (void) at; (void) ids; (void) reason; // no list or map below this record return true; } -// ShipConfigMapExtentAt: the node extent ShipConfig's maps take, PRE-ORDER, advancing the -// running offset exactly as ShipConfigMapPack advances it (docs/SPEC-TABLES.md §2.8). +// ShipConfigExtentAt: the node extent ShipConfig's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as ShipConfigExtentPack advances it (§2.8, §2.9). template -inline bool ShipConfigMapExtentAt( const Ctx & ctx, const ShipConfig & value, int64_t & at ) +inline bool ShipConfigExtentAt( const Ctx & ctx, const ShipConfig & value, int64_t & at ) { - (void) ctx; (void) value; (void) at; // no map below this record + (void) ctx; (void) value; (void) at; // no list or map below this record return true; } -// ShipConfigMapPack: carve ShipConfig's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset ShipConfigMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// ShipConfigExtentPack: carve ShipConfig's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset ShipConfigExtentAt advances (§2.8, §2.9). template -inline bool ShipConfigMapPack( const Ctx & ctx, const ShipConfig & src, ShipConfig & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool ShipConfigExtentPack( const Ctx & ctx, const ShipConfig & src, ShipConfig & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { - (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no map below this record + (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no list or map below this record return true; } -// FleetByIdEntryWireExtent: the extent FleetByIdEntry's maps command, from the FRAMING alone. +// FleetByIdEntryWireExtent: the extent FleetByIdEntry's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool FleetByIdEntryWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool FleetByIdEntryWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { - (void) body; (void) length; (void) at; (void) ids; // no map below this record + (void) body; (void) length; (void) at; (void) ids; (void) reason; // no list or map below this record return true; } -// FleetByIdEntryMapExtentAt: the node extent FleetByIdEntry's maps take, PRE-ORDER, advancing the -// running offset exactly as FleetByIdEntryMapPack advances it (docs/SPEC-TABLES.md §2.8). +// FleetByIdEntryExtentAt: the node extent FleetByIdEntry's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as FleetByIdEntryExtentPack advances it (§2.8, §2.9). template -inline bool FleetByIdEntryMapExtentAt( const Ctx & ctx, const FleetByIdEntry & value, int64_t & at ) +inline bool FleetByIdEntryExtentAt( const Ctx & ctx, const FleetByIdEntry & value, int64_t & at ) { - (void) ctx; (void) value; (void) at; // no map below this record + (void) ctx; (void) value; (void) at; // no list or map below this record return true; } -// FleetByIdEntryMapPack: carve FleetByIdEntry's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset FleetByIdEntryMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// FleetByIdEntryExtentPack: carve FleetByIdEntry's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset FleetByIdEntryExtentAt advances (§2.8, §2.9). template -inline bool FleetByIdEntryMapPack( const Ctx & ctx, const FleetByIdEntry & src, FleetByIdEntry & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool FleetByIdEntryExtentPack( const Ctx & ctx, const FleetByIdEntry & src, FleetByIdEntry & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { - (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no map below this record + (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no list or map below this record return true; } -// FleetLoadoutsEntryWireExtent: the extent FleetLoadoutsEntry's maps command, from the FRAMING alone. +// FleetLoadoutsEntryWireExtent: the extent FleetLoadoutsEntry's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool FleetLoadoutsEntryWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool FleetLoadoutsEntryWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; // the scan's framing damage is the LOAD's to report TableReader r( body, length, &scratch, ids ); @@ -5327,17 +5360,17 @@ inline bool FleetLoadoutsEntryWireExtent( const uint8_t * body, int64_t length, if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetLoadoutsEntryValueEntry ), (int64_t) alignof( FleetLoadoutsEntryValueEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetLoadoutsEntryValueEntry ), (int64_t) alignof( FleetLoadoutsEntryValueEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( !r.skip( field_kind ) ) { return true; } } } -// FleetLoadoutsEntryMapExtentAt: the node extent FleetLoadoutsEntry's maps take, PRE-ORDER, advancing the -// running offset exactly as FleetLoadoutsEntryMapPack advances it (docs/SPEC-TABLES.md §2.8). +// FleetLoadoutsEntryExtentAt: the node extent FleetLoadoutsEntry's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as FleetLoadoutsEntryExtentPack advances it (§2.8, §2.9). template -inline bool FleetLoadoutsEntryMapExtentAt( const Ctx & ctx, const FleetLoadoutsEntry & value, int64_t & at ) +inline bool FleetLoadoutsEntryExtentAt( const Ctx & ctx, const FleetLoadoutsEntry & value, int64_t & at ) { { TableMapCursor cursor = TableMapOrder( ctx, value.value ); @@ -5352,18 +5385,18 @@ inline bool FleetLoadoutsEntryMapExtentAt( const Ctx & ctx, const FleetLoadoutsE // the whole extent of one node, from a fresh offset: what a pack reserves // for it beside the record's own storage. template -inline int64_t FleetLoadoutsEntryMapExtent( const Ctx & ctx, const FleetLoadoutsEntry & value ) +inline int64_t FleetLoadoutsEntryExtent( const Ctx & ctx, const FleetLoadoutsEntry & value ) { int64_t at = 0; - if ( !FleetLoadoutsEntryMapExtentAt( ctx, value, at ) ) { return -1; } + if ( !FleetLoadoutsEntryExtentAt( ctx, value, at ) ) { return -1; } return at; } -// FleetLoadoutsEntryMapPack: carve FleetLoadoutsEntry's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset FleetLoadoutsEntryMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// FleetLoadoutsEntryExtentPack: carve FleetLoadoutsEntry's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset FleetLoadoutsEntryExtentAt advances (§2.8, §2.9). template -inline bool FleetLoadoutsEntryMapPack( const Ctx & ctx, const FleetLoadoutsEntry & src, FleetLoadoutsEntry & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool FleetLoadoutsEntryExtentPack( const Ctx & ctx, const FleetLoadoutsEntry & src, FleetLoadoutsEntry & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { { TableMapCursor cursor = TableMapOrder( ctx, src.value ); @@ -5385,10 +5418,10 @@ inline bool FleetLoadoutsEntryMapPack( const Ctx & ctx, const FleetLoadoutsEntry return true; } -// FleetWireExtent: the extent Fleet's maps command, from the FRAMING alone. +// FleetWireExtent: the extent Fleet's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool FleetWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool FleetWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; // the scan's framing damage is the LOAD's to report TableReader r( body, length, &scratch, ids ); @@ -5407,7 +5440,7 @@ inline bool FleetWireExtent( const uint8_t * body, int64_t length, int64_t & at, if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetShipsEntry ), (int64_t) alignof( FleetShipsEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetShipsEntry ), (int64_t) alignof( FleetShipsEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( field_id == 0x7b024c46e98d3404ull && field_kind == 14 ) // by_id @@ -5416,7 +5449,7 @@ inline bool FleetWireExtent( const uint8_t * body, int64_t length, int64_t & at, if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetByIdEntry ), (int64_t) alignof( FleetByIdEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetByIdEntry ), (int64_t) alignof( FleetByIdEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( field_id == 0x294fa1b3f0f5f070ull && field_kind == 14 ) // loadouts @@ -5425,7 +5458,7 @@ inline bool FleetWireExtent( const uint8_t * body, int64_t length, int64_t & at, if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetLoadoutsEntry ), (int64_t) alignof( FleetLoadoutsEntry ), &FleetLoadoutsEntryWireExtent, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetLoadoutsEntry ), (int64_t) alignof( FleetLoadoutsEntry ), &FleetLoadoutsEntryWireExtent, ids, reason ) ) { return false; } continue; } if ( field_id == 0x6dd8dc6c5fdae3ceull && field_kind == 14 ) // tiers @@ -5434,17 +5467,17 @@ inline bool FleetWireExtent( const uint8_t * body, int64_t length, int64_t & at, if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetTiersEntry ), (int64_t) alignof( FleetTiersEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetTiersEntry ), (int64_t) alignof( FleetTiersEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( !r.skip( field_kind ) ) { return true; } } } -// FleetMapExtentAt: the node extent Fleet's maps take, PRE-ORDER, advancing the -// running offset exactly as FleetMapPack advances it (docs/SPEC-TABLES.md §2.8). +// FleetExtentAt: the node extent Fleet's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as FleetExtentPack advances it (§2.8, §2.9). template -inline bool FleetMapExtentAt( const Ctx & ctx, const Fleet & value, int64_t & at ) +inline bool FleetExtentAt( const Ctx & ctx, const Fleet & value, int64_t & at ) { { TableMapCursor cursor = TableMapOrder( ctx, value.ships ); @@ -5460,7 +5493,7 @@ inline bool FleetMapExtentAt( const Ctx & ctx, const Fleet & value, int64_t & at at += (int64_t) cursor.count * (int64_t) sizeof( FleetByIdEntry ); // the whole array FIRST for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order { - if ( !FleetByIdEntryMapExtentAt( ctx, *cursor[i], at ) ) { TableMapRelease( cursor ); return false; } + if ( !FleetByIdEntryExtentAt( ctx, *cursor[i], at ) ) { TableMapRelease( cursor ); return false; } } TableMapRelease( cursor ); } @@ -5471,7 +5504,7 @@ inline bool FleetMapExtentAt( const Ctx & ctx, const Fleet & value, int64_t & at at += (int64_t) cursor.count * (int64_t) sizeof( FleetLoadoutsEntry ); // the whole array FIRST for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order { - if ( !FleetLoadoutsEntryMapExtentAt( ctx, *cursor[i], at ) ) { TableMapRelease( cursor ); return false; } + if ( !FleetLoadoutsEntryExtentAt( ctx, *cursor[i], at ) ) { TableMapRelease( cursor ); return false; } } TableMapRelease( cursor ); } @@ -5488,18 +5521,18 @@ inline bool FleetMapExtentAt( const Ctx & ctx, const Fleet & value, int64_t & at // the whole extent of one node, from a fresh offset: what a pack reserves // for it beside the record's own storage. template -inline int64_t FleetMapExtent( const Ctx & ctx, const Fleet & value ) +inline int64_t FleetExtent( const Ctx & ctx, const Fleet & value ) { int64_t at = 0; - if ( !FleetMapExtentAt( ctx, value, at ) ) { return -1; } + if ( !FleetExtentAt( ctx, value, at ) ) { return -1; } return at; } -// FleetMapPack: carve Fleet's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset FleetMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// FleetExtentPack: carve Fleet's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset FleetExtentAt advances (§2.8, §2.9). template -inline bool FleetMapPack( const Ctx & ctx, const Fleet & src, Fleet & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool FleetExtentPack( const Ctx & ctx, const Fleet & src, Fleet & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { { TableMapCursor cursor = TableMapOrder( ctx, src.ships ); @@ -5535,7 +5568,7 @@ inline bool FleetMapPack( const Ctx & ctx, const Fleet & src, Fleet & dst, uint8 } for ( int32_t i = 0; i < cursor.count; i++ ) { - if ( !FleetByIdEntryMapPack( ctx, *cursor[i], placed[i], extent, at, capacity ) ) { TableMapRelease( cursor ); return false; } + if ( !FleetByIdEntryExtentPack( ctx, *cursor[i], placed[i], extent, at, capacity ) ) { TableMapRelease( cursor ); return false; } } TableMapRelease( cursor ); } @@ -5556,7 +5589,7 @@ inline bool FleetMapPack( const Ctx & ctx, const Fleet & src, Fleet & dst, uint8 } for ( int32_t i = 0; i < cursor.count; i++ ) { - if ( !FleetLoadoutsEntryMapPack( ctx, *cursor[i], placed[i], extent, at, capacity ) ) { TableMapRelease( cursor ); return false; } + if ( !FleetLoadoutsEntryExtentPack( ctx, *cursor[i], placed[i], extent, at, capacity ) ) { TableMapRelease( cursor ); return false; } } TableMapRelease( cursor ); } @@ -6121,7 +6154,7 @@ inline bool ShipConfigPack( const Ctx & ctx, TablePackMap & seen, const ShipConf int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( ShipConfig ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !ShipConfigMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !ShipConfigExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return ShipConfigPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -6220,7 +6253,7 @@ inline bool FleetByIdEntryPack( const Ctx & ctx, TablePackMap & seen, const Flee int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( FleetByIdEntry ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !FleetByIdEntryMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !FleetByIdEntryExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return FleetByIdEntryPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -6298,7 +6331,7 @@ inline bool FleetLoadoutsEntryPack( const Ctx & ctx, TablePackMap & seen, const int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( FleetLoadoutsEntry ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !FleetLoadoutsEntryMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !FleetLoadoutsEntryExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return FleetLoadoutsEntryPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -6417,7 +6450,7 @@ inline bool FleetPack( const Ctx & ctx, TablePackMap & seen, const Fleet & src, int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Fleet ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !FleetMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !FleetExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return FleetPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -6544,7 +6577,7 @@ inline bool FleetBuilder::Lock() below = FleetPackMeasure( ctx, seen, root ); } if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it - int64_t root_extent = FleetMapExtent( ctx, root ); + int64_t root_extent = FleetExtent( ctx, root ); if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Fleet ) ) + root_extent ) + below; // the AUTHORING path may allocate (§6.5), and it does so through the @@ -6644,10 +6677,11 @@ inline uint32_t FleetNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t // already owns. inline void FleetNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) { - // the node's own EXTENT, where its maps' entry arrays are carved from, - // PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8). The tool's - // path carries a worker instead: there the entries are the arena's. - TableMapCarve carve; + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; carve.worker = nodes.worker; if ( carve.worker == NULL ) { @@ -6797,7 +6831,7 @@ inline int64_t FleetSaveMessage( const FleetBuilder & builder, uint8_t * buffer, // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t FleetLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t FleetLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; TableIdTable ids_table; @@ -6810,8 +6844,9 @@ inline int64_t FleetLoadMeasure( const uint8_t * wire_file, int64_t wire_file_by const uint8_t * const wire = wire_file + 1; const int64_t wire_bytes = body_bytes; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Fleet ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -6821,7 +6856,7 @@ inline int64_t FleetLoadMeasure( const uint8_t * wire_file, int64_t wire_file_by { records++; int64_t storage = FleetNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -6865,8 +6900,9 @@ inline const Fleet * FleetLoad( uint8_t * region, int64_t region_bytes, const ui int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Fleet ) ) + root_extent ); int64_t records = 0; { @@ -6949,7 +6985,7 @@ inline const Fleet * FleetLoad( uint8_t * region, int64_t region_bytes, const ui // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Fleet ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -6965,7 +7001,7 @@ inline const Fleet * FleetLoad( uint8_t * region, int64_t region_bytes, const ui // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t FleetLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t FleetLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } @@ -6973,8 +7009,9 @@ inline int64_t FleetLoadMeasure( const TableVocabulary & vocabulary, const uint8 const uint8_t * const wire = message + 1; const int64_t wire_bytes = message_bytes - 1; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Fleet ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -6984,7 +7021,7 @@ inline int64_t FleetLoadMeasure( const TableVocabulary & vocabulary, const uint8 { records++; int64_t storage = FleetNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -7019,8 +7056,9 @@ inline const Fleet * FleetLoadMessage( uint8_t * region, int64_t region_bytes, c int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Fleet ) ) + root_extent ); int64_t records = 0; { @@ -7103,7 +7141,7 @@ inline const Fleet * FleetLoadMessage( uint8_t * region, int64_t region_bytes, c // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Fleet ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -7158,7 +7196,7 @@ inline bool FleetLoadBuilder( FleetBuilder & builder, const uint8_t * wire_file, nodes.entries = directory; nodes.count = records + 1; nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here - nodes.worker = &builder.main; // and a map's entries are the arena's, not a node extent's (§2.8) + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) { TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); int64_t k = 0; @@ -7197,10 +7235,14 @@ inline bool FleetLoadBuilder( FleetBuilder & builder, const uint8_t * wire_file, } TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.worker = &builder.main; nodes.carve = &root_carve; bool ok = FleetLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; allocator.free( allocator.context, directory ); return ok; } @@ -7333,10 +7375,10 @@ inline void FleetLoadoutsEntryValueEntryCookBody( uint8_t * at, const FleetLoado template inline bool FleetLoadoutsEntryCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const FleetLoadoutsEntry & value, TableByteOrder order ) { - (void) ctx; (void) region; // no reference below this node: the class was decided by a pointer elsewhere in its closure + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure table_cook_bytes( at + 0, value.key, value.key_length, 17 ); table_cook_put( at + 20, (uint64_t) (uint32_t) value.key_length, 4, order ); - table_cook_put( at + 24, 0, 8, order ); // value: the entry array's delta, filled by the extent writer + table_cook_put( at + 24, 0, 8, order ); // value: the array's delta, filled by the extent writer table_cook_put( at + 32, 0, 4, order ); // and its count return true; } @@ -7349,72 +7391,78 @@ inline void FleetTiersEntryCookBody( uint8_t * at, const FleetTiersEntry & value template inline bool FleetCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Fleet & value, TableByteOrder order ) { - table_cook_put( at + 0, 0, 8, order ); // ships: the entry array's delta, filled by the extent writer + table_cook_put( at + 0, 0, 8, order ); // ships: the array's delta, filled by the extent writer table_cook_put( at + 8, 0, 4, order ); // and its count - table_cook_put( at + 16, 0, 8, order ); // by_id: the entry array's delta, filled by the extent writer + table_cook_put( at + 16, 0, 8, order ); // by_id: the array's delta, filled by the extent writer table_cook_put( at + 24, 0, 4, order ); // and its count if ( !table_cook_ref( region, at + 32, (const void *) ShipConfigAt( ctx, value.flagship ), order ) ) { return false; } // flagship - table_cook_put( at + 40, 0, 8, order ); // loadouts: the entry array's delta, filled by the extent writer + table_cook_put( at + 40, 0, 8, order ); // loadouts: the array's delta, filled by the extent writer table_cook_put( at + 48, 0, 4, order ); // and its count - table_cook_put( at + 56, 0, 8, order ); // tiers: the entry array's delta, filled by the extent writer + table_cook_put( at + 56, 0, 8, order ); // tiers: the array's delta, filled by the extent writer table_cook_put( at + 64, 0, 4, order ); // and its count return true; } -template inline bool ShipConfigCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const ShipConfig & value, TableByteOrder order ); -template inline bool ItemCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Item & value, TableByteOrder order ); -template inline bool FleetShipsEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetShipsEntry & value, TableByteOrder order ); -template inline bool FleetByIdEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetByIdEntry & value, TableByteOrder order ); -template inline bool FleetLoadoutsEntryValueEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntryValueEntry & value, TableByteOrder order ); -template inline bool FleetLoadoutsEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntry & value, TableByteOrder order ); -template inline bool FleetTiersEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetTiersEntry & value, TableByteOrder order ); -template inline bool FleetCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Fleet & value, TableByteOrder order ); +template inline bool ShipConfigCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const ShipConfig & value, TableByteOrder order ); +template inline bool ItemCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Item & value, TableByteOrder order ); +template inline bool FleetShipsEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetShipsEntry & value, TableByteOrder order ); +template inline bool FleetByIdEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetByIdEntry & value, TableByteOrder order ); +template inline bool FleetLoadoutsEntryValueEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntryValueEntry & value, TableByteOrder order ); +template inline bool FleetLoadoutsEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntry & value, TableByteOrder order ); +template inline bool FleetTiersEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetTiersEntry & value, TableByteOrder order ); +template inline bool FleetCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Fleet & value, TableByteOrder order ); -// ShipConfigCookMaps: ShipConfig's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool ShipConfigCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const ShipConfig & value, TableByteOrder order ) +// ShipConfigCookExtent: ShipConfig's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool ShipConfigCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const ShipConfig & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// ItemCookMaps: Item's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool ItemCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Item & value, TableByteOrder order ) +// ItemCookExtent: Item's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool ItemCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Item & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// FleetShipsEntryCookMaps: FleetShipsEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool FleetShipsEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetShipsEntry & value, TableByteOrder order ) +// FleetShipsEntryCookExtent: FleetShipsEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FleetShipsEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetShipsEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// FleetByIdEntryCookMaps: FleetByIdEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool FleetByIdEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetByIdEntry & value, TableByteOrder order ) +// FleetByIdEntryCookExtent: FleetByIdEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FleetByIdEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetByIdEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// FleetLoadoutsEntryValueEntryCookMaps: FleetLoadoutsEntryValueEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool FleetLoadoutsEntryValueEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntryValueEntry & value, TableByteOrder order ) +// FleetLoadoutsEntryValueEntryCookExtent: FleetLoadoutsEntryValueEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FleetLoadoutsEntryValueEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntryValueEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// FleetLoadoutsEntryCookMaps: FleetLoadoutsEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool FleetLoadoutsEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntry & value, TableByteOrder order ) +// FleetLoadoutsEntryCookExtent: FleetLoadoutsEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FleetLoadoutsEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntry & value, TableByteOrder order ) { - (void) region; // a map's entries carry their own references through their own bodies + (void) region; // a table element's and an entry's references resolve through their own bodies { // value TableMapCursor cursor = TableMapOrder( ctx, value.value ); if ( !cursor.ok ) { return false; } @@ -7433,19 +7481,21 @@ template inline bool FleetLoadoutsEntryCookMaps( const Ctx & ctx, return true; } -// FleetTiersEntryCookMaps: FleetTiersEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool FleetTiersEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetTiersEntry & value, TableByteOrder order ) +// FleetTiersEntryCookExtent: FleetTiersEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FleetTiersEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetTiersEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// FleetCookMaps: Fleet's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool FleetCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Fleet & value, TableByteOrder order ) +// FleetCookExtent: Fleet's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FleetCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Fleet & value, TableByteOrder order ) { - (void) region; // a map's entries carry their own references through their own bodies + (void) region; // a table element's and an entry's references resolve through their own bodies { // ships TableMapCursor cursor = TableMapOrder( ctx, value.ships ); if ( !cursor.ok ) { return false; } @@ -7491,7 +7541,7 @@ template inline bool FleetCookMaps( const Ctx & ctx, const TableC } for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order { - if ( !FleetLoadoutsEntryCookMaps( ctx, region, extent, at, array + i * 40, *cursor[i], order ) ) { TableMapRelease( cursor ); return false; } + if ( !FleetLoadoutsEntryCookExtent( ctx, region, extent, at, array + i * 40, *cursor[i], order ) ) { TableMapRelease( cursor ); return false; } } TableMapRelease( cursor ); } @@ -7513,68 +7563,68 @@ template inline bool FleetCookMaps( const Ctx & ctx, const TableC return true; } -// ShipConfigCookNode: one node — the record, then the extent its maps take (§2.8). +// ShipConfigCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool ShipConfigCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const ShipConfig & value, TableByteOrder order ) { ShipConfigCookBody( at, value, order ); int64_t extent_at = 0; - return ShipConfigCookMaps( ctx, region, at + 80, extent_at, at, value, order ); + return ShipConfigCookExtent( ctx, region, at + 80, extent_at, at, value, order ); } -// ItemCookNode: one node — the record, then the extent its maps take (§2.8). +// ItemCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool ItemCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Item & value, TableByteOrder order ) { ItemCookBody( at, value, order ); int64_t extent_at = 0; - return ItemCookMaps( ctx, region, at + 8, extent_at, at, value, order ); + return ItemCookExtent( ctx, region, at + 8, extent_at, at, value, order ); } -// FleetShipsEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// FleetShipsEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool FleetShipsEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const FleetShipsEntry & value, TableByteOrder order ) { FleetShipsEntryCookBody( at, value, order ); int64_t extent_at = 0; - return FleetShipsEntryCookMaps( ctx, region, at + 120, extent_at, at, value, order ); + return FleetShipsEntryCookExtent( ctx, region, at + 120, extent_at, at, value, order ); } -// FleetByIdEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// FleetByIdEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool FleetByIdEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const FleetByIdEntry & value, TableByteOrder order ) { if ( !FleetByIdEntryCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return FleetByIdEntryCookMaps( ctx, region, at + 16, extent_at, at, value, order ); + return FleetByIdEntryCookExtent( ctx, region, at + 16, extent_at, at, value, order ); } -// FleetLoadoutsEntryValueEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// FleetLoadoutsEntryValueEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool FleetLoadoutsEntryValueEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const FleetLoadoutsEntryValueEntry & value, TableByteOrder order ) { FleetLoadoutsEntryValueEntryCookBody( at, value, order ); int64_t extent_at = 0; - return FleetLoadoutsEntryValueEntryCookMaps( ctx, region, at + 8, extent_at, at, value, order ); + return FleetLoadoutsEntryValueEntryCookExtent( ctx, region, at + 8, extent_at, at, value, order ); } -// FleetLoadoutsEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// FleetLoadoutsEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool FleetLoadoutsEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const FleetLoadoutsEntry & value, TableByteOrder order ) { if ( !FleetLoadoutsEntryCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return FleetLoadoutsEntryCookMaps( ctx, region, at + 40, extent_at, at, value, order ); + return FleetLoadoutsEntryCookExtent( ctx, region, at + 40, extent_at, at, value, order ); } -// FleetTiersEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// FleetTiersEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool FleetTiersEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const FleetTiersEntry & value, TableByteOrder order ) { FleetTiersEntryCookBody( at, value, order ); int64_t extent_at = 0; - return FleetTiersEntryCookMaps( ctx, region, at + 8, extent_at, at, value, order ); + return FleetTiersEntryCookExtent( ctx, region, at + 8, extent_at, at, value, order ); } -// FleetCookNode: one node — the record, then the extent its maps take (§2.8). +// FleetCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool FleetCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Fleet & value, TableByteOrder order ) { if ( !FleetCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return FleetCookMaps( ctx, region, at + 72, extent_at, at, value, order ); + return FleetCookExtent( ctx, region, at + 72, extent_at, at, value, order ); } // ShipConfigCookMeasure: the whole cooked file's bytes — the header, the data part @@ -7688,15 +7738,15 @@ inline bool ItemCook( const Item & value, void * out, uint64_t capacity, TableBy // eight. The offsets go into the region's table when it has one, and are only // summed when it does not (a measure). A type id the numbering carries that // this root cannot name is the two walks disagreeing, and it is refused. -// A NODE'S SIZE DEPENDS ON ITS VALUE where a map rides in its extent +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent // (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context -// the numbering walked and reads the same maps that walk read. +// the numbering walked and reads the same arrays that walk read. template inline bool FleetCookLayout( const Ctx & ctx, const Fleet & root, const TableNumbering & numbering, TableCookRegion & region ) { region.numbering = &numbering; region.count = numbering.count + 1; - const int64_t root_extent = FleetMapExtent( ctx, root ); + const int64_t root_extent = FleetExtent( ctx, root ); if ( root_extent < 0 ) { return false; } int64_t offset = 72 + root_extent; // the root at zero, its extent behind it int64_t align = 8; @@ -7954,59 +8004,59 @@ extern const TableTypeInfo FleetTiersEntryTableInfo; extern const TableTypeInfo FleetTableInfo; inline const TableFieldInfo ShipConfigTableFields[] = { - { "name", "name", "string", 0xc4bcadba8e631b86ull, 12, false, false, NULL, NULL, true, false, 64, (uint32_t) offsetof( ShipConfig, name ), (uint32_t) sizeof( ShipConfig::name ), (uint32_t) offsetof( ShipConfig, name_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "health", "health", "int32", 0x7f69d4b5288ba9cfull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( ShipConfig, health ), (uint32_t) sizeof( ShipConfig::health ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "name", "name", "string", 0xc4bcadba8e631b86ull, 12, false, false, NULL, NULL, true, false, 64, (uint32_t) offsetof( ShipConfig, name ), (uint32_t) sizeof( ShipConfig::name ), (uint32_t) offsetof( ShipConfig, name_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "health", "health", "int32", 0x7f69d4b5288ba9cfull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( ShipConfig, health ), (uint32_t) sizeof( ShipConfig::health ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo ShipConfigTableInfo = { "ShipConfig", (uint32_t) sizeof( ShipConfig ), 2, ShipConfigTableFields, +[]( void * p ) { ShipConfigReset( *(ShipConfig *) p ); }, false }; inline const TableTypeInfo * ShipConfigTableType() { return &ShipConfigTableInfo; } inline const TableFieldInfo ItemTableFields[] = { - { "count", "count", "int32", 0xb1e5e28e4479a274ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Item, count ), (uint32_t) sizeof( Item::count ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "count", "count", "int32", 0xb1e5e28e4479a274ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Item, count ), (uint32_t) sizeof( Item::count ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo ItemTableInfo = { "Item", (uint32_t) sizeof( Item ), 1, ItemTableFields, +[]( void * p ) { ItemReset( *(Item *) p ); }, false }; inline const TableTypeInfo * ItemTableType() { return &ItemTableInfo; } inline const TableFieldInfo FleetShipsEntryTableFields[] = { - { "key", "key", "string", 0x3dc94a19365b10ecull, 12, false, false, NULL, NULL, true, false, 32, (uint32_t) offsetof( FleetShipsEntry, key ), (uint32_t) sizeof( FleetShipsEntry::key ), (uint32_t) offsetof( FleetShipsEntry, key_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "ShipConfig", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetShipsEntry, value ), (uint32_t) sizeof( FleetShipsEntry::value ), 0xffffffffu, 0xffffffffu, &ShipConfigTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "string", 0x3dc94a19365b10ecull, 12, false, false, NULL, NULL, true, false, 32, (uint32_t) offsetof( FleetShipsEntry, key ), (uint32_t) sizeof( FleetShipsEntry::key ), (uint32_t) offsetof( FleetShipsEntry, key_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "ShipConfig", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetShipsEntry, value ), (uint32_t) sizeof( FleetShipsEntry::value ), 0xffffffffu, 0xffffffffu, &ShipConfigTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo FleetShipsEntryTableInfo = { "FleetShipsEntry", (uint32_t) sizeof( FleetShipsEntry ), 2, FleetShipsEntryTableFields, +[]( void * p ) { FleetShipsEntryReset( *(FleetShipsEntry *) p ); }, false }; inline const TableTypeInfo * FleetShipsEntryTableType() { return &FleetShipsEntryTableInfo; } inline const TableFieldInfo FleetByIdEntryTableFields[] = { - { "key", "key", "uint32", 0x3dc94a19365b10ecull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetByIdEntry, key ), (uint32_t) sizeof( FleetByIdEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "ShipConfig", 0x7ce4fd9430e80ceaull, 17, false, true, []( const void * slot ) -> const void * { return (const void *) ShipConfigAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) ShipConfigEmplace( worker, *(TableRef *) slot ); }, false, false, 0, (uint32_t) offsetof( FleetByIdEntry, value ), (uint32_t) sizeof( TableRef ), 0xffffffffu, 0xffffffffu, &ShipConfigTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "uint32", 0x3dc94a19365b10ecull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetByIdEntry, key ), (uint32_t) sizeof( FleetByIdEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "ShipConfig", 0x7ce4fd9430e80ceaull, 17, false, true, []( const void * slot ) -> const void * { return (const void *) ShipConfigAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) ShipConfigEmplace( worker, *(TableRef *) slot ); }, false, false, 0, (uint32_t) offsetof( FleetByIdEntry, value ), (uint32_t) sizeof( TableRef ), 0xffffffffu, 0xffffffffu, &ShipConfigTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo FleetByIdEntryTableInfo = { "FleetByIdEntry", (uint32_t) sizeof( FleetByIdEntry ), 2, FleetByIdEntryTableFields, +[]( void * p ) { FleetByIdEntryReset( *(FleetByIdEntry *) p ); }, true }; inline const TableTypeInfo * FleetByIdEntryTableType() { return &FleetByIdEntryTableInfo; } inline const TableFieldInfo FleetLoadoutsEntryValueEntryTableFields[] = { - { "key", "key", "uint8", 0x3dc94a19365b10ecull, 6, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetLoadoutsEntryValueEntry, key ), (uint32_t) sizeof( FleetLoadoutsEntryValueEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetLoadoutsEntryValueEntry, value ), (uint32_t) sizeof( FleetLoadoutsEntryValueEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "uint8", 0x3dc94a19365b10ecull, 6, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetLoadoutsEntryValueEntry, key ), (uint32_t) sizeof( FleetLoadoutsEntryValueEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetLoadoutsEntryValueEntry, value ), (uint32_t) sizeof( FleetLoadoutsEntryValueEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo FleetLoadoutsEntryValueEntryTableInfo = { "FleetLoadoutsEntryValueEntry", (uint32_t) sizeof( FleetLoadoutsEntryValueEntry ), 2, FleetLoadoutsEntryValueEntryTableFields, +[]( void * p ) { FleetLoadoutsEntryValueEntryReset( *(FleetLoadoutsEntryValueEntry *) p ); }, false }; inline const TableTypeInfo * FleetLoadoutsEntryValueEntryTableType() { return &FleetLoadoutsEntryValueEntryTableInfo; } inline const TableFieldInfo FleetLoadoutsEntryTableFields[] = { - { "key", "key", "string", 0x3dc94a19365b10ecull, 12, false, false, NULL, NULL, true, false, 16, (uint32_t) offsetof( FleetLoadoutsEntry, key ), (uint32_t) sizeof( FleetLoadoutsEntry::key ), (uint32_t) offsetof( FleetLoadoutsEntry, key_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "map[uint8]Item", 0x7ce4fd9430e80ceaull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetLoadoutsEntry, value ), (uint32_t) sizeof( FleetLoadoutsEntry::value ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &FleetLoadoutsEntryValueEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { FleetLoadoutsEntryValueEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint8_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint8_t) key_value ); } return (void *) placed; }, "" }, + { "key", "key", "string", 0x3dc94a19365b10ecull, 12, false, false, NULL, NULL, true, false, 16, (uint32_t) offsetof( FleetLoadoutsEntry, key ), (uint32_t) sizeof( FleetLoadoutsEntry::key ), (uint32_t) offsetof( FleetLoadoutsEntry, key_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "map[uint8]Item", 0x7ce4fd9430e80ceaull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( FleetLoadoutsEntry, value ), (uint32_t) sizeof( FleetLoadoutsEntryValueEntry ), (uint32_t) offsetof( FleetLoadoutsEntry, value.count ), 0xffffffffu, &FleetLoadoutsEntryValueEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { FleetLoadoutsEntryValueEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint8_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint8_t) key_value ); } return (void *) placed; }, "" }, }; inline const TableTypeInfo FleetLoadoutsEntryTableInfo = { "FleetLoadoutsEntry", (uint32_t) sizeof( FleetLoadoutsEntry ), 2, FleetLoadoutsEntryTableFields, +[]( void * p ) { FleetLoadoutsEntryReset( *(FleetLoadoutsEntry *) p ); }, true }; inline const TableTypeInfo * FleetLoadoutsEntryTableType() { return &FleetLoadoutsEntryTableInfo; } inline const TableFieldInfo FleetTiersEntryTableFields[] = { - { "key", "key", "int16", 0x3dc94a19365b10ecull, 3, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetTiersEntry, key ), (uint32_t) sizeof( FleetTiersEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetTiersEntry, value ), (uint32_t) sizeof( FleetTiersEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "int16", 0x3dc94a19365b10ecull, 3, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetTiersEntry, key ), (uint32_t) sizeof( FleetTiersEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetTiersEntry, value ), (uint32_t) sizeof( FleetTiersEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo FleetTiersEntryTableInfo = { "FleetTiersEntry", (uint32_t) sizeof( FleetTiersEntry ), 2, FleetTiersEntryTableFields, +[]( void * p ) { FleetTiersEntryReset( *(FleetTiersEntry *) p ); }, false }; inline const TableTypeInfo * FleetTiersEntryTableType() { return &FleetTiersEntryTableInfo; } inline const TableFieldInfo FleetTableFields[] = { - { "ships", "ships", "map[string(32)]ShipConfig", 0x294a5c4913e1ad44ull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Fleet, ships ), (uint32_t) sizeof( Fleet::ships ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &FleetShipsEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * { if ( key == NULL || key_length > kFleetShipsEntryKeyBound ) { return NULL; } FleetShipsEntry * placed = TableMapPlace( worker, *(TableMap *) slot, key ); if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }, "" }, - { "by_id", "by_id", "map[uint32]*ShipConfig", 0x7b024c46e98d3404ull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Fleet, by_id ), (uint32_t) sizeof( Fleet::by_id ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &FleetByIdEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { FleetByIdEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint32_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint32_t) key_value ); } return (void *) placed; }, "" }, - { "flagship", "flagship", "ShipConfig", 0x63dfa0c4a4b3815dull, 17, false, true, []( const void * slot ) -> const void * { return (const void *) ShipConfigAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) ShipConfigEmplace( worker, *(TableRef *) slot ); }, false, false, 0, (uint32_t) offsetof( Fleet, flagship ), (uint32_t) sizeof( TableRef ), 0xffffffffu, 0xffffffffu, &ShipConfigTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "loadouts", "loadouts", "map[string(16)]map[uint8]Item", 0x294fa1b3f0f5f070ull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Fleet, loadouts ), (uint32_t) sizeof( Fleet::loadouts ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &FleetLoadoutsEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * { if ( key == NULL || key_length > kFleetLoadoutsEntryKeyBound ) { return NULL; } FleetLoadoutsEntry * placed = TableMapPlace( worker, *(TableMap *) slot, key ); if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }, "" }, - { "tiers", "tiers", "map[int16]Item", 0x6dd8dc6c5fdae3ceull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Fleet, tiers ), (uint32_t) sizeof( Fleet::tiers ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &FleetTiersEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { FleetTiersEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (int16_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (int16_t) key_value ); } return (void *) placed; }, "" }, + { "ships", "ships", "map[string(32)]ShipConfig", 0x294a5c4913e1ad44ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Fleet, ships ), (uint32_t) sizeof( FleetShipsEntry ), (uint32_t) offsetof( Fleet, ships.count ), 0xffffffffu, &FleetShipsEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * { if ( key == NULL || key_length > kFleetShipsEntryKeyBound ) { return NULL; } FleetShipsEntry * placed = TableMapPlace( worker, *(TableMap *) slot, key ); if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }, "" }, + { "by_id", "by_id", "map[uint32]*ShipConfig", 0x7b024c46e98d3404ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Fleet, by_id ), (uint32_t) sizeof( FleetByIdEntry ), (uint32_t) offsetof( Fleet, by_id.count ), 0xffffffffu, &FleetByIdEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { FleetByIdEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint32_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint32_t) key_value ); } return (void *) placed; }, "" }, + { "flagship", "flagship", "ShipConfig", 0x63dfa0c4a4b3815dull, 17, false, true, []( const void * slot ) -> const void * { return (const void *) ShipConfigAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) ShipConfigEmplace( worker, *(TableRef *) slot ); }, false, false, 0, (uint32_t) offsetof( Fleet, flagship ), (uint32_t) sizeof( TableRef ), 0xffffffffu, 0xffffffffu, &ShipConfigTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "loadouts", "loadouts", "map[string(16)]map[uint8]Item", 0x294fa1b3f0f5f070ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Fleet, loadouts ), (uint32_t) sizeof( FleetLoadoutsEntry ), (uint32_t) offsetof( Fleet, loadouts.count ), 0xffffffffu, &FleetLoadoutsEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * { if ( key == NULL || key_length > kFleetLoadoutsEntryKeyBound ) { return NULL; } FleetLoadoutsEntry * placed = TableMapPlace( worker, *(TableMap *) slot, key ); if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }, "" }, + { "tiers", "tiers", "map[int16]Item", 0x6dd8dc6c5fdae3ceull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Fleet, tiers ), (uint32_t) sizeof( FleetTiersEntry ), (uint32_t) offsetof( Fleet, tiers.count ), 0xffffffffu, &FleetTiersEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { FleetTiersEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (int16_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (int16_t) key_value ); } return (void *) placed; }, "" }, }; inline const TableTypeInfo FleetTableInfo = { "Fleet", (uint32_t) sizeof( Fleet ), 5, FleetTableFields, +[]( void * p ) { FleetReset( *(Fleet *) p ); }, true }; inline const TableTypeInfo * FleetTableType() { return &FleetTableInfo; } diff --git a/testdata/golden/tables/maps/RowsTable.cpp b/testdata/golden/tables/maps/RowsTable.cpp index 5d6de21fa..537449dfd 100644 --- a/testdata/golden/tables/maps/RowsTable.cpp +++ b/testdata/golden/tables/maps/RowsTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2731,14 +2748,33 @@ inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * inf // ---- json graph walk: end ---- +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + // ---- json map walk: begin ---- -inline bool TableJsonIsMap( const TableFieldInfo * f ) { return f->entry != NULL; } +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} // the entry's two rows: fields[0] IS the key and fields[1] IS the value, which // is what makes a user's own table of pairs the same bytes (§2.8) -inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->entry->fields[0]; } -inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->entry->fields[1]; } +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } @@ -2798,16 +2834,17 @@ inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const // A region holds them in that order already, so this is the array in place. inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) { - const int32_t count = f->map_count( slot ); + const int32_t count = TableJsonExtentCount( slot ); if ( count == 0 ) { out.raw( "{}", 2 ); return true; } const TableFieldInfo * key = TableJsonMapKeyField( f ); const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); out.put( '{' ); for ( int32_t i = 0; i < count; i++ ) { if ( i > 0 ) { out.put( ',' ); } out.line( depth + 1 ); - const void * entry = f->map_at( slot, i ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); TableJsonWriteMapKey( out, entry, key ); out.raw( ": ", 2 ); if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } @@ -2896,15 +2933,15 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf } if ( !fits ) { in.report->kind_mismatch++; place = false; } } - const int32_t before = f->map_count( (const void *) slot ); - void * entry = place ? f->map_insert( *graph->worker, slot, token, token_length, key_value ) : NULL; + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; if ( place && entry == NULL ) { // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the // wire's rule, because a clamped key is a merged entry (§2.8). in.report->clamped++; } - else if ( entry != NULL && f->map_count( (const void *) slot ) == before ) + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) { in.report->duplicate++; // last-wins, the object rule inside the map } @@ -2955,6 +2992,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf // ---- json map walk: end ---- +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace mapdemo #endif // MAPDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/maps/RowsTable.h b/testdata/golden/tables/maps/RowsTable.h index 394e6212b..59f2bb7cf 100644 --- a/testdata/golden/tables/maps/RowsTable.h +++ b/testdata/golden/tables/maps/RowsTable.h @@ -110,6 +110,18 @@ struct TableReport TableMessageReason reason = newer_form; }; + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; // ---- reflection (tables only, docs/SPEC-TABLES.md) ---- // // Static field descriptors for every type in the table closure: name, wire @@ -225,18 +237,13 @@ struct TableFieldInfo // a function pointer at compile time; the arms themselves are a static // inside it). NULL for every other kind. const TableUnionInfo * (*arms)(); - // a MAP (docs/SPEC-TABLES.md §2.8): the generated ENTRY's descriptor — - // fields[0] is the key and fields[1] the value — and the three the ONE - // text walk cannot spell for itself, because TableMap is a type - // it has no name for. NULL on every field that is not a map. - const TableTypeInfo * entry; - int32_t ( * map_count )( const void * slot ); - const void * ( * map_at )( const void * slot, int32_t index ); - // place one entry BY KEY and hand back the entry, at its defaults: a - // string key comes in as the bytes and the length, an integer key as - // the value, and NULL is NOT INSERTED — a key past the bound, or an - // arena that could not carve another segment. - void * ( * map_insert )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded }; @@ -1292,8 +1299,8 @@ struct TableWorker return blob; } - // RAW, ZEROED storage of the bytes asked for, at the alignment asked for — a MAP's builder head and its - // entry segments (docs/SPEC-TABLES.md §2.8). It is not a node: it carries + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries // no type id, takes no index and has no Reset, so it goes through the same // slab and span the blob path uses rather than through Alloc. uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) @@ -1577,12 +1584,6 @@ static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts th // resolving through it yields NULL and can never fabricate the root. static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; -// What a node's storage answers when the FRAMING ITSELF is refused rather than -// merely unnameable: a map whose N cannot fit in its L (docs/SPEC-TABLES.md -// §2.8). An unnameable type id commands no storage and keeps its index; this -// one makes the whole measure answer -1 (§7.6). -static const int64_t kTableNodeRefused = -2; - // ---- the numbering, on the SAVE side ---- // // One entry per reachable node in FIRST-VISIT order, so entry k is node index @@ -1780,9 +1781,9 @@ struct TableNodeDirEntry uint64_t type_id; }; -// a map's extent cursor, defined with the map runtime (docs/SPEC-TABLES.md -// §2.8); the node map names it only through a pointer. -struct TableMapCarve; +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; // TableNodeMap is what a pointer slot resolves through while a body decodes. struct TableNodeMap @@ -1795,17 +1796,21 @@ struct TableNodeMap // takes the SELF-RELATIVE delta so a deref is one add, and the tool's // builder path takes the node's ARENA OFFSET (§6.3). bool arena = false; - // WHERE A MAP'S ENTRIES LAND while this node's body decodes - // (docs/SPEC-TABLES.md §2.8): the node's own extent on the region path - // and the builder's arena on the tool's. It is MUTABLE because the - // cursor belongs to ONE node's decode and the dispatch that owns that - // node holds the map by const reference, exactly as it did before maps - // existed — the decoder's signature does not move for a construct it - // may not carry. - mutable TableMapCarve * carve = NULL; - // and the TOOL's path's allocation front, set once: there a map's - // entries are the builder's arena's rather than a node's extent. + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; }; // TableNodeResolve places one node index in a pointer slot, and every failure @@ -1949,6 +1954,90 @@ inline bool TableNodeScanWhole( TableNodeScan & s ) #endif // MAPDEMO_SCHEMA_TABLE_ARENA +#ifndef MAPDEMO_SCHEMA_TABLE_EXTENT +#define MAPDEMO_SCHEMA_TABLE_EXTENT + +namespace mapdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace mapdemo + +#endif // MAPDEMO_SCHEMA_TABLE_EXTENT + #ifndef MAPDEMO_SCHEMA_TABLE_MAP #define MAPDEMO_SCHEMA_TABLE_MAP @@ -2379,12 +2468,6 @@ inline TableMapEach TableMapEachOf( const TableArena & arena, const Table return each; } -// AN UNREACHED SLOT MUST HOLD NO MAP WITH ENTRIES IN IT (§2.8, §7.6). An empty -// map takes no bytes, so a record whose extent measures ZERO is a record whose -// every by-value map is empty; a measure that REFUSED answers non-zero here -// too, and refusing on it is the same answer one level up. -inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } - // ---- the LOAD side: where a decoded entry lands (§2.8) ---- // // THE READER TRUSTS NOTHING and spends one compare per entry. Every load path @@ -2394,16 +2477,9 @@ inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } // out of the holder node's own extent, and the TOOL's path appends into the // builder's arena, and the decoder above them cannot tell which it has. -// TableMapCarve is a node's extent cursor, PRE-ORDER: a map's whole entry -// array first, then, entry by entry in key order, the arrays of any map an -// entry's value holds by value. The cursor is the node map's, because the -// generated decoder is threaded with that and not with a region. -struct TableMapCarve -{ - uint8_t * at = NULL; // the region path: the node's extent, unspent - int64_t left = 0; - TableWorker * worker = NULL; // the TOOL's path: entries come from the arena -}; +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. // TableMapFill is one map field being decoded: where the next entry lands, and // the entry that last LANDED, which is what the ascending check compares @@ -2528,10 +2604,9 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // LoadMeasure's term for a map is N x sizeof( Entry ) rounded to // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value -// holds a map of its own, the entries' headers under it. The caller owns the -// allocation precisely so it can refuse a number it did not expect. -typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ); - +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect, +// and a refusal carries its reason (§6.5). // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2539,8 +2614,8 @@ typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, i static const int64_t kTableMapEntryFloor = 2; inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, - int64_t entry_size, int64_t entry_align, TableMapWireExtentFn inner, - const TableIdTable * ids ) + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; TableReader r( body, length, &scratch, ids ); @@ -2549,7 +2624,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } const int64_t rest = length - r.offset; - if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { return false; } // an N the map's L cannot carry + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); at += (int64_t) n * entry_size; if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term @@ -2557,49 +2632,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & { uint64_t elem = 0; if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// the same framing walk over an ARRAY OF TABLES that is not a map: its -// elements' own maps are part of this node's extent too -inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each -// length-prefixed element (docs/SPEC-TABLES.md §3.2) -inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t key = 0; - if ( !r.getleb( key ) ) { return true; } - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } r.offset += (int64_t) elem; } return true; @@ -2996,7 +3029,7 @@ inline void RowEntriesEntryReset( RowEntriesEntry & value ) inline void RowReset( Row & value ) { - value.entries.entries.value = 0; // map[string(8)]Item — empty + value.entries.entries.value = 0; // map[string(8)]Item: empty value.entries.count = 0; value.entries.padding = 0; value.after = 0; @@ -3010,7 +3043,7 @@ inline void WideRowEntriesEntryReset( WideRowEntriesEntry & value ) inline void WideRowReset( WideRow & value ) { - value.entries.entries.value = 0; // map[uint32]Item — empty + value.entries.entries.value = 0; // map[uint32]Item: empty value.entries.count = 0; value.entries.padding = 0; value.after = 0; @@ -3868,10 +3901,10 @@ inline bool WideRowLoadBody( TableReader & r, const TableNodeMap & nodes, WideRo } } -// RowWireExtent: the extent Row's maps command, from the FRAMING alone. +// RowWireExtent: the extent Row's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool RowWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool RowWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; // the scan's framing damage is the LOAD's to report TableReader r( body, length, &scratch, ids ); @@ -3890,17 +3923,17 @@ inline bool RowWireExtent( const uint8_t * body, int64_t length, int64_t & at, c if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( RowEntriesEntry ), (int64_t) alignof( RowEntriesEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( RowEntriesEntry ), (int64_t) alignof( RowEntriesEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( !r.skip( field_kind ) ) { return true; } } } -// RowMapExtentAt: the node extent Row's maps take, PRE-ORDER, advancing the -// running offset exactly as RowMapPack advances it (docs/SPEC-TABLES.md §2.8). +// RowExtentAt: the node extent Row's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as RowExtentPack advances it (§2.8, §2.9). template -inline bool RowMapExtentAt( const Ctx & ctx, const Row & value, int64_t & at ) +inline bool RowExtentAt( const Ctx & ctx, const Row & value, int64_t & at ) { { TableMapCursor cursor = TableMapOrder( ctx, value.entries ); @@ -3915,18 +3948,18 @@ inline bool RowMapExtentAt( const Ctx & ctx, const Row & value, int64_t & at ) // the whole extent of one node, from a fresh offset: what a pack reserves // for it beside the record's own storage. template -inline int64_t RowMapExtent( const Ctx & ctx, const Row & value ) +inline int64_t RowExtent( const Ctx & ctx, const Row & value ) { int64_t at = 0; - if ( !RowMapExtentAt( ctx, value, at ) ) { return -1; } + if ( !RowExtentAt( ctx, value, at ) ) { return -1; } return at; } -// RowMapPack: carve Row's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset RowMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// RowExtentPack: carve Row's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset RowExtentAt advances (§2.8, §2.9). template -inline bool RowMapPack( const Ctx & ctx, const Row & src, Row & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool RowExtentPack( const Ctx & ctx, const Row & src, Row & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { { TableMapCursor cursor = TableMapOrder( ctx, src.entries ); @@ -3948,10 +3981,10 @@ inline bool RowMapPack( const Ctx & ctx, const Row & src, Row & dst, uint8_t * e return true; } -// WideRowWireExtent: the extent WideRow's maps command, from the FRAMING alone. +// WideRowWireExtent: the extent WideRow's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool WideRowWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool WideRowWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; // the scan's framing damage is the LOAD's to report TableReader r( body, length, &scratch, ids ); @@ -3970,17 +4003,17 @@ inline bool WideRowWireExtent( const uint8_t * body, int64_t length, int64_t & a if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( WideRowEntriesEntry ), (int64_t) alignof( WideRowEntriesEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( WideRowEntriesEntry ), (int64_t) alignof( WideRowEntriesEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( !r.skip( field_kind ) ) { return true; } } } -// WideRowMapExtentAt: the node extent WideRow's maps take, PRE-ORDER, advancing the -// running offset exactly as WideRowMapPack advances it (docs/SPEC-TABLES.md §2.8). +// WideRowExtentAt: the node extent WideRow's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as WideRowExtentPack advances it (§2.8, §2.9). template -inline bool WideRowMapExtentAt( const Ctx & ctx, const WideRow & value, int64_t & at ) +inline bool WideRowExtentAt( const Ctx & ctx, const WideRow & value, int64_t & at ) { { TableMapCursor cursor = TableMapOrder( ctx, value.entries ); @@ -3995,18 +4028,18 @@ inline bool WideRowMapExtentAt( const Ctx & ctx, const WideRow & value, int64_t // the whole extent of one node, from a fresh offset: what a pack reserves // for it beside the record's own storage. template -inline int64_t WideRowMapExtent( const Ctx & ctx, const WideRow & value ) +inline int64_t WideRowExtent( const Ctx & ctx, const WideRow & value ) { int64_t at = 0; - if ( !WideRowMapExtentAt( ctx, value, at ) ) { return -1; } + if ( !WideRowExtentAt( ctx, value, at ) ) { return -1; } return at; } -// WideRowMapPack: carve WideRow's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset WideRowMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// WideRowExtentPack: carve WideRow's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset WideRowExtentAt advances (§2.8, §2.9). template -inline bool WideRowMapPack( const Ctx & ctx, const WideRow & src, WideRow & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool WideRowExtentPack( const Ctx & ctx, const WideRow & src, WideRow & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { { TableMapCursor cursor = TableMapOrder( ctx, src.entries ); @@ -4270,7 +4303,7 @@ inline bool RowPack( const Ctx & ctx, TablePackMap & seen, const Row & src, Row int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Row ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !RowMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !RowExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return RowPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -4323,7 +4356,7 @@ inline bool WideRowPack( const Ctx & ctx, TablePackMap & seen, const WideRow & s int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( WideRow ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !WideRowMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !WideRowExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return WideRowPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -4415,7 +4448,7 @@ inline bool RowBuilder::Lock() below = RowPackMeasure( ctx, seen, root ); } if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it - int64_t root_extent = RowMapExtent( ctx, root ); + int64_t root_extent = RowExtent( ctx, root ); if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ) + below; // the AUTHORING path may allocate (§6.5), and it does so through the @@ -4513,10 +4546,11 @@ inline uint32_t RowNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t le // already owns. inline void RowNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) { - // the node's own EXTENT, where its maps' entry arrays are carved from, - // PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8). The tool's - // path carries a worker instead: there the entries are the arena's. - TableMapCarve carve; + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; carve.worker = nodes.worker; if ( carve.worker == NULL ) { @@ -4665,7 +4699,7 @@ inline int64_t RowSaveMessage( const RowBuilder & builder, uint8_t * buffer, int // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t RowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t RowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; TableIdTable ids_table; @@ -4678,8 +4712,9 @@ inline int64_t RowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_byte const uint8_t * const wire = wire_file + 1; const int64_t wire_bytes = body_bytes; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -4689,7 +4724,7 @@ inline int64_t RowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_byte { records++; int64_t storage = RowNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -4733,8 +4768,9 @@ inline const Row * RowLoad( uint8_t * region, int64_t region_bytes, const uint8_ int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); int64_t records = 0; { @@ -4817,7 +4853,7 @@ inline const Row * RowLoad( uint8_t * region, int64_t region_bytes, const uint8_ // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Row ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -4833,7 +4869,7 @@ inline const Row * RowLoad( uint8_t * region, int64_t region_bytes, const uint8_ // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t RowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t RowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } @@ -4841,8 +4877,9 @@ inline int64_t RowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t const uint8_t * const wire = message + 1; const int64_t wire_bytes = message_bytes - 1; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -4852,7 +4889,7 @@ inline int64_t RowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t { records++; int64_t storage = RowNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -4887,8 +4924,9 @@ inline const Row * RowLoadMessage( uint8_t * region, int64_t region_bytes, const int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); int64_t records = 0; { @@ -4971,7 +5009,7 @@ inline const Row * RowLoadMessage( uint8_t * region, int64_t region_bytes, const // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Row ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -5026,7 +5064,7 @@ inline bool RowLoadBuilder( RowBuilder & builder, const uint8_t * wire_file, int nodes.entries = directory; nodes.count = records + 1; nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here - nodes.worker = &builder.main; // and a map's entries are the arena's, not a node extent's (§2.8) + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) { TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); int64_t k = 0; @@ -5065,10 +5103,14 @@ inline bool RowLoadBuilder( RowBuilder & builder, const uint8_t * wire_file, int } TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.worker = &builder.main; nodes.carve = &root_carve; bool ok = RowLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; allocator.free( allocator.context, directory ); return ok; } @@ -5154,7 +5196,7 @@ inline bool WideRowBuilder::Lock() below = WideRowPackMeasure( ctx, seen, root ); } if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it - int64_t root_extent = WideRowMapExtent( ctx, root ); + int64_t root_extent = WideRowExtent( ctx, root ); if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( WideRow ) ) + root_extent ) + below; // the AUTHORING path may allocate (§6.5), and it does so through the @@ -5252,10 +5294,11 @@ inline uint32_t WideRowNodeAlloc( uint64_t type_id, TableWorker & worker, int64_ // already owns. inline void WideRowNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) { - // the node's own EXTENT, where its maps' entry arrays are carved from, - // PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8). The tool's - // path carries a worker instead: there the entries are the arena's. - TableMapCarve carve; + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; carve.worker = nodes.worker; if ( carve.worker == NULL ) { @@ -5404,7 +5447,7 @@ inline int64_t WideRowSaveMessage( const WideRowBuilder & builder, uint8_t * buf // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t WideRowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t WideRowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; TableIdTable ids_table; @@ -5417,8 +5460,9 @@ inline int64_t WideRowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_ const uint8_t * const wire = wire_file + 1; const int64_t wire_bytes = body_bytes; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( WideRow ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -5428,7 +5472,7 @@ inline int64_t WideRowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_ { records++; int64_t storage = WideRowNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -5472,8 +5516,9 @@ inline const WideRow * WideRowLoad( uint8_t * region, int64_t region_bytes, cons int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( WideRow ) ) + root_extent ); int64_t records = 0; { @@ -5556,7 +5601,7 @@ inline const WideRow * WideRowLoad( uint8_t * region, int64_t region_bytes, cons // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( WideRow ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -5572,7 +5617,7 @@ inline const WideRow * WideRowLoad( uint8_t * region, int64_t region_bytes, cons // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t WideRowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t WideRowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } @@ -5580,8 +5625,9 @@ inline int64_t WideRowLoadMeasure( const TableVocabulary & vocabulary, const uin const uint8_t * const wire = message + 1; const int64_t wire_bytes = message_bytes - 1; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( WideRow ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -5591,7 +5637,7 @@ inline int64_t WideRowLoadMeasure( const TableVocabulary & vocabulary, const uin { records++; int64_t storage = WideRowNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -5626,8 +5672,9 @@ inline const WideRow * WideRowLoadMessage( uint8_t * region, int64_t region_byte int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( WideRow ) ) + root_extent ); int64_t records = 0; { @@ -5710,7 +5757,7 @@ inline const WideRow * WideRowLoadMessage( uint8_t * region, int64_t region_byte // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( WideRow ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -5765,7 +5812,7 @@ inline bool WideRowLoadBuilder( WideRowBuilder & builder, const uint8_t * wire_f nodes.entries = directory; nodes.count = records + 1; nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here - nodes.worker = &builder.main; // and a map's entries are the arena's, not a node extent's (§2.8) + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) { TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); int64_t k = 0; @@ -5804,10 +5851,14 @@ inline bool WideRowLoadBuilder( WideRowBuilder & builder, const uint8_t * wire_f } TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.worker = &builder.main; nodes.carve = &root_carve; bool ok = WideRowLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; allocator.free( allocator.context, directory ); return ok; } @@ -5887,8 +5938,8 @@ inline void RowEntriesEntryCookBody( uint8_t * at, const RowEntriesEntry & value template inline bool RowCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Row & value, TableByteOrder order ) { - (void) ctx; (void) region; // no reference below this node: the class was decided by a pointer elsewhere in its closure - table_cook_put( at + 0, 0, 8, order ); // entries: the entry array's delta, filled by the extent writer + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // entries: the array's delta, filled by the extent writer table_cook_put( at + 8, 0, 4, order ); // and its count table_cook_put( at + 16, (uint64_t) value.after, 4, order ); return true; @@ -5902,31 +5953,33 @@ inline void WideRowEntriesEntryCookBody( uint8_t * at, const WideRowEntriesEntry template inline bool WideRowCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const WideRow & value, TableByteOrder order ) { - (void) ctx; (void) region; // no reference below this node: the class was decided by a pointer elsewhere in its closure - table_cook_put( at + 0, 0, 8, order ); // entries: the entry array's delta, filled by the extent writer + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // entries: the array's delta, filled by the extent writer table_cook_put( at + 8, 0, 4, order ); // and its count table_cook_put( at + 16, (uint64_t) value.after, 4, order ); return true; } -template inline bool RowEntriesEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const RowEntriesEntry & value, TableByteOrder order ); -template inline bool RowCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Row & value, TableByteOrder order ); -template inline bool WideRowEntriesEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRowEntriesEntry & value, TableByteOrder order ); -template inline bool WideRowCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRow & value, TableByteOrder order ); +template inline bool RowEntriesEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const RowEntriesEntry & value, TableByteOrder order ); +template inline bool RowCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Row & value, TableByteOrder order ); +template inline bool WideRowEntriesEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRowEntriesEntry & value, TableByteOrder order ); +template inline bool WideRowCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRow & value, TableByteOrder order ); -// RowEntriesEntryCookMaps: RowEntriesEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool RowEntriesEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const RowEntriesEntry & value, TableByteOrder order ) +// RowEntriesEntryCookExtent: RowEntriesEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool RowEntriesEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const RowEntriesEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// RowCookMaps: Row's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool RowCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Row & value, TableByteOrder order ) +// RowCookExtent: Row's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool RowCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Row & value, TableByteOrder order ) { - (void) region; // a map's entries carry their own references through their own bodies + (void) region; // a table element's and an entry's references resolve through their own bodies { // entries TableMapCursor cursor = TableMapOrder( ctx, value.entries ); if ( !cursor.ok ) { return false; } @@ -5945,19 +5998,21 @@ template inline bool RowCookMaps( const Ctx & ctx, const TableCoo return true; } -// WideRowEntriesEntryCookMaps: WideRowEntriesEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool WideRowEntriesEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRowEntriesEntry & value, TableByteOrder order ) +// WideRowEntriesEntryCookExtent: WideRowEntriesEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool WideRowEntriesEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRowEntriesEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// WideRowCookMaps: WideRow's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool WideRowCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRow & value, TableByteOrder order ) +// WideRowCookExtent: WideRow's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool WideRowCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRow & value, TableByteOrder order ) { - (void) region; // a map's entries carry their own references through their own bodies + (void) region; // a table element's and an entry's references resolve through their own bodies { // entries TableMapCursor cursor = TableMapOrder( ctx, value.entries ); if ( !cursor.ok ) { return false; } @@ -5976,36 +6031,36 @@ template inline bool WideRowCookMaps( const Ctx & ctx, const Tabl return true; } -// RowEntriesEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// RowEntriesEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool RowEntriesEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const RowEntriesEntry & value, TableByteOrder order ) { RowEntriesEntryCookBody( at, value, order ); int64_t extent_at = 0; - return RowEntriesEntryCookMaps( ctx, region, at + 24, extent_at, at, value, order ); + return RowEntriesEntryCookExtent( ctx, region, at + 24, extent_at, at, value, order ); } -// RowCookNode: one node — the record, then the extent its maps take (§2.8). +// RowCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool RowCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Row & value, TableByteOrder order ) { if ( !RowCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return RowCookMaps( ctx, region, at + 24, extent_at, at, value, order ); + return RowCookExtent( ctx, region, at + 24, extent_at, at, value, order ); } -// WideRowEntriesEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// WideRowEntriesEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool WideRowEntriesEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const WideRowEntriesEntry & value, TableByteOrder order ) { WideRowEntriesEntryCookBody( at, value, order ); int64_t extent_at = 0; - return WideRowEntriesEntryCookMaps( ctx, region, at + 8, extent_at, at, value, order ); + return WideRowEntriesEntryCookExtent( ctx, region, at + 8, extent_at, at, value, order ); } -// WideRowCookNode: one node — the record, then the extent its maps take (§2.8). +// WideRowCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool WideRowCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const WideRow & value, TableByteOrder order ) { if ( !WideRowCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return WideRowCookMaps( ctx, region, at + 24, extent_at, at, value, order ); + return WideRowCookExtent( ctx, region, at + 24, extent_at, at, value, order ); } // RowCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one @@ -6015,15 +6070,15 @@ template inline bool WideRowCookNode( const Ctx & ctx, const Tabl // eight. The offsets go into the region's table when it has one, and are only // summed when it does not (a measure). A type id the numbering carries that // this root cannot name is the two walks disagreeing, and it is refused. -// A NODE'S SIZE DEPENDS ON ITS VALUE where a map rides in its extent +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent // (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context -// the numbering walked and reads the same maps that walk read. +// the numbering walked and reads the same arrays that walk read. template inline bool RowCookLayout( const Ctx & ctx, const Row & root, const TableNumbering & numbering, TableCookRegion & region ) { region.numbering = &numbering; region.count = numbering.count + 1; - const int64_t root_extent = RowMapExtent( ctx, root ); + const int64_t root_extent = RowExtent( ctx, root ); if ( root_extent < 0 ) { return false; } int64_t offset = 24 + root_extent; // the root at zero, its extent behind it int64_t align = 8; @@ -6178,15 +6233,15 @@ inline bool RowCook( const RowBuilder & builder, void * out, uint64_t capacity, // eight. The offsets go into the region's table when it has one, and are only // summed when it does not (a measure). A type id the numbering carries that // this root cannot name is the two walks disagreeing, and it is refused. -// A NODE'S SIZE DEPENDS ON ITS VALUE where a map rides in its extent +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent // (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context -// the numbering walked and reads the same maps that walk read. +// the numbering walked and reads the same arrays that walk read. template inline bool WideRowCookLayout( const Ctx & ctx, const WideRow & root, const TableNumbering & numbering, TableCookRegion & region ) { region.numbering = &numbering; region.count = numbering.count + 1; - const int64_t root_extent = WideRowMapExtent( ctx, root ); + const int64_t root_extent = WideRowExtent( ctx, root ); if ( root_extent < 0 ) { return false; } int64_t offset = 24 + root_extent; // the root at zero, its extent behind it int64_t align = 8; @@ -6399,29 +6454,29 @@ extern const TableTypeInfo WideRowEntriesEntryTableInfo; extern const TableTypeInfo WideRowTableInfo; inline const TableFieldInfo RowEntriesEntryTableFields[] = { - { "key", "key", "string", 0x3dc94a19365b10ecull, 12, false, false, NULL, NULL, true, false, 8, (uint32_t) offsetof( RowEntriesEntry, key ), (uint32_t) sizeof( RowEntriesEntry::key ), (uint32_t) offsetof( RowEntriesEntry, key_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( RowEntriesEntry, value ), (uint32_t) sizeof( RowEntriesEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "string", 0x3dc94a19365b10ecull, 12, false, false, NULL, NULL, true, false, 8, (uint32_t) offsetof( RowEntriesEntry, key ), (uint32_t) sizeof( RowEntriesEntry::key ), (uint32_t) offsetof( RowEntriesEntry, key_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( RowEntriesEntry, value ), (uint32_t) sizeof( RowEntriesEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo RowEntriesEntryTableInfo = { "RowEntriesEntry", (uint32_t) sizeof( RowEntriesEntry ), 2, RowEntriesEntryTableFields, +[]( void * p ) { RowEntriesEntryReset( *(RowEntriesEntry *) p ); }, false }; inline const TableTypeInfo * RowEntriesEntryTableType() { return &RowEntriesEntryTableInfo; } inline const TableFieldInfo RowTableFields[] = { - { "entries", "entries", "map[string(8)]Item", 0xc5b2a72c0845a253ull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Row, entries ), (uint32_t) sizeof( Row::entries ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &RowEntriesEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * { if ( key == NULL || key_length > kRowEntriesEntryKeyBound ) { return NULL; } RowEntriesEntry * placed = TableMapPlace( worker, *(TableMap *) slot, key ); if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }, "" }, - { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Row, after ), (uint32_t) sizeof( Row::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "entries", "entries", "map[string(8)]Item", 0xc5b2a72c0845a253ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Row, entries ), (uint32_t) sizeof( RowEntriesEntry ), (uint32_t) offsetof( Row, entries.count ), 0xffffffffu, &RowEntriesEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * { if ( key == NULL || key_length > kRowEntriesEntryKeyBound ) { return NULL; } RowEntriesEntry * placed = TableMapPlace( worker, *(TableMap *) slot, key ); if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Row, after ), (uint32_t) sizeof( Row::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo RowTableInfo = { "Row", (uint32_t) sizeof( Row ), 2, RowTableFields, +[]( void * p ) { RowReset( *(Row *) p ); }, true }; inline const TableTypeInfo * RowTableType() { return &RowTableInfo; } inline const TableFieldInfo WideRowEntriesEntryTableFields[] = { - { "key", "key", "uint32", 0x3dc94a19365b10ecull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRowEntriesEntry, key ), (uint32_t) sizeof( WideRowEntriesEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRowEntriesEntry, value ), (uint32_t) sizeof( WideRowEntriesEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "uint32", 0x3dc94a19365b10ecull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRowEntriesEntry, key ), (uint32_t) sizeof( WideRowEntriesEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRowEntriesEntry, value ), (uint32_t) sizeof( WideRowEntriesEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo WideRowEntriesEntryTableInfo = { "WideRowEntriesEntry", (uint32_t) sizeof( WideRowEntriesEntry ), 2, WideRowEntriesEntryTableFields, +[]( void * p ) { WideRowEntriesEntryReset( *(WideRowEntriesEntry *) p ); }, false }; inline const TableTypeInfo * WideRowEntriesEntryTableType() { return &WideRowEntriesEntryTableInfo; } inline const TableFieldInfo WideRowTableFields[] = { - { "entries", "entries", "map[uint32]Item", 0xc5b2a72c0845a253ull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRow, entries ), (uint32_t) sizeof( WideRow::entries ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &WideRowEntriesEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { WideRowEntriesEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint32_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint32_t) key_value ); } return (void *) placed; }, "" }, - { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRow, after ), (uint32_t) sizeof( WideRow::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "entries", "entries", "map[uint32]Item", 0xc5b2a72c0845a253ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( WideRow, entries ), (uint32_t) sizeof( WideRowEntriesEntry ), (uint32_t) offsetof( WideRow, entries.count ), 0xffffffffu, &WideRowEntriesEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { WideRowEntriesEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint32_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint32_t) key_value ); } return (void *) placed; }, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRow, after ), (uint32_t) sizeof( WideRow::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo WideRowTableInfo = { "WideRow", (uint32_t) sizeof( WideRow ), 2, WideRowTableFields, +[]( void * p ) { WideRowReset( *(WideRow *) p ); }, true }; inline const TableTypeInfo * WideRowTableType() { return &WideRowTableInfo; } diff --git a/testdata/golden/tables/messages/MessagesTable.cpp b/testdata/golden/tables/messages/MessagesTable.cpp index 9121df2ba..99d41c3b2 100644 --- a/testdata/golden/tables/messages/MessagesTable.cpp +++ b/testdata/golden/tables/messages/MessagesTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace messagedemo #endif // MESSAGEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/pointers/GraphTable.cpp b/testdata/golden/tables/pointers/GraphTable.cpp index 72392fb14..2fe115c19 100644 --- a/testdata/golden/tables/pointers/GraphTable.cpp +++ b/testdata/golden/tables/pointers/GraphTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2750,6 +2767,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace graphdemo #endif // GRAPHDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/pointers/MarksTable.cpp b/testdata/golden/tables/pointers/MarksTable.cpp index 6a01b6738..4a36577ac 100644 --- a/testdata/golden/tables/pointers/MarksTable.cpp +++ b/testdata/golden/tables/pointers/MarksTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2750,6 +2767,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace graphdemo #endif // GRAPHDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/pointers/PartsTable.cpp b/testdata/golden/tables/pointers/PartsTable.cpp index cac4f7cff..28baf609f 100644 --- a/testdata/golden/tables/pointers/PartsTable.cpp +++ b/testdata/golden/tables/pointers/PartsTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2750,6 +2767,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace graphdemo #endif // GRAPHDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/scalars/ScalarsTable.cpp b/testdata/golden/tables/scalars/ScalarsTable.cpp index 10fdae356..2bce9867d 100644 --- a/testdata/golden/tables/scalars/ScalarsTable.cpp +++ b/testdata/golden/tables/scalars/ScalarsTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace scalardemo #endif // SCALARDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/stream/StreamTable.cpp b/testdata/golden/tables/stream/StreamTable.cpp index 13aa93919..9cdb19c3c 100644 --- a/testdata/golden/tables/stream/StreamTable.cpp +++ b/testdata/golden/tables/stream/StreamTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2750,6 +2767,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace streamdemo #endif // STREAMDEMO_SCHEMA_TABLE_JSON From d4a0b4991e5d816766a1218074d53bd965f15b46 Mon Sep 17 00:00:00 2001 From: Glenn Fiedler Date: Sat, 5 Sep 2026 08:33:15 -0700 Subject: [PATCH 6/7] tables: the list test's loop ranges over the count Co-Authored-By: Claude Fable 5.1 --- internal/tablecook/list_test.go | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/internal/tablecook/list_test.go b/internal/tablecook/list_test.go index 0308df824..99701342c 100644 --- a/internal/tablecook/list_test.go +++ b/internal/tablecook/list_test.go @@ -34,7 +34,7 @@ func intsCook(u *ir.Unit, delta int64, count int32) []byte { le.PutUint64(record[0:], uint64(delta)) le.PutUint32(record[8:], uint32(count)) le.PutUint32(record[16:], 5) // after - for i := 0; i < 3; i++ { + for i := range 3 { le.PutUint32(record[24+i*4:], uint32(10*(i+1))) } dir := out[header+data:] From 2352f8be50456094df0801e250e1d688650dbeab Mon Sep 17 00:00:00 2001 From: Glenn Fiedler Date: Sat, 5 Sep 2026 08:56:55 -0700 Subject: [PATCH 7/7] tables: the cold read's five changes on the list codec The update-goldens recipe's for-loop is whole again: the blank line after its first continuation is gone, and GNU Make 3.81 runs it through. The preorder negative control sabotages both writers of a list whose element holds a map, the pack's extent walk and the cook's extent writer, and runs schema cook-check on the cook the sabotaged gate wrote. Both of the instruments the page names go red: the pinned list_of_maps_cook byte compare, and the containment clause, the array leaves the node. The page states the cook-check stopgap under 7.4 item 4: the tool refuses a map slot by name where its scan meets one until schema#380 lands the clause, and the #380 entry in 15 names that clause as the piece owed. The map's wire extent tests the int32 cap before the body's L, so a map count above the cap answers count_over_extent_cap exactly as a list's does, one rule for both constructs. The measure-refusals battery builds a synthetic Army wire with the map's count as its knob and asserts both reasons, with a clean map-holding wire beside them. The eight Table.h goldens the emission moves are re-pinned. The em dash in TestToolRefusesMapsByName's comment is a period. Co-Authored-By: Claude Fable 5.1 --- Makefile | 37 ++++++--- compiler/tablesmaps_test.go | 2 +- docs/SPEC-TABLES.md | 9 +- internal/codegen/cpptable/maps.go | 7 +- test/tables/lists_main.cpp | 91 +++++++++++++++++++-- testdata/golden/tables/lists/HoldersTable.h | 7 +- testdata/golden/tables/lists/MigrateTable.h | 7 +- testdata/golden/tables/lists/ReportTable.h | 7 +- testdata/golden/tables/lists/SaveTable.h | 7 +- testdata/golden/tables/lists/SharedTable.h | 7 +- testdata/golden/tables/maps/DepthTable.h | 7 +- testdata/golden/tables/maps/FleetTable.h | 7 +- testdata/golden/tables/maps/RowsTable.h | 7 +- 13 files changed, 166 insertions(+), 36 deletions(-) diff --git a/Makefile b/Makefile index 612a12ccc..5d223f225 100644 --- a/Makefile +++ b/Makefile @@ -2725,9 +2725,10 @@ tables-lists: build/schema_test_lists build/schema_test_lists_asan @grep -q "leaves\|extent" build/lists-cooks/forged.log || { echo "LIST GATE FAILED: the forgery was refused, but not on the element-array clause"; cat build/lists-cooks/forged.log; exit 1; } @echo "list gate: cook-check reads two cooks the runtime wrote, refuses the forged list slot, and refuses the map-holding cook by name" -# THE FOUR LoadMeasure REFUSALS are a unit test and not a `report` row (§2.9, -# §6.5): each wire is built in memory with a SYNTHETIC count, and the answer -# and the REASON are asserted, with a clean wire beside them that must measure. +# THE SIX LoadMeasure REFUSALS are a unit test and not a `report` row (§2.8, +# §2.9, §6.5): each wire is built in memory with a SYNTHETIC count, a list's +# and a map's, and the answer and the REASON are asserted, with a clean wire +# beside them that must measure. .PHONY: tables-list-measure-refusals tables-list-measure-refusals: build/schema_test_lists ./build/schema_test_lists measure-refusals @@ -2801,14 +2802,31 @@ tables-lists-dead-element-negative-control: bin/schema build/tables-generated/.s $(call list_negative_control,dead,'s@if ( !TableListSegmentDead( segment->dead, within ) ) { break; }@break; // SABOTAGED@',internal/codegen/cpptable/lists.go,a dead element riding on the wire left the list gate GREEN) # THE ELEMENT ARRAY IS LAID OUT AFTER A NESTED CONTAINER'S, breaking the -# pre-order rule: the cook's extent writer no longer steps past the element -# array before it lays each element's map, so the maps land where the elements -# are. `list_of_maps` meets it: the pinned cook's byte compare goes red, and -# `schema cook-check`'s no-overlap clause refuses the file the gate wrote. +# pre-order rule in BOTH writers of a list whose element holds a map: the +# pack's extent walk stops reserving the element array ahead of the elements' +# maps, and the cook's extent writer stops stepping past it, so the maps are +# laid where the elements are and the node's extent is short of the array. +# `list_of_maps` meets it, and the two instruments §2.9 names go red +# together: the pinned cook's byte compare, and `schema cook-check`'s +# containment clause on the cook the sabotaged gate wrote, which must be that +# clause and not the map slot's refusal by name. .PHONY: tables-lists-preorder-negative-control tables-lists-preorder-negative-control: bin/schema build/tables-generated/.stamp - $(call list_negative_control,preorder,'s@g.pf(" at += (int64_t) cursor.count \* %d; // the whole array FIRST\\n", size)@g.pf(" // SABOTAGED: %d\\n", size)@',internal/codegen/cpptable/extent.go,laying the element array after a nested container left the list gate GREEN) - + @mkdir -p build + @printf 'func listElementHoldsMap(f *ir.Field) bool {\n\tref := listElementStruct(f)\n\tif ref == nil {\n\t\treturn false\n\t}\n\tfor i := range ref.Fields {\n\t\tif ref.Fields[i].IsMap() {\n\t\t\treturn true\n\t\t}\n\t}\n\treturn false\n}\n' > build/list-preorder-helper.txt + $(call list_negative_control,preorder,'s@g.pf(" at += (int64_t) cursor.count \* %d; // the whole array FIRST\\n", size)@if !listElementHoldsMap(f) { g.pf(" at += (int64_t) cursor.count * %d; // the whole array FIRST\\n", size) } // SABOTAGED@' -e 's@g.pf("%s at += (int64_t) cursor.count \* (int64_t) sizeof( %s ); // the whole array FIRST\\n", ind, elem)@if !listElementHoldsMap(f) { g.pf("%s at += (int64_t) cursor.count * (int64_t) sizeof( %s ); // the whole array FIRST\\n", ind, elem) } // SABOTAGED@' -e '$$r build/list-preorder-helper.txt',internal/codegen/cpptable/extent.go,laying the element array after a nested container left the list gate GREEN) + @grep -c "SABOTAGED" build/list-preorder.gotext | grep -qx 2 || \ + { echo "NEGATIVE CONTROL FAILED: the preorder sabotage did not reach both the pack's and the cook's list branch"; exit 1; } + @grep -q "^FAIL.*list_of_maps_cook" build/list-preorder.log || \ + { echo "NEGATIVE CONTROL FAILED: the pinned list_of_maps_cook byte compare stayed GREEN"; cat build/list-preorder.log; exit 1; } + @rm -rf build/tables-list-preorder/cooks && mkdir -p build/tables-list-preorder/cooks + @SCHEMA_LIST_COOK_DIR=build/tables-list-preorder/cooks ./build/schema_test_lists_preorder > /dev/null 2>&1 || true + @if ./bin/schema cook-check --root Army build/tables-list-preorder/cooks/army.cook tables/lists > build/list-preorder-check.log 2>&1; then \ + echo "NEGATIVE CONTROL FAILED: cook-check accepted the cook the sabotaged writer laid"; exit 1; \ + fi + @grep -q "the array leaves the node\|overlaps another array" build/list-preorder-check.log || \ + { echo "NEGATIVE CONTROL FAILED: cook-check refused the sabotaged cook, but not on the containment clause"; cat build/list-preorder-check.log; exit 1; } + @echo "negative control: preorder turns the pinned list_of_maps_cook compare and cook-check's containment clause red: $$(grep -o 'the array leaves the node\|overlaps another array' build/list-preorder-check.log | head -1)" # THE WALK VISITS LISTS OUT OF DECLARATION ORDER, grouped after the pointer # fields: the edge walk is made to take every list field last, so @@ -2909,7 +2927,6 @@ update-goldens: build/schema_test build/schema_test_ludicrous build/schema_test_ SCHEMA_UPDATE_WIRE_GOLDENS=1 ./build/schema_test_maps SCHEMA_UPDATE_WIRE_GOLDENS=1 ./build/schema_test_lists @for d in examples pointers block blockhome messages stream blobs scalars maps lists; do \ - mkdir -p testdata/golden/tables/$$d; \ cp build/tables-generated/$$d/*Table.h build/tables-generated/$$d/*Table.cpp testdata/golden/tables/$$d/ 2>/dev/null || true; \ done diff --git a/compiler/tablesmaps_test.go b/compiler/tablesmaps_test.go index 63b371fd4..f56be1e5b 100644 --- a/compiler/tablesmaps_test.go +++ b/compiler/tablesmaps_test.go @@ -271,7 +271,7 @@ func TestMapEntryIsNotARoot(t *testing.T) { } // TestToolRefusesMapsByName: the tool's WIRE and TEXT halves carry maps now -// (docs/SPEC-TABLES.md §2.8), and its COOK half does not — so the cook and +// (docs/SPEC-TABLES.md §2.8), and its COOK half does not. So the cook and // uncook surfaces refuse a map-bearing unit BY NAME rather than laying out an // entry array they have no placement for. Without the refusal a caller gets a // cook whose region is short of the entries, which is worse than a diagnostic. diff --git a/docs/SPEC-TABLES.md b/docs/SPEC-TABLES.md index 04057105e..ecbecbd9b 100644 --- a/docs/SPEC-TABLES.md +++ b/docs/SPEC-TABLES.md @@ -7545,7 +7545,10 @@ no reference: the check exactly as the pack order does. The entries' own slots, companions and tags are then walked as a bounded array's elements are. The KEYS are read too, ascending with no repeat, because a cook `Find` cannot - search is a forgery. + search is a forgery. Until schema#380 lands this clause in the tool, `schema + cook-check` refuses a map slot by name where its scan meets one, so a + cook that holds one is refused rather than walked past, and the C++ + reference reads it. 5. **Every UNBOUNDED-ARRAY SLOT** (§2.9). The same four clauses as a map's: CONTAINMENT, ALIGNMENT, FIT and NO OVERLAP, against the holder's own extent and against every other element or entry array in that node, and then the @@ -10559,7 +10562,9 @@ inspects everything in the schema built: C++ reference and the tool are first: the builder surface (insert, erase, find, iterate), the sort in the four walks, the region load's ascending check with its `duplicate` and `malformed` events, the const `Find`, the text - form's object and `schema cook-check`'s order check. What a port needs is the entry as an ordinary array-of-tables element in its + form's object and `schema cook-check`'s map-slot clause with its order + check, which is the one piece still owed: the tool refuses a map slot by + name until it lands (§7.4). What a port needs is the entry as an ordinary array-of-tables element in its measure, save and load, the writer's sort, the reader's one compare with its two events, the const `Find` as a binary search that allocates nothing, ascending iteration, and the text form's keyed object; each holds the same diff --git a/internal/codegen/cpptable/maps.go b/internal/codegen/cpptable/maps.go index f842edec8..9e1809a19 100644 --- a/internal/codegen/cpptable/maps.go +++ b/internal/codegen/cpptable/maps.go @@ -661,8 +661,10 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value // holds a map or a list of its own, the entries' headers under it. The caller -// owns the allocation precisely so it can refuse a number it did not expect, -// and a refusal carries its reason (§6.5). +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -679,6 +681,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); diff --git a/test/tables/lists_main.cpp b/test/tables/lists_main.cpp index a5cf015fd..85039ab10 100644 --- a/test/tables/lists_main.cpp +++ b/test/tables/lists_main.cpp @@ -7,7 +7,7 @@ // Compiled WITHOUT the serialize include path: the Table headers stand alone. // // schema_test_lists every battery -// schema_test_lists measure-refusals the four LoadMeasure refusals alone +// schema_test_lists measure-refusals the six LoadMeasure refusals alone // (make tables-list-measure-refusals) #include @@ -777,14 +777,15 @@ static void test_clamp_control() free( wire ); } -// ---- THE FOUR LoadMeasure REFUSALS (docs/SPEC-TABLES.md §2.9, §6.5) ---- +// ---- THE SIX LoadMeasure REFUSALS (docs/SPEC-TABLES.md §2.8, §2.9, §6.5) ---- // // A unit test and not a `report` row, because a refusal produces no counters. // Each wire is built in memory with a SYNTHETIC count rather than a golden: // a count above the int32 cap, which no golden could carry because the file // would be two gigabytes, a count whose elements cannot fit the field's L, the -// same two at DEPTH, inside an element's own list, and a clean wire beside -// them, which must measure. Red if any of the four answers something other +// same two at DEPTH, inside an element's own list, the same two inside an +// element's MAP, which answers by the one rule a list does, and a clean wire beside +// them, which must measure. Red if any of the six answers something other // than -1 with its own reason, if the clean one refuses, or if any of them // moves one of the report's counters. @@ -822,6 +823,51 @@ static Wire build_sheet( uint64_t rows, uint64_t items, int32_t real_items ) return w; } +// an `Army` body written FROM THE GRAMMAR: a kind 14 array of kind 13 `Squad` +// elements, each holding its `roster` MAP as a kind 14 array of kind 13 +// entries (§2.8), with the map's declared count a knob of its own +static Wire build_army( uint64_t squads, uint64_t entries, int32_t real_entries ) +{ + WireBuilder b; + b.field( "squads", 14 ); + const int64_t body = b.open_len(); + b.u8( 13 ); + b.leb( squads ); + { + const int64_t squad = b.open_len(); + b.field( "roster", 14 ); + const int64_t inner = b.open_len(); + b.u8( 13 ); + b.leb( entries ); + for ( int32_t i = 0; i < real_entries; i++ ) + { + const int64_t entry = b.open_len(); + b.field( "key", 6 ); + b.u8( (uint8_t) ( 2 + i ) ); + b.field( "value", 13 ); + const int64_t item = b.open_len(); + b.field( "count", 4 ); + b.u32( (uint32_t) ( 20 + i ) ); + b.end(); + b.close_len( item ); + b.end(); + b.close_len( entry ); + } + b.close_len( inner ); + b.field( "name", 4 ); + b.u32( 50 ); + b.end(); + b.close_len( squad ); + } + b.close_len( body ); + b.field( "after", 4 ); + b.u32( 8 ); + b.end(); + Wire w; + w.size = b.finish( w.bytes ); + return w; +} + static void test_measure_refusals() { // a count above the int32 cap, at the ROOT @@ -865,6 +911,41 @@ static void test_measure_refusals() CHECK_EQ( SheetLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); CHECK( reason == count_over_length ); } + // the same two at DEPTH, inside an element's MAP (§2.8): a map's term + // answers the reasons a list's does, the int32 cap first, one rule for + // both constructs (§6.5) + { + Wire w = build_army( 1, 0x80000000ull, 1 ); + TableRefuseReason reason = count_over_length; + CHECK_EQ( ArmyLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_extent_cap ); + } + { + Wire w = build_army( 1, 100000, 1 ); + TableRefuseReason reason = count_over_extent_cap; + CHECK_EQ( ArmyLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_length ); + } + // and a clean map-holding wire beside them, which must measure and load silently + { + Wire w = build_army( 1, 2, 2 ); + TableRefuseReason reason = count_over_length; + const int64_t need = ArmyLoadMeasure( w.bytes, w.size, NULL, &reason ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Army * army = ArmyLoad( region, need, w.bytes, w.size, &r ); + CHECK( army != NULL ); + report_silent( r, "the clean map-holding wire beside the refusals" ); + if ( army != NULL && army->squads.size() == 1 ) + { + CHECK_EQ( army->squads[0].roster.size(), 2 ); + const Item * item = army->squads[0].roster.Find( (uint8_t) 3 ); + CHECK( item != NULL && item->count == 21 ); + } + free( region ); + } // and a clean wire beside them, which must measure and load silently { Wire w = build_sheet( 1, 2, 2 ); @@ -1378,7 +1459,7 @@ int main( int argc, char ** argv ) printf( "\n%d measure refusal check(s) failed\n", failures ); return 1; } - printf( "list measure refusals: four -1s with their reasons, one clean measure, no counter moved (docs/SPEC-TABLES.md §2.9, §6.5)\n" ); + printf( "list measure refusals: six -1s with their reasons, two clean measures, no counter moved (docs/SPEC-TABLES.md §2.8, §2.9, §6.5)\n" ); return 0; } test_writer(); diff --git a/testdata/golden/tables/lists/HoldersTable.h b/testdata/golden/tables/lists/HoldersTable.h index 6ae1b0d5b..e3bf2cfa0 100644 --- a/testdata/golden/tables/lists/HoldersTable.h +++ b/testdata/golden/tables/lists/HoldersTable.h @@ -2519,8 +2519,10 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value // holds a map or a list of its own, the entries' headers under it. The caller -// owns the allocation precisely so it can refuse a number it did not expect, -// and a refusal carries its reason (§6.5). +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2537,6 +2539,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); diff --git a/testdata/golden/tables/lists/MigrateTable.h b/testdata/golden/tables/lists/MigrateTable.h index b7c7fe283..9bc92d3cd 100644 --- a/testdata/golden/tables/lists/MigrateTable.h +++ b/testdata/golden/tables/lists/MigrateTable.h @@ -2519,8 +2519,10 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value // holds a map or a list of its own, the entries' headers under it. The caller -// owns the allocation precisely so it can refuse a number it did not expect, -// and a refusal carries its reason (§6.5). +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2537,6 +2539,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); diff --git a/testdata/golden/tables/lists/ReportTable.h b/testdata/golden/tables/lists/ReportTable.h index b8b99f464..5139791b6 100644 --- a/testdata/golden/tables/lists/ReportTable.h +++ b/testdata/golden/tables/lists/ReportTable.h @@ -2519,8 +2519,10 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value // holds a map or a list of its own, the entries' headers under it. The caller -// owns the allocation precisely so it can refuse a number it did not expect, -// and a refusal carries its reason (§6.5). +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2537,6 +2539,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); diff --git a/testdata/golden/tables/lists/SaveTable.h b/testdata/golden/tables/lists/SaveTable.h index a6400e50d..3566e3b49 100644 --- a/testdata/golden/tables/lists/SaveTable.h +++ b/testdata/golden/tables/lists/SaveTable.h @@ -2519,8 +2519,10 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value // holds a map or a list of its own, the entries' headers under it. The caller -// owns the allocation precisely so it can refuse a number it did not expect, -// and a refusal carries its reason (§6.5). +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2537,6 +2539,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); diff --git a/testdata/golden/tables/lists/SharedTable.h b/testdata/golden/tables/lists/SharedTable.h index 1442d4542..de8d0934d 100644 --- a/testdata/golden/tables/lists/SharedTable.h +++ b/testdata/golden/tables/lists/SharedTable.h @@ -2519,8 +2519,10 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value // holds a map or a list of its own, the entries' headers under it. The caller -// owns the allocation precisely so it can refuse a number it did not expect, -// and a refusal carries its reason (§6.5). +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2537,6 +2539,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); diff --git a/testdata/golden/tables/maps/DepthTable.h b/testdata/golden/tables/maps/DepthTable.h index 347219edb..ccff4552c 100644 --- a/testdata/golden/tables/maps/DepthTable.h +++ b/testdata/golden/tables/maps/DepthTable.h @@ -2605,8 +2605,10 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value // holds a map or a list of its own, the entries' headers under it. The caller -// owns the allocation precisely so it can refuse a number it did not expect, -// and a refusal carries its reason (§6.5). +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2623,6 +2625,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); diff --git a/testdata/golden/tables/maps/FleetTable.h b/testdata/golden/tables/maps/FleetTable.h index 02df5cfd9..057861ccb 100644 --- a/testdata/golden/tables/maps/FleetTable.h +++ b/testdata/golden/tables/maps/FleetTable.h @@ -2604,8 +2604,10 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value // holds a map or a list of its own, the entries' headers under it. The caller -// owns the allocation precisely so it can refuse a number it did not expect, -// and a refusal carries its reason (§6.5). +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2622,6 +2624,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); diff --git a/testdata/golden/tables/maps/RowsTable.h b/testdata/golden/tables/maps/RowsTable.h index 59f2bb7cf..f538e3f9b 100644 --- a/testdata/golden/tables/maps/RowsTable.h +++ b/testdata/golden/tables/maps/RowsTable.h @@ -2605,8 +2605,10 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value // holds a map or a list of its own, the entries' headers under it. The caller -// owns the allocation precisely so it can refuse a number it did not expect, -// and a refusal carries its reason (§6.5). +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2623,6 +2625,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 );