diff --git a/Makefile b/Makefile index e9f360adf..5d223f225 100644 --- a/Makefile +++ b/Makefile @@ -36,6 +36,8 @@ SCHEMAS_TABLES_BLOBS := $(wildcard tables/blobs/*.schema) SCHEMAS_TABLES_SCALARS := $(wildcard tables/scalars/*.schema) # the MAP corpus (docs/SPEC-TABLES.md §2.8) SCHEMAS_TABLES_MAPS := $(wildcard tables/maps/*.schema) +# the UNBOUNDED ARRAY corpus (docs/SPEC-TABLES.md §2.9) +SCHEMAS_TABLES_LISTS := $(wildcard tables/lists/*.schema) # the MESSAGE FORM's corpora (docs/SPEC-TABLES.md §3.3): the three backend # messages the ruling measured, and the WIDE-VOCABULARY unit test/vocabgen # writes, whose vocabulary passes 127 ids so its message names slots on both @@ -134,6 +136,7 @@ define tables_generate $(1) generate --lang cpp --out $(2)/a2 test/tables/A2.schema $(1) generate --lang cpp --out $(2)/scalars tables/scalars $(1) generate --lang cpp --out $(2)/maps tables/maps + $(1) generate --lang cpp --out $(2)/lists tables/lists $(1) generate --lang cpp --out $(2)/scalars2 test/tables/Scalars2.schema $(1) generate --lang cpp --out $(2)/backend tables/backend $(1) generate --lang cpp --out $(2)/vocab tables/vocab @@ -141,9 +144,9 @@ endef tables_includes = -I$(1)/examples -I$(1)/pointers -I$(1)/block -I$(1)/blockhome -Itest/tables \ -I$(1)/v1 -I$(1)/v2 -I$(1)/p1 -I$(1)/p2 -I$(1)/p3 -I$(1)/jsonkeys \ - -I$(1)/messages -I$(1)/stream -I$(1)/blobs -I$(1)/m1 -I$(1)/m2 -I$(1)/a1 -I$(1)/a2 -I$(1)/g1 -I$(1)/k1 -I$(1)/k2 -I$(1)/scalars -I$(1)/scalars2 -I$(1)/maps -I$(1)/backend -I$(1)/vocab -I$(SERIALIZE) + -I$(1)/messages -I$(1)/stream -I$(1)/blobs -I$(1)/m1 -I$(1)/m2 -I$(1)/a1 -I$(1)/a2 -I$(1)/g1 -I$(1)/k1 -I$(1)/k2 -I$(1)/scalars -I$(1)/scalars2 -I$(1)/maps -I$(1)/lists -I$(1)/backend -I$(1)/vocab -I$(SERIALIZE) -build/tables-generated/.stamp: bin/schema $(SCHEMAS_TABLES) $(SCHEMAS_TABLES_POINTERS) $(SCHEMAS_TABLES_BLOCK) $(SCHEMAS_TABLES_MESSAGES) $(SCHEMAS_TABLES_BLOBS) $(SCHEMAS_TABLES_SCALARS) $(SCHEMAS_TABLES_MAPS) $(SCHEMAS_TABLES_BACKEND) $(SCHEMAS_TABLES_VOCAB) test/tables/V1.schema test/tables/V2.schema test/tables/P1.schema test/tables/P2.schema test/tables/P3.schema test/tables/JsonKeys.schema test/tables/M1.schema test/tables/M2.schema test/tables/A1.schema test/tables/A2.schema test/tables/G1.schema test/tables/K1.schema test/tables/K2.schema test/tables/Scalars2.schema +build/tables-generated/.stamp: bin/schema $(SCHEMAS_TABLES) $(SCHEMAS_TABLES_POINTERS) $(SCHEMAS_TABLES_BLOCK) $(SCHEMAS_TABLES_MESSAGES) $(SCHEMAS_TABLES_BLOBS) $(SCHEMAS_TABLES_SCALARS) $(SCHEMAS_TABLES_MAPS) $(SCHEMAS_TABLES_LISTS) $(SCHEMAS_TABLES_BACKEND) $(SCHEMAS_TABLES_VOCAB) test/tables/V1.schema test/tables/V2.schema test/tables/P1.schema test/tables/P2.schema test/tables/P3.schema test/tables/JsonKeys.schema test/tables/M1.schema test/tables/M2.schema test/tables/A1.schema test/tables/A2.schema test/tables/G1.schema test/tables/K1.schema test/tables/K2.schema test/tables/Scalars2.schema @mkdir -p build/tables-generated $(call tables_generate,./bin/schema,build/tables-generated) @touch $@ @@ -165,7 +168,7 @@ build/tables-generated/.stamp: bin/schema $(SCHEMAS_TABLES) $(SCHEMAS_TABLES_POI # THE SCAN IS BY SYMBOL, not by line, and `TableNode` is matched with its whole # spelling so a node symbol nobody has written yet is still refused. Exactly one # of those spellings is ALLOWED in a pointer-free unit and it is named below. -TABLES_ZERO_COST_SYMBOLS := TableArena|TableSlot|TableWorker|TableRef|TableRegion|kTableSegment|kTableSlab|TablePack|[A-Za-z_]*TableNode[A-Za-z_]*|is_pointer|Builder|PackMeasure|LoadMeasure|TableBlob|TableBytesView|TableStringView|AllocBytes|AllocString|BytesEmplace|StringEmplace|TableMap|TableMapHead|TableMapSegment|TableMapOrder|TableMapCursor|TableEntryKey|TableKeyOrder|TableResetMapValue|TableEntrySetKey|kTableMapSegment +TABLES_ZERO_COST_SYMBOLS := TableArena|TableSlot|TableWorker|TableRef|TableRegion|kTableSegment|kTableSlab|TablePack|[A-Za-z_]*TableNode[A-Za-z_]*|is_pointer|Builder|PackMeasure|LoadMeasure|TableBlob|TableBytesView|TableStringView|AllocBytes|AllocString|BytesEmplace|StringEmplace|TableMap|TableMapHead|TableMapSegment|TableMapOrder|TableMapCursor|TableEntryKey|TableKeyOrder|TableResetMapValue|TableEntrySetKey|kTableMapSegment|TableList|TableListHead|TableListSegment|TableListCursor|TableListElements|TableListPlace|kTableListSegment|TableExtentCarve|TableExtentUnreachedEmpty|TableWireExtent|TableRefuseReason|count_over_length|count_over_extent_cap # THE ONE NODE SPELLING A POINTER-FREE UNIT CARRIES, and it is the FORM's and # not the pointer machinery's (docs/SPEC-TABLES.md §3, §3.1): the reserved @@ -190,10 +193,10 @@ tables-zero-cost: build/tables-generated/.stamp @for f in $(TABLES_ZERO_COST_HEADERS); do \ if grep -ohE "$(TABLES_ZERO_COST_SYMBOLS)" $$f | grep -vxE "$(TABLES_ZERO_COST_ALLOWED)" | sort -u | grep -q .; then \ grep -nE "$(TABLES_ZERO_COST_SYMBOLS)" $$f | grep -vE "$(TABLES_ZERO_COST_ALLOWED)"; \ - echo "ZERO-COST GATE FAILED: pointer or map machinery leaked into $$f"; exit 1; \ + echo "ZERO-COST GATE FAILED: pointer, map or list machinery leaked into $$f"; exit 1; \ fi; \ done - @echo "tables zero-cost gate: value-only tables carry no pointer or map machinery" + @echo "tables zero-cost gate: value-only tables carry no pointer, map or list machinery" # THE NEGATIVE CONTROL. The gate above sanctions ONE node spelling, so it owes a # demonstration that it still refuses the others: the nearest neighbour of the @@ -1011,7 +1014,7 @@ tables-block-zero-cost: build/tables-generated/.stamp build/tables-generated-cs/ testdata/golden/tables/block/*Table.* testdata/golden/tables/blockhome/*Table.* \ testdata/golden/tables/messages/*Table.* testdata/golden/tables/stream/*Table.* \ testdata/golden/tables/blobs/*Table.* testdata/golden/tables/scalars/*Table.* \ - testdata/golden/tables/maps/*Table.* ; do \ + testdata/golden/tables/maps/*Table.* testdata/golden/tables/lists/*Table.* ; do \ dir=$$(basename $$(dirname $$f)); \ n=$$(( n + 1 )); \ cmp -s $$f build/tables-generated/$$dir/$$(basename $$f) || \ @@ -2066,6 +2069,12 @@ build/schema_test_maps_be: build/tables-generated/.stamp test/tables/maps_main.c @mkdir -p build $(BE_CXX) $(TABLES_CXXFLAGS) -static $(TABLES_INCLUDES) test/tables/maps_main.cpp $(MAPS_SOURCES) -o $@ +# and the LIST gate on the same host: a list's element array is the same bytes +# on both hosts because the count and the reference are the region's own scalars +build/schema_test_lists_be: build/tables-generated/.stamp test/tables/lists_main.cpp + @mkdir -p build + $(BE_CXX) $(TABLES_CXXFLAGS) -static $(TABLES_INCLUDES) test/tables/lists_main.cpp $(LISTS_SOURCES) -o $@ + # The COOK's read side, for a BIG-ENDIAN target. A cook is produced in the byte # order of the build it is cooked for (docs/SPEC-TABLES.md §7), so this is where that # stops being a sentence: the big-endian build opens the big-endian cook @@ -2088,10 +2097,11 @@ build/schema_test_block_endian_be: build/tables-generated/.stamp test/tables/blo $(BE_CXX) $(BLOCK_CXXFLAGS) -static $(BLOCK_INCLUDES) test/tables/block_endian_main.cpp $(BLOCK_SOURCES) -o $@ .PHONY: tables-big-endian -tables-big-endian: build/schema_test_tables_be build/schema_test_maps_be build/schema_test_block_endian build/schema_test_block_endian_be build/schema_test_cook build/schema_test_cook_be build/cook-open/.stamp +tables-big-endian: build/schema_test_tables_be build/schema_test_maps_be build/schema_test_lists_be build/schema_test_block_endian build/schema_test_block_endian_be build/schema_test_cook build/schema_test_cook_be build/cook-open/.stamp $(BE_RUN) ./build/schema_test_tables_be $(BE_RUN) ./build/schema_test_maps_be - @echo "big-endian leg: the wire crosses the byte order, a map's framing and its sorted entry array with it" + $(BE_RUN) ./build/schema_test_lists_be + @echo "big-endian leg: the wire crosses the byte order, a map's framing and its sorted entry array with it, and a list's element array" ./build/schema_test_block_endian write build/block-host.bin $(BE_RUN) ./build/schema_test_block_endian_be write build/block-target.bin $(BE_RUN) ./build/schema_test_block_endian_be accept build/block-target.bin @@ -2421,6 +2431,10 @@ test: build/schema_test build/schema_test_guard build/schema_test_tables build/s $(MAKE) tables-maps $(MAKE) tables-json-map-walk $(MAKE) tables-maps-negative-controls + $(MAKE) tables-lists + $(MAKE) tables-list-measure-refusals + $(MAKE) tables-json-list-walk + $(MAKE) tables-lists-negative-controls $(MAKE) tables-json-walk $(MAKE) tables-json-graph-walk $(MAKE) tables-json-negative-control @@ -2663,17 +2677,256 @@ tables-maps-negative-controls: tables-maps-sort-negative-control \ tables-maps-text-order-negative-control \ tables-maps-unreached-negative-control +# ---- THE LIST GATE (docs/SPEC-TABLES.md §2.9) ------------------------------ +# +# One binary over the `tables/lists` corpus: the builder's three, the four +# writing walks in INDEX order, the node extent a region and a cook carry, +# every reader rule §2.9 states, the migration golden, and the clamp control +# at 100,000. Its wire goldens are the reference's, pinned like every other +# table golden, and the cooks it writes are read by `schema cook-check`, whose +# scan carries §7.4's element-array clause, beside a FORGERY whose list slot +# points its array past the holder's extent, which the tool must refuse. +# +# THE SANITIZED TWIN rides beside it for the map gate's reason: segments, an +# element array carved from a node's own extent and indexing over mapped +# bytes are lifetime and bounds questions -Werror cannot see. + +LISTS_SOURCES = $$(ls build/tables-generated/lists/*Table.cpp) + +build/schema_test_lists: build/tables-generated/.stamp test/tables/lists_main.cpp + @mkdir -p build + $(CXX) $(TABLES_CXXFLAGS) $(TABLES_INCLUDES) test/tables/lists_main.cpp $(LISTS_SOURCES) -o $@ + +build/schema_test_lists_asan: build/tables-generated/.stamp test/tables/lists_main.cpp + @mkdir -p build + $(CXX) $(TABLES_CXXFLAGS) -fsanitize=address,undefined -fno-omit-frame-pointer -g \ + $(TABLES_INCLUDES) test/tables/lists_main.cpp $(LISTS_SOURCES) -o $@ + +.PHONY: tables-lists +tables-lists: build/schema_test_lists build/schema_test_lists_asan + @rm -rf build/lists-cooks && mkdir -p build/lists-cooks + SCHEMA_LIST_COOK_DIR=build/lists-cooks ./build/schema_test_lists + ./build/schema_test_lists_asan + # `schema cook-check` reads what the runtime cooked (§7.4): the root's list + # slot, every element's own slots and companions, and a pointed-at holder's + # list, and refuses the forgery beside them. The cook whose element holds a + # MAP is refused by name at the map slot, because the tool's map-slot + # clause is schema#380's next PR, and the refusal must be that one and not + # a list clause's + ./bin/schema cook-check --root Save build/lists-cooks/save.cook tables/lists + ./bin/schema cook-check --root Sheet build/lists-cooks/sheet.cook tables/lists + @if ./bin/schema cook-check --root Army build/lists-cooks/army.cook tables/lists > build/lists-cooks/army.log 2>&1; then \ + echo "LIST GATE FAILED: cook-check walked past an element's map slot, which it has no clause for"; exit 1; \ + fi + @grep -q "Squad.roster.*schema#380" build/lists-cooks/army.log || { echo "LIST GATE FAILED: the map-holding cook was refused, but not by name at the map slot"; cat build/lists-cooks/army.log; exit 1; } + @if ./bin/schema cook-check --root Sheet build/lists-cooks/sheet-forged.cook tables/lists > build/lists-cooks/forged.log 2>&1; then \ + echo "LIST GATE FAILED: cook-check accepted a list slot pointing past its holder's extent"; exit 1; \ + fi + @grep -q "leaves\|extent" build/lists-cooks/forged.log || { echo "LIST GATE FAILED: the forgery was refused, but not on the element-array clause"; cat build/lists-cooks/forged.log; exit 1; } + @echo "list gate: cook-check reads two cooks the runtime wrote, refuses the forged list slot, and refuses the map-holding cook by name" + +# THE SIX LoadMeasure REFUSALS are a unit test and not a `report` row (§2.8, +# §2.9, §6.5): each wire is built in memory with a SYNTHETIC count, a list's +# and a map's, and the answer and the REASON are asserted, with a clean wire +# beside them that must measure. +.PHONY: tables-list-measure-refusals +tables-list-measure-refusals: build/schema_test_lists + ./build/schema_test_lists measure-refusals + +# THE LIST-WALK GATE (docs/SPEC-TABLES.md §2.9, §16): the list's half of the +# text form is emitted only in a unit that declares one, it is ONE half, the +# same bytes in every list-bearing .cpp, and none of it reaches a list-free +# unit, which is the zero-cost property (§2.2) holding for the text form. +.PHONY: tables-json-list-walk +tables-json-list-walk: build/tables-generated/.stamp + @rm -rf build/json-list-walk && mkdir -p build/json-list-walk + @for f in build/tables-generated/lists/*Table.cpp; do \ + out=build/json-list-walk/$$(echo $$f | tr / _); \ + awk '/---- json list walk: begin ----/,/---- json list walk: end ----/' $$f > $$out; \ + if [ ! -s $$out ]; then echo "LIST-WALK GATE FAILED: no list half in $$f"; exit 1; fi; \ + done + @first=""; for f in build/json-list-walk/*; do \ + if [ -z "$$first" ]; then first=$$f; else \ + cmp -s $$first $$f || { echo "LIST-WALK GATE FAILED: the list half in $$f is not the list half in $$first"; exit 1; }; \ + fi; \ + done + @for f in build/tables-generated/examples/*Table.cpp build/tables-generated/pointers/*Table.cpp build/tables-generated/maps/*Table.cpp; do \ + if grep -q "json list walk: begin" $$f; then \ + echo "LIST-WALK GATE FAILED: the list half reached the list-free unit $$f"; exit 1; \ + fi; \ + done + @echo "tables list-walk gate: one list half, byte-identical in $$(ls build/json-list-walk | wc -l | tr -d ' ') list-bearing .cpp files, and none in a list-free one" + +# ---- the NEGATIVE CONTROLS §2.9 names ------------------------------------ +# +# The map controls' shape: each names the sabotage, patches the GENERATOR +# through a Go overlay, regenerates the corpus, rebuilds the gate and requires +# it to go RED on a CHECK. A sabotage that patches nothing is itself a failure. +# +# $(1) the control's short name, $(2) the sed script, $(3) the file to patch, +# $(4) the sentence a reader gets when the gate stayed green. +# a comma inside a $(call) argument, spelled so the call does not split on it +comma := , + +define list_negative_control + @mkdir -p build + @sed -e $(2) $(3) > build/list-$(1).gotext + @cmp -s build/list-$(1).gotext $(3) && \ + { echo "NEGATIVE CONTROL FAILED: the $(1) sabotage patched nothing"; exit 1; } || true + @printf '{"Replace":{"%s/$(3)":"%s/build/list-$(1).gotext"}}\n' "$(CURDIR)" "$(CURDIR)" > build/list-$(1)-overlay.json + @go build -overlay=build/list-$(1)-overlay.json -o build/schema-list-$(1) ./cmd/schema + @rm -rf build/tables-list-$(1) && mkdir -p build/tables-list-$(1) + @./build/schema-list-$(1) generate --lang cpp --out build/tables-list-$(1)/lists tables/lists + @$(CXX) $(TABLES_CXXFLAGS) -Ibuild/tables-list-$(1)/lists -Itest/tables test/tables/lists_main.cpp \ + build/tables-list-$(1)/lists/*Table.cpp -o build/schema_test_lists_$(1) + @if ./build/schema_test_lists_$(1) > build/list-$(1).log 2>&1; then \ + echo "NEGATIVE CONTROL FAILED: $(4)"; exit 1; \ + fi + @grep -q "^FAIL" build/list-$(1).log || \ + { echo "NEGATIVE CONTROL FAILED: the gate went red, but not on a CHECK"; cat build/list-$(1).log; exit 1; } + @echo "negative control: $(1) turns the LIST GATE red: $$(grep -c '^FAIL' build/list-$(1).log) failures" +endef + +# THE WRITER EMITS THE ELEMENTS OUT OF ORDER. The cursor is made to step the +# builder's segments from the LAST slot back; `list_scalars` meets it, and the +# byte compare against its pinned wire goes red while measure == save holds. +.PHONY: tables-lists-order-negative-control +tables-lists-order-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,order,'s@return segment->elements + within;@return segment->elements + ( segment->used - 1 - within ); // SABOTAGED@',internal/codegen/cpptable/lists.go,the writer emitting elements out of order left the list gate GREEN) + +# `Save` EMITS A DEAD ELEMENT. `list_erased` meets it, an erase from the +# MIDDLE with an add after it, and the byte compare goes red while +# measure == save still holds, which says the sabotage is the skip. +.PHONY: tables-lists-dead-element-negative-control +tables-lists-dead-element-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,dead,'s@if ( !TableListSegmentDead( segment->dead, within ) ) { break; }@break; // SABOTAGED@',internal/codegen/cpptable/lists.go,a dead element riding on the wire left the list gate GREEN) + +# THE ELEMENT ARRAY IS LAID OUT AFTER A NESTED CONTAINER'S, breaking the +# pre-order rule in BOTH writers of a list whose element holds a map: the +# pack's extent walk stops reserving the element array ahead of the elements' +# maps, and the cook's extent writer stops stepping past it, so the maps are +# laid where the elements are and the node's extent is short of the array. +# `list_of_maps` meets it, and the two instruments §2.9 names go red +# together: the pinned cook's byte compare, and `schema cook-check`'s +# containment clause on the cook the sabotaged gate wrote, which must be that +# clause and not the map slot's refusal by name. +.PHONY: tables-lists-preorder-negative-control +tables-lists-preorder-negative-control: bin/schema build/tables-generated/.stamp + @mkdir -p build + @printf 'func listElementHoldsMap(f *ir.Field) bool {\n\tref := listElementStruct(f)\n\tif ref == nil {\n\t\treturn false\n\t}\n\tfor i := range ref.Fields {\n\t\tif ref.Fields[i].IsMap() {\n\t\t\treturn true\n\t\t}\n\t}\n\treturn false\n}\n' > build/list-preorder-helper.txt + $(call list_negative_control,preorder,'s@g.pf(" at += (int64_t) cursor.count \* %d; // the whole array FIRST\\n", size)@if !listElementHoldsMap(f) { g.pf(" at += (int64_t) cursor.count * %d; // the whole array FIRST\\n", size) } // SABOTAGED@' -e 's@g.pf("%s at += (int64_t) cursor.count \* (int64_t) sizeof( %s ); // the whole array FIRST\\n", ind, elem)@if !listElementHoldsMap(f) { g.pf("%s at += (int64_t) cursor.count * (int64_t) sizeof( %s ); // the whole array FIRST\\n", ind, elem) } // SABOTAGED@' -e '$$r build/list-preorder-helper.txt',internal/codegen/cpptable/extent.go,laying the element array after a nested container left the list gate GREEN) + @grep -c "SABOTAGED" build/list-preorder.gotext | grep -qx 2 || \ + { echo "NEGATIVE CONTROL FAILED: the preorder sabotage did not reach both the pack's and the cook's list branch"; exit 1; } + @grep -q "^FAIL.*list_of_maps_cook" build/list-preorder.log || \ + { echo "NEGATIVE CONTROL FAILED: the pinned list_of_maps_cook byte compare stayed GREEN"; cat build/list-preorder.log; exit 1; } + @rm -rf build/tables-list-preorder/cooks && mkdir -p build/tables-list-preorder/cooks + @SCHEMA_LIST_COOK_DIR=build/tables-list-preorder/cooks ./build/schema_test_lists_preorder > /dev/null 2>&1 || true + @if ./bin/schema cook-check --root Army build/tables-list-preorder/cooks/army.cook tables/lists > build/list-preorder-check.log 2>&1; then \ + echo "NEGATIVE CONTROL FAILED: cook-check accepted the cook the sabotaged writer laid"; exit 1; \ + fi + @grep -q "the array leaves the node\|overlaps another array" build/list-preorder-check.log || \ + { echo "NEGATIVE CONTROL FAILED: cook-check refused the sabotaged cook, but not on the containment clause"; cat build/list-preorder-check.log; exit 1; } + @echo "negative control: preorder turns the pinned list_of_maps_cook compare and cook-check's containment clause red: $$(grep -o 'the array leaves the node\|overlaps another array' build/list-preorder-check.log | head -1)" + +# THE WALK VISITS LISTS OUT OF DECLARATION ORDER, grouped after the pointer +# fields: the edge walk is made to take every list field last, so +# `list_before_pointer`'s `cover` reaches the shared node before the list +# does, and the pinned wire goes red on the node numbering. +.PHONY: tables-lists-walk-order-negative-control +tables-lists-walk-order-negative-control: bin/schema build/tables-generated/.stamp + @mkdir -p build + @printf 'func listsLast(fields []*ir.Field) []*ir.Field {\n\tvar first, last []*ir.Field\n\tfor _, f := range fields {\n\t\tif f.IsList() {\n\t\t\tlast = append(last, f)\n\t\t} else {\n\t\t\tfirst = append(first, f)\n\t\t}\n\t}\n\treturn append(first, last...)\n}\n' > build/list-walkorder-helper.txt + $(call list_negative_control,walkorder,'/guards := guardWalk(st$(comma) v.read+".")/$(comma)/^}/ s@for _$(comma) f := range st.Fields {@for _$(comma) f := range listsLast(st.Fields) { // SABOTAGED@' -e '$$r build/list-walkorder-helper.txt',internal/codegen/cpptable/pointers.go,grouping the lists after the pointer fields left the list gate GREEN) + +# A SHARED NODE IS WRITTEN TWICE: `list_shared`, whose two slots name one +# node, meets it, and the region's byte count and the text round trip's &node +# resolution go red. The sabotage reaches a fresh map entry per visit. +.PHONY: tables-lists-shared-negative-control +tables-lists-shared-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,shared,'s@const TablePackEntry \* entry = TablePackMapReach( seen, (const void \*) pointee, 0, taken, slot );@const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, 0, taken, slot ); taken = true; // SABOTAGED@',internal/codegen/cpptable/pointers.go,writing a shared node twice left the list gate GREEN) + +# THE READER CLAMPS THE COUNT against something: the 100,000-element row +# meets it, and the decoded count goes red. The sabotage clamps at 2^16. +.PHONY: tables-lists-clamp-negative-control +tables-lists-clamp-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,clamp,'s@ fill.capacity = (int32_t) n;@ fill.capacity = n > 65536 ? 65536 : (int32_t) n; // SABOTAGED@',internal/codegen/cpptable/lists.go,clamping the count left the list gate GREEN) + +# THE ELEMENT-KIND RULE DECODES ANYWAY: the `Ints`-as-`Floats` row meets it, +# and the decoded values go red. +.PHONY: tables-lists-element-kind-negative-control +tables-lists-element-kind-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,elemkind,'s@else if ( elem_kind != %d ) { r.report->kind_mismatch++; r.offset = body_end; break; }@else if ( elem_kind != %d \&\& false ) { r.report->kind_mismatch++; r.offset = body_end; break; }@',internal/codegen/cpptable/lists.go,decoding under a changed element kind left the list gate GREEN) + +# LoadMeasure OVER A LIST OF TABLES HOLDING LISTS, summed at ONE DEPTH only: +# `list_nested` meets it, and the measure goes red against the region Load +# fills. +.PHONY: tables-lists-depth-negative-control +tables-lists-depth-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,depth,'s@if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term@return true; // SABOTAGED@',internal/codegen/cpptable/lists.go,summing the extent at one depth only left the list gate GREEN) + +# THE FIT CHECK IS DROPPED: an N the list's L cannot carry measures, and the +# refusals battery goes red on the reason. +.PHONY: tables-lists-fit-negative-control +tables-lists-fit-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,fit,'s@if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; }@if ( n > (uint64_t) ( rest / elem_floor ) \&\& false ) { reason = count_over_length; return false; }@',internal/codegen/cpptable/lists.go,an N the list L cannot carry left the list gate GREEN) + +# AN UNREACHED NON-EMPTY LIST SLOT IS REFUSED by Cook and by Lock: the `Deck` +# instance whose counted array holds a list PAST ITS LIVE COUNT meets it. +.PHONY: tables-lists-unreached-negative-control +tables-lists-unreached-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,unreached,'s@inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; }@inline bool TableExtentUnreachedEmpty( int64_t ) { return true; }@',internal/codegen/cpptable/extent.go,writing an unreached non-empty list left the list gate GREEN) + +# `schema cook-check`'S ELEMENT-ARRAY CLAUSE IS DROPPED (§7.4): the forged +# cook whose list slot points past its holder's extent passes the tool, and the +# Go test that holds the clause goes red. +.PHONY: tables-lists-cook-check-negative-control +tables-lists-cook-check-negative-control: + @rm -rf build/list-cook-check-control && mkdir -p build/list-cook-check-control + @sed -e 's@if start < s.base || end > s.extent {@if ( start < s.base || end > s.extent ) \&\& false { // SABOTAGED@' \ + internal/tablecook/check.go > build/list-cook-check-control/check.go.txt + + @cmp -s internal/tablecook/check.go build/list-cook-check-control/check.go.txt && \ + { echo "NEGATIVE CONTROL FAILED: the cook-check list sabotage patched nothing"; exit 1; } || true + @printf '{"Replace":{"%s/internal/tablecook/check.go":"%s/build/list-cook-check-control/check.go.txt"}}\n' \ + "$(CURDIR)" "$(CURDIR)" > build/list-cook-check-control/overlay.json + @if go test -count=1 -overlay=build/list-cook-check-control/overlay.json \ + -run 'TestCookCheckListSlot' ./internal/tablecook/ > build/list-cook-check-control/log 2>&1; then \ + echo "NEGATIVE CONTROL FAILED: dropping cook-check's element-array clause left its test GREEN"; exit 1; \ + fi + @echo "negative control: dropping cook-check's element-array clause turns its test RED" + +# AN ALLOCATION IS PLANTED IN Load: the gate's operator new counter sees it on +# the reading path, and the allocation audit goes red. +.PHONY: tables-lists-allocation-negative-control +tables-lists-allocation-negative-control: bin/schema build/tables-generated/.stamp + $(call list_negative_control,allocation,'s@ Element \* element = fill.array + fill.list->count;@ Element * element = fill.array + fill.list->count; delete new int; // SABOTAGED@',internal/codegen/cpptable/lists.go,an allocation planted in Load left the allocation audit GREEN) + +.PHONY: tables-lists-negative-controls +tables-lists-negative-controls: tables-lists-allocation-negative-control \ + tables-lists-order-negative-control \ + tables-lists-dead-element-negative-control \ + tables-lists-preorder-negative-control \ + tables-lists-walk-order-negative-control \ + tables-lists-shared-negative-control \ + tables-lists-clamp-negative-control \ + tables-lists-element-kind-negative-control \ + tables-lists-depth-negative-control \ + tables-lists-fit-negative-control \ + tables-lists-unreached-negative-control \ + tables-lists-cook-check-negative-control + # Re-pin the goldens DELIBERATELY (SPEC §7.2 gates 1, 2, 7). A wire golden # breaking under an unchanged schema is stop-the-line, never a quiet re-pin # (SPEC §3.1) — this target is for intentional emitter/schema changes only. -update-goldens: build/schema_test build/schema_test_ludicrous build/schema_test_bench build/schema_test_bench_table build/schema_test_tables build/schema_test_block build/schema_test_maps +update-goldens: build/schema_test build/schema_test_ludicrous build/schema_test_bench build/schema_test_bench_table build/schema_test_tables build/schema_test_block build/schema_test_maps build/schema_test_lists @mkdir -p testdata/golden testdata/wire testdata/wire/tables go test ./internal/goldens -update -run 'TestGolden' SCHEMA_UPDATE_WIRE_GOLDENS=1 ./build/schema_test SCHEMA_UPDATE_WIRE_GOLDENS=1 ./build/schema_test_tables SCHEMA_UPDATE_WIRE_GOLDENS=1 ./build/schema_test_block SCHEMA_UPDATE_WIRE_GOLDENS=1 ./build/schema_test_maps - @for d in examples pointers block blockhome messages stream blobs scalars maps; do \ + SCHEMA_UPDATE_WIRE_GOLDENS=1 ./build/schema_test_lists + @for d in examples pointers block blockhome messages stream blobs scalars maps lists; do \ mkdir -p testdata/golden/tables/$$d; \ cp build/tables-generated/$$d/*Table.h build/tables-generated/$$d/*Table.cpp testdata/golden/tables/$$d/ 2>/dev/null || true; \ done diff --git a/compiler/cook.go b/compiler/cook.go index 107e86ed1..67012431c 100644 --- a/compiler/cook.go +++ b/compiler/cook.go @@ -117,12 +117,6 @@ func orderWord(big bool) string { // It is a person's decision to run it, not a parameter on a load: the runtime // keeps one `Open` that matches the header and points. func (c *Compiler) CookCheck(u *ir.Unit, root string, file []byte) (CookReport, error) { - if err := refuseToolMaps(u); err != nil { - return CookReport{}, err - } - if err := refuseToolLists(u); err != nil { - return CookReport{}, err - } m := tabletext.NewModel(u) res, err := tablecook.Check(m, file) if err != nil { diff --git a/compiler/tableslists.go b/compiler/tableslists.go index fc9b0eacb..41604e4d2 100644 --- a/compiler/tableslists.go +++ b/compiler/tableslists.go @@ -1,8 +1,8 @@ // The UNBOUNDED ARRAY cross-target refusal (docs/SPEC-TABLES.md §2.9, §11): -// its own file, per the registry split — a construct's refusal adds a file -// beside builtin.go rather than growing it. Every target's Generate calls -// [refuseLists], because no code generator carries the construct yet; the -// carrier registry the map's file has lands here with the first carrier. +// its own file, per the registry split: a construct's carrier registry and +// its refusal add a file beside builtin.go rather than growing it. A target +// that carries the construct registers through [registerListCarrier] from its +// own file's init, and every other target's Generate calls [refuseLists]. package compiler import ( @@ -11,29 +11,40 @@ import ( "github.com/mas-bandwidth/schema/v2/ir" ) -// refuseLists is the named refusal every target gives a unit whose table +// listTargets is the canonical name of every built-in target whose table +// backend carries an UNBOUNDED ARRAY (docs/SPEC-TABLES.md §2.9). refuseLists +// names them. +var listTargets []string + +// registerListCarrier is what a carrying target's file calls from its init, +// beside its registerBuiltin call. +func registerListCarrier(name string) { listTargets = append(listTargets, name) } + +// refuseLists is the named refusal every PORT gives a unit whose table // closure declares a `[]T` (docs/SPEC-TABLES.md §2.9, §11, §15). // // An unbounded array is a VARIABLE-CLASS construct, and the variable class is // the C++ reference's alone — the arena, the region, the node extent and the -// walks a list's elements ride in are all the reference's. The LANGUAGE takes -// the spelling and the tool's WIRE and TEXT halves carry it, so `pack` and -// `unpack` read and write one, and a generator that emitted a codec for it -// would emit one that never met an element array. So every target refuses by -// name until the reference lands the codec. +// walks a list's elements ride in are all the reference's. So the reference +// carries the codec, registers through [registerListCarrier] from its own +// init and never reaches here, and every port refuses loudly rather than +// emitting a codec that never met an element array. func refuseLists(u *ir.Unit, target string) error { fields := ir.ListFields(u) if len(fields) == 0 { return nil } - return fmt.Errorf("unit declares an unbounded array in a table closure (%s) — no code generator carries `[]T` yet, %s included: the language takes the spelling and the tool's `pack` and `unpack` read and write it, and the C++ reference lands the codec first (docs/SPEC-TABLES.md §2.9, §11, §15). Declare the array at a bound, [..N]T, which is the same bytes, and remove the bound when the reference carries it", - englishList(fields), target) + carry, flags := carriers(listTargets) + return fmt.Errorf("unit declares an unbounded array in a table closure (%s): a []T is %s only today, and the %s form is a named follow-on. Generate with %s, or declare the array at a bound, [..N]T, which is the same bytes (docs/SPEC-TABLES.md §2.9, §11, §15)", + englishList(fields), englishList(carry), target, englishList(flags)) } // refuseToolLists is the TOOL's COOK refusal (docs/SPEC-TABLES.md §2.9, §15), // the map's own, one construct over: a unit whose table closure declares a -// `[]T` is refused by name at the tool's COOK surfaces, because -// internal/tablecook does not lay out the element arrays yet. +// `[]T` is refused by name at the tool's COOK and UNCOOK surfaces, because +// internal/tablecook does not lay out the element arrays yet. `cook-check` +// is not among them: its scan carries §7.4's element-array clause, and a cook +// the C++ reference wrote is checked there. // // It is here, at the surface, rather than in the engine, and it is NAMED // rather than left to the layout. Without it the engine lays out a region @@ -44,6 +55,6 @@ func refuseToolLists(u *ir.Unit) error { if len(fields) == 0 { return nil } - return fmt.Errorf("unit declares an unbounded array in a table closure (%s) — the tool's WIRE and TEXT halves carry the construct, and its COOK half does not, so this command would lay out a region short of the element arrays rather than refusing; the C++ reference lands the cook (--lang cpp) (docs/SPEC-TABLES.md §2.9, §15)", + return fmt.Errorf("unit declares an unbounded array in a table closure (%s): the tool's WIRE and TEXT halves carry the construct and `cook-check` reads one, and its COOK half does not, so this command would lay out a region short of the element arrays rather than refusing. The C++ reference carries the cook (--lang cpp) (docs/SPEC-TABLES.md §2.9, §15)", englishList(fields)) } diff --git a/compiler/tableslists_test.go b/compiler/tableslists_test.go index 9abd614cb..aaf8fb861 100644 --- a/compiler/tableslists_test.go +++ b/compiler/tableslists_test.go @@ -1,9 +1,8 @@ package compiler // The UNBOUNDED ARRAY's cross-target refusals (docs/SPEC-TABLES.md §2.9, §11, -// §15): no code generator carries the construct yet, so every one of them -// refuses a unit that declares one BY NAME, and none of them refuses a -// list-free unit for it. +// §15): the C++ reference carries the codec, every port refuses a unit that +// declares one BY NAME, and none of them refuses a list-free unit for it. import ( "strings" @@ -26,22 +25,38 @@ table Save } ` -// TestEveryTargetRefusesAList: the refusal is a REFUSAL and not a silent -// emission of an array whose elements no backend laid out. -func TestEveryTargetRefusesAList(t *testing.T) { +// TestListsAreRefusedByEveryPort: the refusal is a REFUSAL and not a silent +// emission of an array whose elements no port laid out, and the reference +// does not refuse. +func TestListsAreRefusedByEveryPort(t *testing.T) { u := unitFromSource(t, listSrc) c := New() for _, target := range c.Targets() { - _, err := c.Generate(u, target, Options{}) - if err == nil { - t.Errorf("--lang %s emitted for a unit declaring an unbounded array", target) - continue - } - for _, want := range []string{"unbounded array", "Save.placements", "Save.scores", "[..N]T"} { - if !strings.Contains(err.Error(), want) { - t.Errorf("--lang %s: the refusal does not name %q: %v", target, want, err) + t.Run(target, func(t *testing.T) { + _, err := c.Generate(u, target, Options{}) + if target == "cpp" { + if err != nil { + t.Fatalf("--lang cpp refused an unbounded array: the reference carries the codec (schema#531): %v", err) + } + return } - } + if err == nil { + t.Fatalf("--lang %s emitted for a unit declaring an unbounded array", target) + } + for _, want := range []string{"unbounded array", "Save.placements", "Save.scores", "[..N]T", "cpp"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("--lang %s: the refusal does not name %q: %v", target, want, err) + } + } + }) + } +} + +// TestListCarrierIsTheReferenceAlone: exactly one target carries the +// construct, and it is the C++ reference (docs/SPEC-TABLES.md §2.9, §15). +func TestListCarrierIsTheReferenceAlone(t *testing.T) { + if len(listTargets) != 1 || listTargets[0] != "cpp" { + t.Fatalf("listTargets = %v, want exactly [cpp]: the variable class is the reference's (docs/SPEC-TABLES.md §2.9, §15)", listTargets) } } @@ -75,11 +90,27 @@ func TestListFieldsNamesWhatAnAuthorWrote(t *testing.T) { } } +// TestListRefusalNamesTheCarrier: what a port's refusal says: the carrier, +// the flag that generates, and the fields an author wrote. +func TestListRefusalNamesTheCarrier(t *testing.T) { + err := refuseLists(unitFromSource(t, listSrc), "go") + if err == nil { + t.Fatalf("refuseLists accepted a list-bearing unit for a non-carrier") + } + for _, want := range []string{"a []T is cpp only today", "Save.placements", "--lang cpp"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("the carrier-form refusal does not name %q: %v", want, err) + } + } +} + // TestTheToolsCookRefusesAList: the tool's WIRE and TEXT halves carry the -// construct and its COOK half does not, so the cook surfaces refuse by name -// rather than laying out a region short of the element arrays. +// construct and `cook-check` reads one, and its COOK half does not, so the +// cook and uncook surfaces refuse by name rather than laying out a region +// short of the element arrays. func TestTheToolsCookRefusesAList(t *testing.T) { - err := refuseToolLists(unitFromSource(t, listSrc)) + u := unitFromSource(t, listSrc) + err := refuseToolLists(u) if err == nil { t.Fatal("the tool's cook accepted a unit declaring an unbounded array") } @@ -91,4 +122,16 @@ func TestTheToolsCookRefusesAList(t *testing.T) { if err := refuseToolLists(unitFromSource(t, mapSrc)); err != nil { t.Fatalf("the tool refused a list-free unit: %v", err) } + c := New() + if _, _, _, err := c.Cook(u, "Save", nil, CookOptions{}); err == nil || !strings.Contains(err.Error(), "unbounded array") { + t.Errorf("the tool's cook did not refuse a list-bearing unit by name: %v", err) + } + if _, err := c.Uncook(u, "Save", nil); err == nil || !strings.Contains(err.Error(), "unbounded array") { + t.Errorf("the tool's uncook did not refuse a list-bearing unit by name: %v", err) + } + // cook-check reaches its scan: the refusal it answers for an empty file is + // the header's, not the construct's + if _, err := c.CookCheck(u, "Save", nil); err == nil || strings.Contains(err.Error(), "unbounded array") { + t.Errorf("cook-check refused a list-bearing unit by construct rather than reading the file: %v", err) + } } diff --git a/compiler/tablesmaps.go b/compiler/tablesmaps.go index 9b94e30a4..7a13a63f4 100644 --- a/compiler/tablesmaps.go +++ b/compiler/tablesmaps.go @@ -39,16 +39,17 @@ func refuseMaps(u *ir.Unit, target string) error { } // refuseToolMaps is the TOOL's COOK refusal (docs/SPEC-TABLES.md §2.8, §15): a unit -// whose table closure declares a map is refused by name at every table surface -// the tool has — pack, unpack, cook, cook-check and uncook — because -// internal/tablewire does not carry the construct yet. +// whose table closure declares a map is refused by name at the tool's COOK and +// UNCOOK surfaces, because internal/tablecook does not lay out the entry +// arrays yet. `cook-check` is not among them: its scan refuses a map SLOT by +// name where it meets one (internal/tablecook), so a cook of a map-free root +// in a unit that declares a map elsewhere is checked as any other is. // // It is here, at the surface, rather than in the engine, and it is NAMED rather -// than left to the decoder. Without it the engine meets a kind 14 field whose -// element kind is 13, decodes nothing into a slot it has no shape for, and -// reports FRAMING DAMAGE — an answer that sends its reader looking for a -// corrupt file when the file is fine and the reader is the one that is short. -// A refusal that says which is which is the whole difference. +// than left to the layout. Without it the engine lays out a region short of +// the entry arrays and a reader meets a slot pointing past its holder's +// extent, which is a corrupt file with nothing saying who wrote it. A refusal +// that says which is which is the whole difference. func refuseToolMaps(u *ir.Unit) error { fields := ir.MapFields(u) if len(fields) == 0 { diff --git a/compiler/tablesmaps_test.go b/compiler/tablesmaps_test.go index 67b620858..f56be1e5b 100644 --- a/compiler/tablesmaps_test.go +++ b/compiler/tablesmaps_test.go @@ -271,17 +271,18 @@ func TestMapEntryIsNotARoot(t *testing.T) { } // TestToolRefusesMapsByName: the tool's WIRE and TEXT halves carry maps now -// (docs/SPEC-TABLES.md §2.8), and its COOK half does not — so the cook -// surfaces refuse a map-bearing unit BY NAME rather than laying out an entry -// array they have no placement for. Without the refusal a caller gets a cook -// whose region is short of the entries, which is worse than a diagnostic. +// (docs/SPEC-TABLES.md §2.8), and its COOK half does not. So the cook and +// uncook surfaces refuse a map-bearing unit BY NAME rather than laying out an +// entry array they have no placement for. Without the refusal a caller gets a +// cook whose region is short of the entries, which is worse than a diagnostic. +// `cook-check` refuses at the SLOT instead, where its scan meets one, and +// internal/tablecook's TestCookCheckMapSlotRefusedByName holds that. func TestToolRefusesMapsByName(t *testing.T) { u := unitFromSource(t, mapSrc) c := New() surfaces := map[string]func() error{ - "Cook": func() error { _, _, _, err := c.Cook(u, "Fleet", nil, CookOptions{}); return err }, - "Uncook": func() error { _, err := c.Uncook(u, "Fleet", nil); return err }, - "CookCheck": func() error { _, err := c.CookCheck(u, "Fleet", nil); return err }, + "Cook": func() error { _, _, _, err := c.Cook(u, "Fleet", nil, CookOptions{}); return err }, + "Uncook": func() error { _, err := c.Uncook(u, "Fleet", nil); return err }, } for name, call := range surfaces { t.Run(name, func(t *testing.T) { diff --git a/compiler/target_cpp.go b/compiler/target_cpp.go index 6023346ab..392d2e53f 100644 --- a/compiler/target_cpp.go +++ b/compiler/target_cpp.go @@ -22,13 +22,6 @@ func (cppTarget) Generate(u *ir.Unit, _ Options) (map[string][]byte, error) { if err := refusePacketVoidArms(u, "cpp"); err != nil { return nil, err } - // the UNBOUNDED ARRAY is OWED in this target (docs/SPEC-TABLES.md §2.9): - // the reference lands the codec first and registers through - // registers a carrier when it does. Until then it refuses one by name - // rather than emitting an array whose elements it never laid out. - if err := refuseLists(u, "cpp"); err != nil { - return nil, err - } files, err := cpp.Generate(u) if err != nil { return nil, err @@ -53,5 +46,6 @@ func init() { registerBuiltin(cppTarget{}, true, true, true, true) registerWideTextCarrier("cpp") // the C++ reference carries wstring(N) on the packet wire (SPEC §4.12) registerOptionalArrayCarrier("cpp") - registerMapCarrier("cpp") // the C++ reference carries the map codecs (docs/SPEC-TABLES.md §2.8) + registerMapCarrier("cpp") // the C++ reference carries the map codecs (docs/SPEC-TABLES.md §2.8) + registerListCarrier("cpp") // and the unbounded array codec (docs/SPEC-TABLES.md §2.9) } diff --git a/docs/COMPARISON-TABLES.md b/docs/COMPARISON-TABLES.md index cdc500742..bcab29d17 100644 --- a/docs/COMPARISON-TABLES.md +++ b/docs/COMPARISON-TABLES.md @@ -214,7 +214,7 @@ the source list at the end. | Optional fields | `?T` on a nested table, a type, an enum, a flags mask, a scalar and a bounded array: the value plus a `_present` bool, fixed size, no allocation (§2.3); on a string, on `bytes` and on a value whose closure is variable, a named follow-on (§15) | optional scalars via `= null`; references null when absent (FB-schema) | explicit presence: proto2 all, proto3 `optional`, editions explicit by default (PB-presence) | | Defaults | `= v` on scalars; part of the wire contract because a default is elided; changing one is a silent edit the baseline refuses (§4, §4.1, §18) | scalar defaults; "don't change existing default values" (FB-evolution) | proto2 `[default]`; proto3 zero, not serialized; editions serialize set defaults (PB-presence) | | Union | `union` with implicit `None`; arms by name hash, so add, remove, reorder freely; an arm IS a field line, so its type is any type a field's is — a `table` inside a table closure included — and an arm may carry no payload at all (§2.6, §5) | `union` of tables; structs and strings experimental; vectors of unions C++ only (FB-schema) | `oneof`; `Any` for open typing (PB-proto3) | -| Arrays | `[N]T`, `[..N]T`, `[A..B]T` on both wires, and `[]T` unbounded in a table body, whose count the data decides (§2.9, taken by the front end and by the tool, emitted by no backend yet); the bound is not wire identity, so `[]T` and `[..N]T` are the same bytes (§2, §4) | `[T]` unbounded; `[T:N]` in structs only (FB-schema) | `repeated`, unbounded, packed scalars (PB-proto3) | +| Arrays | `[N]T`, `[..N]T`, `[A..B]T` on both wires, and `[]T` unbounded in a table body, whose count the data decides (§2.9, carried by the C++ reference and the tool, refused by every port by name); the bound is not wire identity, so `[]T` and `[..N]T` are the same bytes (§2, §4) | `[T]` unbounded; `[T:N]` in structs only (FB-schema) | `repeated`, unbounded, packed scalars (PB-proto3) | | Enum-keyed arrays | `[E]T`: one slot per variant, no `None` slot, bad keys refused in every build, slots ride by name (§2.4, §3.2) | — | — | | Maps | `map[K]V` in a table body, a lookup over entries the wire carries as a sorted array of one generated `{ key, value }` table; keys are `string(N)` and the integer kinds, every other key refused by name; makes the holder variable (§2.8), in the reference and the tool | sorted vector of tables with `key` plus `LookupByKey` (FB-cpp) | `map`, integral or string keys, order undefined (PB-proto3) | | Sorted lookup in a buffer | the map's `Find`: a binary search in place over the sorted entry array, the same call in a locked region, a loaded one and an opened cook, plus an optional index the caller builds at load and never stores (§2.8) | `key`, `CreateVectorOfSortedTables`, `LookupByKey` (FB-cpp) | — | @@ -268,7 +268,7 @@ the source list at the end. | Union evolution | arms by name; add anywhere, remove, reorder (§2.6, §5) | append or explicit discriminant (FB-evolution) | adding is fine; moving an existing field into a oneof is unsafe (PB-editions) | | Flags evolution | append only, retire in place; the baseline refuses the rest (§4.1) | explicit values, any order (FB-schema) | — | | Array bound change | prefix kept, `clamped` counted; a short array fills with defaults (§4) | unbounded | unbounded | -| Unbounded arrays | `[]T` and `[]*T` in a TABLE body, whose count the data decides (§2.9, taken by the front end and by the tool, emitted by no backend yet); the same bytes as `[..N]T`, so the bound is a declaration-side fact and moving between them is silent or a clamp. Refused in a `type` body by name, which is what keeps the packet wire bounded | unbounded vectors | unbounded repeated fields | +| Unbounded arrays | `[]T` and `[]*T` in a TABLE body, whose count the data decides (§2.9, carried by the C++ reference and the tool, refused by every port by name); the same bytes as `[..N]T`, so the bound is a declaration-side fact and moving between them is silent or a clamp. Refused in a `type` body by name, which is what keeps the packet wire bounded | unbounded vectors | unbounded repeated fields | | `T` to `?T` to `*T` | `T` and `?T` are byte-identical for non-default content; to or from `*T` is a counted mismatch (§2.3, §4) | changes default semantics; required/optional changes break (FB-schema) | presence changes round-trip of defaults (PB-presence) | | Unknown fields on read | skipped by length, counted (§3, §4) | ignored (FB-evolution) | retained in the unknown set (PB-proto3) | | Unknown fields on rewrite | dropped and counted, by design; the writer has a schema | the buffer keeps them if forwarded whole (FB-evolution) | preserved and re-serialized (PB-proto3) | diff --git a/docs/SPEC-TABLES.md b/docs/SPEC-TABLES.md index 18dad63ca..ecbecbd9b 100644 --- a/docs/SPEC-TABLES.md +++ b/docs/SPEC-TABLES.md @@ -2268,12 +2268,15 @@ table that holds `[]*Self` is the ordinary legal recursion through a pointer. A hold one node, one index on the wire (§3.1), one body in a region (§6.3), one `&node` in the text (§16.7). -**THE COUNT IS THE DATA'S, and what bounds it is stated.** There is no `| max` -on a `[]T` and there is no `?[]T`, for the reasons §2.8 gives a map: a bound -would buy only a CLAMP, which drops a tail, and a fresh list is empty and an -empty list is elided under §3's by-value elision rule, the rule that elides an -empty counted array. What bounds it is three things and each belongs to one -side: +**THE COUNT IS THE DATA'S, and what bounds it is stated.** There is no +attribute that bounds a `[]T`'s COUNT and there is no `?[]T`, for the reasons +§2.8 gives a map: a bound would buy only a CLAMP, which drops a tail, and a +fresh list is empty and an empty list is elided under §3's by-value elision +rule, the rule that elides an empty counted array. **A bar attribute on a +`[]T` qualifies the ELEMENT, exactly as it does on a `[..N]T`**: `scores +[]int32 | min = 0, max = 100` bounds each score, `was` renames the field and +`json` keys its text, and none of them is a count. What bounds the count is +three things and each belongs to one side: - **On the AUTHORING side, the ARENA.** Elements are carved from the builder's arena in bulk segments (below), so an `Add` that cannot carve one answers @@ -2754,20 +2757,25 @@ each: | byte-stable output | `measure == save`, index order, one image from one value (§9) | field order is the writer's | the builder's order | | a fixed-table user pays nothing | a list-free unit carries no list machinery, held by the zero-cost gate's header scan (§2.2) | every runtime carries the repeated codec | every runtime carries the vector | -**BACKEND STATUS: OWED, not emitted.** This section is specified ahead of its -implementation, on the same terms §3.3 and §6.6 take: **the FRONT END takes the -spelling and holds every refusal above, and the TOOL's WIRE and TEXT halves -carry the construct**, so `pack` and `unpack` read and write a `[]T` and the -projections render it. **No CODE GENERATOR carries it**, every one of them -refuses a unit that declares one by name (§11), and the corpus holds no -`tables/lists`. The C++ REFERENCE lands the codec next and every other backend -keeps refusing, with the ports a named follow-on (§15). The corpus the implementation owes is `tables/lists`: -`list_empty` (an empty list beside a full one), `list_scalars`, `list_tables`, -`list_shared` (two slots naming one node beside a null slot), -`list_before_pointer` (the walk-order control above), `list_erased` (an erase -from the middle with an add after it), `list_of_maps` and `list_nested` (a list -of tables that hold lists), each crossing the wire, the text and the cook in -the harness, with the report rows the negative controls above name and +**WHERE IT IS CARRIED.** The FRONT END takes the spelling and holds every +refusal above. The C++ REFERENCE carries the codec: the `TableList` runtime, +the builder's three, the five element classes on the wire, the node extent, the +cook's write side, the text form and the descriptors, held by the corpus below. +The TOOL's WIRE and TEXT halves carry the construct, so `pack` and `unpack` +read and write a `[]T` and the projections render it, and `schema cook-check` +reads one (§7.4); the tool's COOK and UNCOOK halves do not lay out the element +arrays yet and refuse a unit that declares one by name, beside the map's own +refusal there (§15). **Every PORT refuses a unit that declares one by name** +(§11), naming the reference as the carrier, with the ports a named follow-on +(§15). The corpus is `tables/lists`: `list_empty` (an empty list beside a full +one), `list_scalars`, `list_tables`, `list_mixed` (an enum, a flags mask, a +union and a bounded scalar as elements), `list_shared` (two slots naming one +node beside a null slot), `list_before_pointer` (the walk-order control above), +`list_erased` (an erase from the middle with an add after it), `list_of_maps` +and `list_nested` (a list of tables that hold lists, and a pointed-at holder +with a list of its own), each crossing the wire, the text and the cook in +`test/tables/lists_main.cpp`, with the report rows the negative controls above +name, the two cooks pinned beside the wires, and `make tables-list-measure-refusals` beside them. **AND ONE GOLDEN IS THE MIGRATION ITSELF**, `list_migrates`, because "the same @@ -6241,13 +6249,15 @@ The builder is designed to go wide, lock-free by ownership: with the accelerators' refusal and lands with it**, so a build that has one has the other. - **BACKEND STATUS: OWED, not emitted.** The enum is specified ahead of its - implementation, on the terms §3.3 and §6.6 take. `TableRefuseReason` is - spelled in no target, in no runtime and in no tool, a bare `-1` is the whole - of a measure's answer today, and the name is not claimed either, so a unit - declaring a table or a type called `TableRefuseReason` compiles. - Owed as schema#523, with §7's check order and §19.2's block clauses, and - this line is deleted by the implementation PR that lands the behavior. + **WHERE IT IS CARRIED.** The C++ reference spells `TableRefuseReason` with + the two values a map's and an unbounded array's framing can raise, + `count_over_length` and `count_over_extent_cap`, as a native enum a unit + that declares either construct emits, and `LoadMeasure` there takes it as a + trailing out-parameter, `TableRefuseReason * reason_out = NULL`, so a caller + that does not ask keeps the signature it had. A unit with neither construct + carries neither the enum nor the parameter (§2.2). The other three values, + §7's check order and §19.2's block clauses are owed as schema#523, and no + port spells the enum yet. - **Into a builder** — the tool's path. The same tolerant decode into a fresh builder, so loaded data can be edited and locked again. **Its own refusal is a NULL** rather than a `-1`, and the report it leaves behind is @@ -7535,7 +7545,10 @@ no reference: the check exactly as the pack order does. The entries' own slots, companions and tags are then walked as a bounded array's elements are. The KEYS are read too, ascending with no repeat, because a cook `Find` cannot - search is a forgery. + search is a forgery. Until schema#380 lands this clause in the tool, `schema + cook-check` refuses a map slot by name where its scan meets one, so a + cook that holds one is refused rather than walked past, and the C++ + reference reads it. 5. **Every UNBOUNDED-ARRAY SLOT** (§2.9). The same four clauses as a map's: CONTAINMENT, ALIGNMENT, FIT and NO OVERLAP, against the holder's own extent and against every other element or entry array in that node, and then the @@ -8172,8 +8185,12 @@ leaves all three NULL. INLINE, and `array_bound = 0` is what says so.** Neither had a written rule before this section, so the rule is here, one shape for both: -- **`kind` is `14`** and the ELEMENT kind is `13` for a map, the element's own - kind for a list, exactly as the wire carries them (§2.8, §2.9). +- **`kind` is the ELEMENT kind, as on every array line**: `13` for a map, the + generated entry being a table; the element's own kind for a list, and `17` + for a `[]*T`, whose elements are node indices. The wire's kind `14` is what + `is_array` says, exactly as it says it for a `[..N]T`, and the `type_name` + says which of the two constructs it is, `map[...]` or the element's own, + the way `bytes` separates a byte buffer from an array of `u8` (below). - **`element_size` is the pitch**: `sizeof( Entry )` for a map, `sizeof( T )` for a list. It is the stride a walker steps, as on every array line. - **`counted` is set and `count_offset` names the `int32` count**, which sits @@ -8194,21 +8211,17 @@ before this section, so the rule is here, one shape for both: map's key is a field of the entry and a walker meets it there, and a list has no key at all. -**BACKEND STATUS, because the reference does not carry this yet.** The C++ -map descriptor emitted today leaves `kind` at `0` and `is_array` false and -describes the map through the ENTRY's own `TableTypeInfo` beside three -map-specific FUNCTION columns, `map_count`, `map_at` and `map_insert`, which -is what let the text walk reach a `TableMap` it has no name for. **It -moves to the columns above when the list lands**, and the two land together -for one reason: a second out-of-line shape would otherwise need a second set -of function columns, and three per construct is how a descriptor becomes a -per-construct API instead of a vocabulary. **The function columns do not all -go**: what the walk cannot spell for itself it still cannot spell, so a -resolver stays where a resolver is needed, and what changes is that the SHAPE -is read from `kind`, `is_array`, `counted`, `element_size` and -`array_bound = 0` like every other array's rather than inferred from a -non-NULL `entry`. A port that has neither construct carries neither column -set, which is §2.2's gate doing its job. +**ONE FUNCTION COLUMN SERVES BOTH, and it is `place`.** What the ONE text walk +cannot spell for itself it still cannot spell: placing an entry by key in a +`TableMap` or appending an element to a `TableList` needs the type +the walk has no name for, so a resolver stays where a resolver is needed, and +it is one column, `place( worker, slot, key, key_length, key_value )`, which a +map reads as an insert by key and a list reads as an append. The SHAPE is read +from `kind`, `is_array`, `counted`, `element_size` and `array_bound = 0` like +every other array's, the count from `count_offset`, and the array from the +reference at `offset`, so nothing about the two constructs is inferred from a +column of their own. A unit that has neither construct carries no `place` +column, which is §2.2's gate doing its job. **The public currency is the KEY; the storage index is private** (§2.4). `array_bound` on a keyed field is the STORAGE EXTENT, `E.Max` — derived @@ -9049,8 +9062,10 @@ in build version (§20.5). diagnostic naming the table wrapper that serves, which is the refusal a `map` takes there on the same ground; the near-miss spellings `[..]T` and `[0..]T`, each naming `[]T` as the fix, because a count bound is - a range literal and never a truncated one (SPEC.md §4.2); `?[]T`, a specified - default on one, and `| max` on one; the bounded spellings of the construct + a range literal and never a truncated one (SPEC.md §4.2); `?[]T` and a + specified default on one, while the bar attributes qualify the ELEMENT + exactly as they do on a `[..N]T` and no attribute names a count bound + (§2.9); the bounded spellings of the construct itself, `[..N][]T` and `[N][]T`, which are arrays of arrays and refused as those are (SPEC.md §4.3); a table that holds a `[]` of ITSELF by value, directly or through any chain (the @@ -10547,7 +10562,9 @@ inspects everything in the schema built: C++ reference and the tool are first: the builder surface (insert, erase, find, iterate), the sort in the four walks, the region load's ascending check with its `duplicate` and `malformed` events, the const `Find`, the text - form's object and `schema cook-check`'s order check. What a port needs is the entry as an ordinary array-of-tables element in its + form's object and `schema cook-check`'s map-slot clause with its order + check, which is the one piece still owed: the tool refuses a map slot by + name until it lands (§7.4). What a port needs is the entry as an ordinary array-of-tables element in its measure, save and load, the writer's sort, the reader's one compare with its two events, the const `Find` as a binary search that allocates nothing, ascending iteration, and the text form's keyed object; each holds the same @@ -10556,13 +10573,14 @@ inspects everything in the schema built: own call, because it is never stored and no golden names it (§2.8's memory layout): what is deferred is the BENCH NUMBER that says the size above which a caller should reach for it, not the surface. -- **UNBOUNDED ARRAYS IN EVERY BACKEND** (§2.9). The LANGUAGE carries the - construct, which is the parser's `[]T`, the checker's refusals and its two - claimed names, and the record's reference-and-count slot, and every backend - refuses a unit that declares one, by name (§11), until its codec lands. The C++ - reference and the tool are first: the builder's segments and `Add`, the four - walks in index order, the region load, the const `TableList` surface, the - text form's array and `schema cook-check`'s element-array clause. What a port +- **UNBOUNDED ARRAYS IN EVERY PORT** (§2.9). The LANGUAGE carries the + construct, which is the parser's `[]T`, the checker's refusals and its three + claimed names, and the record's reference-and-count slot, and every port + refuses a unit that declares one, by name (§11), until its codec lands. The + C++ reference carries it: the builder's segments and `Add`, the four walks + in index order, the region load, the const `TableList` surface, the text + form's array, and `schema cook-check`'s element-array clause in the tool; + the tool's COOK and UNCOOK halves are owed beside the map's. What a port needs is SMALLER than what a map needed, and by exactly the key: the element is an ordinary array element its measure, save and load already carry, there is no sort, no key compare and neither of the map's two reader events, and @@ -13123,7 +13141,12 @@ of declaration it names"*: carries a `bound=`**, because neither declares an extent and both take their count from the wire. An unbounded array generates no second record, because it generates no entry (§2.9), so its element's own `record` line is the only - one it needs and that line is already there under the element's name. + one it needs and that line is already there under the element's name. **A + list's line carries `kind=14`**, the array's own kind, as a map's does, + where a fixed or bounded array's line carries its ELEMENT's kind: the two + spellings are one wire (§2.9) and two storages, and this projection digests + storage. + - **A MAP's generated ENTRY takes a `record` line of its own, and it is ANONYMOUS.** The line carries the HOLDER's wire id and the MAP FIELD's wire id, joined by a dot, in place of a name, and it sorts with the named records diff --git a/docs/USAGE.md b/docs/USAGE.md index 21fee28c0..3b68f7708 100644 --- a/docs/USAGE.md +++ b/docs/USAGE.md @@ -2337,10 +2337,13 @@ its quoted decimal spelling, entries in ascending key order. ### Unbounded arrays: `placements []Placement` -*Partly landed. The front end takes the `[]T` spelling and holds every -refusal below, and `pack` and `unpack` read and write one. No backend carries -the construct — every one of them refuses a unit that declares one, by name — -and the corpus holds no `tables/lists` (SPEC-TABLES.md §2.9).* +*Carried by the C++ reference: the `TableList` runtime, the builder's `Add`, +`Each` and `Erase`, the wire, the region, the cook, the text form and the +descriptors, held by the `tables/lists` corpus. `pack`, `unpack` and +`cook-check` read and write one; the tool's `cook` and `uncook` halves do not +yet, and every port refuses a unit that declares one, by name +(SPEC-TABLES.md §2.9, §15).* + **An unbounded array is a counted array whose count the DATA decides.** It is the map with the key and the sort taken out: the same kind `14` body a diff --git a/generated/bench/tables/cpp/BenchTableTable.cpp b/generated/bench/tables/cpp/BenchTableTable.cpp index 72b431655..a431bb721 100644 --- a/generated/bench/tables/cpp/BenchTableTable.cpp +++ b/generated/bench/tables/cpp/BenchTableTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace benchtable #endif // BENCHTABLE_SCHEMA_TABLE_JSON diff --git a/internal/check/tablelist.go b/internal/check/tablelist.go index 8fc23d688..10b2fdc87 100644 --- a/internal/check/tablelist.go +++ b/internal/check/tablelist.go @@ -5,8 +5,8 @@ // almost nothing here is new machinery: the ELEMENT resolves through the // ordinary array path, so whatever `[..N]T` admits `[]T` admits and whatever // `[..N]T` refuses `[]T` refuses on the bounded array's own diagnostic. What -// this file adds is the placements the construct is refused in and the -// qualifications it does not take. +// this file adds is the placements the construct is refused in and the two +// qualifications it does not take, `?` and a specified default. package check import ( @@ -42,18 +42,11 @@ func (c *checker) checkListSpelling(f *ast.Field, inTable bool) bool { f.Name) return false } - for i := range f.Attrs { - a := &f.Attrs[i] - switch a.Key { - case "was", "json": - // a list is renamed under `was` as any field is, and takes a - // `json` key as any field does: both are about the field, not the - // construct (docs/SPEC-TABLES.md §2.9, §5, §16.4) - default: - c.errf(a.Pos, "field %s: %s does not apply to an unbounded array — THE COUNT IS THE DATA'S, and a bound would buy only a CLAMP, which drops a tail; drop the qualification, or declare the array at a bound, [..N]T, which is the same bytes with a bound (docs/SPEC-TABLES.md §2.9, §11)", - f.Name, a.Key) - return false - } - } + // THE BAR ATTRIBUTES QUALIFY THE ELEMENT, exactly as they do on a `[..N]T` + // (docs/SPEC-TABLES.md §2.9, §11): `min` and `max` bound each element, `was` + // renames the field and `json` keys its text. What the construct has no + // spelling for is a COUNT bound, and there is no attribute that names one, + // so nothing is refused here and the element path judges each attribute + // on the bounded array's own terms. return true } diff --git a/internal/check/tables_test.go b/internal/check/tables_test.go index afed786e3..5df514076 100644 --- a/internal/check/tables_test.go +++ b/internal/check/tables_test.go @@ -349,8 +349,6 @@ func TestTableRefusals(t *testing.T) { src: "package t\ntable E { a uint32 }\ntable Tab { xs ?[]E }\n"}, {name: "a default on a []T is refused by name", want: "a []T takes no specified default", src: "package t\ntable Tab { xs []int32 = 0 }\n"}, - {name: "a qualification on a []T is refused by name", want: "does not apply to an unbounded array", - src: "package t\ntable Tab { xs []int32 | max = 4 }\n"}, // the bounded spellings of the construct itself, and the element set's // own four refusals, each on the bounded array's own diagnostic {name: "[][]T is an array of arrays", want: "an array of arrays is not supported in v1", @@ -1307,3 +1305,19 @@ func TestTheIdRefusalsUnderAPlantedCollision(t *testing.T) { } } } + +// TestListQualificationBoundsTheElement: a bar attribute on a `[]T` qualifies +// the ELEMENT, exactly as it does on a `[..N]T` (docs/SPEC-TABLES.md §2.9, +// §11). The construct has no spelling for a count bound, and `max` is not +// one: it is the element's range, and the checker judges it on the bounded +// array's own terms. +func TestListQualificationBoundsTheElement(t *testing.T) { + u := buildUnit(t, "package t\ntable Tab { xs []int32 | min = 0, max = 4 }\n") + f := u.Tables["Tab"].Fields[0] + if !f.IsList() { + t.Fatalf("xs is not an unbounded array: %+v", f.Array) + } + if !f.HasIntRange || f.IntMin.Int64() != 0 || f.IntMax.Int64() != 4 { + t.Fatalf("the qualification did not bound the element: has=%v min=%v max=%v", f.HasIntRange, f.IntMin, f.IntMax) + } +} diff --git a/internal/codegen/cpptable/arena.go b/internal/codegen/cpptable/arena.go index da8befe43..dde48a91e 100644 --- a/internal/codegen/cpptable/arena.go +++ b/internal/codegen/cpptable/arena.go @@ -25,29 +25,20 @@ import ( // tableArenaRuntime is the variable-length runtime, guarded per package like // tablePrimitives so one definition survives any include order. -func tableArenaRuntime(pkg string, anyMap bool) string { +func tableArenaRuntime(pkg string, anyExtent bool) string { guard := strings.ToUpper(pkg) + "_SCHEMA_TABLE_ARENA" - // A MAP-FREE UNIT CARRIES NOT ONE SYMBOL OF THE MAP MACHINERY - // (docs/SPEC-TABLES.md §2.2, §2.8), the node map's extent cursor included: - // a pointered unit with no map emits exactly the arena runtime it always - // emitted, to the byte. - // AllocRaw is a MAP symbol (docs/SPEC-TABLES.md §2.8) and stays out of a - // map-free unit's header with the rest of them: nothing else allocates + // A UNIT WITH NEITHER A MAP NOR A LIST CARRIES NOT ONE SYMBOL OF THE EXTENT + // MACHINERY (docs/SPEC-TABLES.md §2.2, §2.8, §2.9), the node map's extent + // cursor included: a pointered unit with neither emits exactly the arena + // runtime it always emitted, to the byte. + // AllocRaw is an EXTENT symbol (docs/SPEC-TABLES.md §2.8, §2.9) and stays + // out of such a unit's header with the rest of them: nothing else allocates // storage that is not a node. - // and the refusal a node's storage answers when the FRAMING ITSELF is bad - // rather than merely unnameable — a map whose N its L cannot carry. - refusedConstant := "" - if anyMap { - refusedConstant = "\n\n// What a node's storage answers when the FRAMING ITSELF is refused rather than\n" + - "// merely unnameable: a map whose N cannot fit in its L (docs/SPEC-TABLES.md\n" + - "// §2.8). An unnameable type id commands no storage and keeps its index; this\n" + - "// one makes the whole measure answer -1 (§7.6).\nstatic const int64_t kTableNodeRefused = -2;" - } allocRaw := "" carveDecl, carveMember := "\n", "" - if anyMap { - allocRaw = ` // RAW, ZEROED storage of the bytes asked for, at the alignment asked for — a MAP's builder head and its - // entry segments (docs/SPEC-TABLES.md §2.8). It is not a node: it carries + if anyExtent { + allocRaw = ` // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries // no type id, takes no index and has no Reset, so it goes through the same // slab and span the blob path uses rather than through Alloc. uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) @@ -75,18 +66,23 @@ func tableArenaRuntime(pkg string, anyMap bool) string { return TableArenaAt( *arena, at ); // the segment came back zeroed } ` - carveDecl = "\n\n// a map's extent cursor, defined with the map runtime (docs/SPEC-TABLES.md\n// §2.8); the node map names it only through a pointer.\nstruct TableMapCarve;\n" - carveMember = "\n // WHERE A MAP'S ENTRIES LAND while this node's body decodes\n" + - " // (docs/SPEC-TABLES.md §2.8): the node's own extent on the region path\n" + - " // and the builder's arena on the tool's. It is MUTABLE because the\n" + - " // cursor belongs to ONE node's decode and the dispatch that owns that\n" + - " // node holds the map by const reference, exactly as it did before maps\n" + - " // existed — the decoder's signature does not move for a construct it\n" + - " // may not carry.\n mutable TableMapCarve * carve = NULL;\n" + - " // and the TOOL's path's allocation front, set once: there a map's\n" + - " // entries are the builder's arena's rather than a node's extent.\n" + - " TableWorker * worker = NULL;" + carveDecl = "\n\n// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md\n// §2.8, §2.9); the node map names it only through a pointer.\nstruct TableExtentCarve;\n" + carveMember = "\n // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body\n" + + " // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the\n" + + " // region path and the builder's arena on the tool's. It is MUTABLE\n" + + " // because the cursor belongs to ONE node's decode and the dispatch that\n" + + " // owns that node holds the map by const reference, exactly as it did\n" + + " // before either construct existed. The decoder's signature does not\n" + + " // move for a construct it may not carry.\n mutable TableExtentCarve * carve = NULL;\n" + + " // and the TOOL's path's allocation front, set once: there the arrays\n" + + " // are the builder's arena's rather than a node's extent.\n" + + " TableWorker * worker = NULL;\n" + + " // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the\n" + + " // int32 cap met while a body decoded. LoadBuilder answers NULL for it\n" + + " // and moves no counter; mutable for the reason the cursor is.\n" + + " mutable bool refused = false;" } + return `#ifndef ` + guard + ` #define ` + guard + ` @@ -731,7 +727,7 @@ static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts th // The not-materialized sentinel (§6.3): a record whose type id this build could // not name. Distinct from every real offset including the root's 0, so an index // resolving through it yields NULL and can never fabricate the root. -static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull;` + refusedConstant + ` +static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; // ---- the numbering, on the SAVE side ---- // diff --git a/internal/codegen/cpptable/codecs.go b/internal/codegen/cpptable/codecs.go index c430e9164..292331510 100644 --- a/internal/codegen/cpptable/codecs.go +++ b/internal/codegen/cpptable/codecs.go @@ -219,6 +219,15 @@ func (g *tableGen) emitTableStorageField(f *ir.Field) { g.pf(" TableMap<%s> %s; // %s — the sorted entry array, empty until an insert\n", f.MapEntry.Name, f.Name, ir.TableTypeSpelling(f)) return } + if f.IsList() { + // AN UNBOUNDED ARRAY FIELD IS SIXTEEN BYTES (docs/SPEC-TABLES.md §2.9, + // §7.2): the map's slot exactly, a self-relative reference to the + // element array and the live count, then padding to eight. The ELEMENTS + // are not here: they are by-value records inside the holder's node + // extent, laid after the record's own storage. + g.pf(" %s %s; // %s: the element array, empty until an Add\n", g.listStorageType(f), f.Name, ir.TableTypeSpelling(f)) + return + } if f.Type.Pointer { // a pointer is EIGHT BYTES and no address: an arena offset while the // builder is mutable, a self-relative delta once packed. That is what @@ -346,7 +355,15 @@ func (g *tableGen) emitTableResetField(f *ir.Field) { // null in both encodings and the live count is zero. The builder's // head and its segments are the arena's, and Reset does not free them // — the arena's own reset is what reclaims a dead entry's storage. - g.pf(" value.%s.entries.value = 0; // %s — empty\n", f.Name, ir.TableTypeSpelling(f)) + g.pf(" value.%s.entries.value = 0; // %s: empty\n", f.Name, ir.TableTypeSpelling(f)) + g.pf(" value.%s.count = 0;\n", f.Name) + g.pf(" value.%s.padding = 0;\n", f.Name) + return + } + if f.IsList() { + // a fresh list is EMPTY (docs/SPEC-TABLES.md §2.9), on the map's terms: + // the reference is null in both encodings and the live count is zero + g.pf(" value.%s.elements.value = 0; // %s: empty\n", f.Name, ir.TableTypeSpelling(f)) g.pf(" value.%s.count = 0;\n", f.Name) g.pf(" value.%s.padding = 0;\n", f.Name) return @@ -712,6 +729,10 @@ func (g *tableGen) emitTableMeasureField(f *ir.Field) { g.emitMapMeasureField(f) return } + if f.IsList() { + g.emitListMeasureField(f) + return + } id := tableFieldWireId(f) kind := tableScalarKind(f) width := tableKindWidth(kind) @@ -1103,6 +1124,10 @@ func (g *tableGen) emitTableWriteField(f *ir.Field) { g.emitMapWriteField(f) return } + if f.IsList() { + g.emitListWriteField(f) + return + } id := tableFieldWireId(f) kind := tableScalarKind(f) elemKind := kind @@ -1579,6 +1604,10 @@ func (g *tableGen) emitTableReadField(f *ir.Field, kind int) { g.emitMapReadField(f) return } + if f.IsList() { + g.emitListReadField(f) + return + } switch { case f.KeyEnum != "": // each triple is placed by its KEY REFERENCE, so a slot lands by name @@ -2146,8 +2175,38 @@ func (g *tableGen) emitFieldInfo(f *ir.Field, sp fieldSpelling, hoisted bool) { elemSize = elemSizeOverride countOffset = "0xffffffffu" } + if f.IsMap() || f.IsList() { + // AN OUT-OF-LINE ARRAY (docs/SPEC-TABLES.md §8.1): an array field whose + // elements are not inline, and array_bound = 0 is what says so. kind is + // the ELEMENT's, as on every array line: 13 for a map's entry, the + // element's own for a list, 17 for a []*T. is_array and counted are + // set, elem_size is the pitch, count_offset names the int32 count + // beside the reference in the sixteen-byte slot, and offset names the + // REFERENCE, which a walker resolves before it steps. + isArray = true + counted = true + bound = "0" + countOffset = fmt.Sprintf("(uint32_t) offsetof( %s, %s.count )", sp.owner, sp.member) + if f.IsMap() { + kind = tkTable + elemSize = fmt.Sprintf("(uint32_t) sizeof( %s )", f.MapEntry.Name) + } else { + kind = listElementWireKind(f) + elemSize = fmt.Sprintf("(uint32_t) sizeof( %s )", g.listElementType(f)) + } + } table := "NULL" + if f.IsMap() { + // the generated ENTRY's descriptor: fields[0] is the key and fields[1] + // the value (§2.8, §8.1) + if hoisted { + table = "&" + f.MapEntry.Name + "TableInfo" + } else { + table = fmt.Sprintf("%sTableType()", f.MapEntry.Name) + } + } if _, isStruct := f.Type.Ref.(*ir.Struct); f.Type.Kind == ir.TNamed && isStruct { + if hoisted { // an ADDRESS, not a call: constant-initialisable, so a // self-reference (Node -> *Node) is simply &NodeTableInfo @@ -2269,5 +2328,5 @@ func (g *tableGen) emitFieldInfo(f *ir.Field, sp fieldSpelling, hoisted bool) { sp.indent, f.Name, ir.TableFieldJsonKey(f), tableFieldTypeName(f), id, kind, isArray, pointerColumn, counted, f.Type.Optional, bound, sp.owner, sp.member, elemSize, countOffset, presentOffset, table, hasRange, rangeMin, rangeMax, fracBits, wide, enumMax, enumName, variantId, - keyTypeName, keyName, keyId, arms, g.mapColumn(f), sp.guard) + keyTypeName, keyName, keyId, arms, g.placeColumn(f), sp.guard) } diff --git a/internal/codegen/cpptable/cookwrite.go b/internal/codegen/cpptable/cookwrite.go index aa1200fab..77d815af4 100644 --- a/internal/codegen/cpptable/cookwrite.go +++ b/internal/codegen/cpptable/cookwrite.go @@ -220,7 +220,8 @@ func (g *tableGen) emitCookWriteSurface(members []*ir.Struct) { for _, st := range bodies { g.emitCookWriteBody(st) } - g.emitCookMapSurface(members) + g.emitCookExtentSurface(members) + for _, st := range bodies { if !st.IsTable || st.IsMapEntry() { // a `type` is no root, and a map's generated ENTRY is §2.8's one @@ -250,18 +251,16 @@ func (g *tableGen) emitCookWriteBody(st *ir.Struct) { g.pf(" (void) at; (void) value; (void) order; // a record with no field writes nothing\n") } } - if variable && len(ml.Fields) > 0 && g.noVariableEdges(st) { - g.pf(" (void) ctx; (void) region; // no reference below this node: the class was decided by a pointer elsewhere in its closure\n") + if variable && len(ml.Fields) > 0 && g.noCookRefs(st) { + g.pf(" (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure\n") } - if len(ml.Fields) > 0 && onlyMapFields(st) { - // every field is a MAP, whose slot the extent writer fills: this body - // writes the empty sixteen bytes and reads nothing off the value, and - // no reference of its own resolves here + if len(ml.Fields) > 0 && onlyExtentFields(st) { + // every field is a LIST or a MAP, whose slot the extent writer fills: + // this body writes the empty sixteen bytes and reads nothing off the + // value g.pf(" (void) value;\n") - if variable && !g.noVariableEdges(st) { - g.pf(" (void) ctx; (void) region;\n") - } } + for i := range ml.Fields { fl := &ml.Fields[i] g.emitCookWriteField(st, fl.Field, fl.Offset) @@ -281,13 +280,14 @@ func (g *tableGen) emitCookWriteField(st *ir.Struct, f *ir.Field, offset int64) // in a member of the record (docs/SPEC-TABLES.md §2.6), and its pieces sit at // the arms' shared offset. func (g *tableGen) emitCookWriteFieldAs(st *ir.Struct, f *ir.Field, offset int64, name, base, sfx string) { - if f.IsMap() { - // THE SLOT IS THE EXTENT WRITER'S (docs/SPEC-TABLES.md §2.8): the + if f.IsMap() || f.IsList() { + // THE SLOT IS THE EXTENT WRITER'S (docs/SPEC-TABLES.md §2.8, §2.9): the // reference is a delta to an array this record's own extent holds, and - // only CookMaps knows where that landed. The record's sixteen bytes + // only CookExtent knows where that landed. The record's sixteen bytes // are written EMPTY here, which is what a node the walk never reaches - // keeps — and is why an unreached non-empty map is refused (§7.6). - g.pf(" table_cook_put( %s + %d, 0, 8, order ); // %s: the entry array's delta, filled by the extent writer\n", base, offset, f.Name) + // keeps, and is why an unreached non-empty list or map is refused + // (§7.6). + g.pf(" table_cook_put( %s + %d, 0, 8, order ); // %s: the array's delta, filled by the extent writer\n", base, offset, f.Name) g.pf(" table_cook_put( %s + %d, 0, 4, order ); // and its count\n", base, offset+8) return } @@ -531,22 +531,22 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf("// eight. The offsets go into the region's table when it has one, and are only\n") g.pf("// summed when it does not (a measure). A type id the numbering carries that\n") g.pf("// this root cannot name is the two walks disagreeing, and it is refused.\n") - if g.anyMap { - g.pf("// A NODE'S SIZE DEPENDS ON ITS VALUE where a map rides in its extent\n") + if g.anyExtent { + g.pf("// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent\n") g.pf("// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context\n") - g.pf("// the numbering walked and reads the same maps that walk read.\n") + g.pf("// the numbering walked and reads the same arrays that walk read.\n") g.pf("template \ninline bool %sCookLayout( const Ctx & ctx, const %s & root, const TableNumbering & numbering, TableCookRegion & region )\n{\n", n, n) } else { g.pf("inline bool %sCookLayout( const TableNumbering & numbering, TableCookRegion & region )\n{\n", n) } g.pf(" region.numbering = &numbering;\n") g.pf(" region.count = numbering.count + 1;\n") - if g.anyMap && g.hasMapExtent(st) { - g.pf(" const int64_t root_extent = %sMapExtent( ctx, root );\n", n) + if g.anyExtent && g.hasExtent(st) { + g.pf(" const int64_t root_extent = %sExtent( ctx, root );\n", n) g.pf(" if ( root_extent < 0 ) { return false; }\n") g.pf(" int64_t offset = %d + root_extent; // the root at zero, its extent behind it\n", cookAlignUp(ml.Size, ir.RegionAlignFloor)) } else { - if g.anyMap { + if g.anyExtent { g.pf(" (void) root;\n") } g.pf(" int64_t offset = %d; // the root, at zero\n", ml.Size) @@ -559,7 +559,7 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf(" switch ( numbering.entries[k].type_id )\n {\n") for _, t := range reachable { tl := ir.RecordLayout(g.unit, t) - if g.anyMap && g.hasMapExtent(t) { + if g.anyExtent && g.hasExtent(t) { g.pf(" case 0x%016xull: // %s\n", ir.TableWireId(t.Name), t.Name) g.emitCookNodeBytes(t, " ", fmt.Sprintf("*(const %s *) numbering.entries[k].node", t.Name), "return false;") g.pf(" break;\n") @@ -596,7 +596,7 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf(" TableNumberingInit( numbering, allocator );\n") g.pf(" TableCookRegion region;\n") g.pf(" int64_t bytes = -1;\n") - if g.anyMap { + if g.anyExtent { g.pf(" if ( %sNumberFrom( ctx, numbering, root ) && %sCookLayout( ctx, root, numbering, region ) )\n {\n", n, n) } else { g.pf(" if ( %sNumberFrom( ctx, numbering, root ) && %sCookLayout( numbering, region ) )\n {\n", n, n) @@ -627,7 +627,7 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf(" bool ok = %sNumberFrom( ctx, numbering, root );\n", n) g.pf(" if ( ok )\n {\n") g.pf(" region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) );\n") - if g.anyMap { + if g.anyExtent { g.pf(" ok = region.offsets != NULL && %sCookLayout( ctx, root, numbering, region );\n", n) } else { g.pf(" ok = region.offsets != NULL && %sCookLayout( numbering, region );\n", n) @@ -644,7 +644,7 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf(" region.base = raw + data_offset;\n") g.pf(" // the DATA part: the root at the region's base, then every numbered\n") g.pf(" // node at the offset the layout gave it, each through its own writer\n") - if g.anyMap { + if g.anyExtent { g.pf(" ok = %sCookNode( ctx, region, region.base, root, order );\n", n) } else { g.pf(" ok = %sCookBody( ctx, region, region.base, root, order );\n", n) @@ -660,7 +660,7 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf(" switch ( numbering.entries[k].type_id )\n {\n") } for _, t := range reachable { - if g.anyMap { + if g.anyExtent { g.pf(" case 0x%016xull: ok = %sCookNode( ctx, region, at, *(const %s *) node, order ); break; // %s\n", ir.TableWireId(t.Name), t.Name, t.Name, t.Name) continue } @@ -737,3 +737,20 @@ func (g *tableGen) emitCookWriteVariableRoot(st *ir.Struct) { g.pf(" TableArenaCtx ctx = { &builder.arena };\n") g.pf(" return %sCookFrom( ctx, *(const %s *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator );\n}\n\n", n, n) } + +// noCookRefs reports a record whose cook body resolves no reference through +// the context: no pointer or byte buffer slot, no by-value nesting or union arm +// that holds one. A list's and a map's slots are not references here, the +// extent writer fills them (§2.8, §2.9), so a record whose only edges are +// those has nothing to resolve, and its ctx and region are named unused. +func (g *tableGen) noCookRefs(st *ir.Struct) bool { + for _, f := range st.Fields { + if f.IsMap() || f.IsList() { + continue + } + if g.edgeOf(f) != edgeNone { + return false + } + } + return true +} diff --git a/internal/codegen/cpptable/cpptable.go b/internal/codegen/cpptable/cpptable.go index 544cd25ce..da168373a 100644 --- a/internal/codegen/cpptable/cpptable.go +++ b/internal/codegen/cpptable/cpptable.go @@ -89,6 +89,13 @@ type tableGen struct { // §2.8). It gates the map runtime and every map-shaped walk, so not one // symbol of the machinery reaches a map-free unit's header (§2.2). anyMap bool + // anyList is the unit declaring at least one `[]T` (docs/SPEC-TABLES.md + // §2.9). It gates the list runtime and the builder's three the same way. + anyList bool + // anyExtent is either: the unit carries the NODE EXTENT machinery both + // constructs share: the carve, the framing walk, the extent walks and the + // refusal reason (§2.8, §2.9, §6.5). A unit with neither carries none of it. + anyExtent bool // blocks is the unit's BLOCK FORM surface (docs/SPEC-TABLES.md §19), nil when // no table is marked `| block`. Nil is what makes the zero-cost gate // answerable by asking one question (§2.2). @@ -355,7 +362,7 @@ struct TableKeyed // same way would be a redefinition. func tableInlineMacro(pkg string) string { return strings.ToUpper(pkg) + "_TABLE_INLINE" } -func tablePrimitives(pkg string, anyVariable bool, anyKeyed bool, anyMap bool, idCap int, u *ir.Unit) string { +func tablePrimitives(pkg string, anyVariable bool, anyKeyed bool, anyExtent bool, idCap int, u *ir.Unit) string { // THE ID TABLE'S CAPACITY IS A COMPILE-TIME FACT of the unit (§3): the // distinct names its table closure can spell, so a save allocates nothing. // The bucket count is the next power of two at twice the capacity, so the @@ -379,24 +386,35 @@ func tablePrimitives(pkg string, anyVariable bool, anyKeyed bool, anyMap bool, i // the two pointer-era descriptor members exist only in a unit that HAS // pointers: a unit of value-only tables emits the descriptor surface it // always emitted, to the byte (docs/SPEC-TABLES.md §2, the zero-cost gate) - // the MAP columns (docs/SPEC-TABLES.md §2.8, §16), emitted only into a unit - // that declares one: the generated ENTRY's descriptor, whose two fields - // ARE the key and the value, and the three thunks the ONE walk cannot - // spell for itself because TableMap is a type it has no name for. - mapFieldMember := "" - if anyMap { - mapFieldMember = "\n // a MAP (docs/SPEC-TABLES.md §2.8): the generated ENTRY's descriptor —\n" + - " // fields[0] is the key and fields[1] the value — and the three the ONE\n" + - " // text walk cannot spell for itself, because TableMap is a type\n" + - " // it has no name for. NULL on every field that is not a map.\n" + - " const TableTypeInfo * entry;\n" + - " int32_t ( * map_count )( const void * slot );\n" + - " const void * ( * map_at )( const void * slot, int32_t index );\n" + - " // place one entry BY KEY and hand back the entry, at its defaults: a\n" + - " // string key comes in as the bytes and the length, an integer key as\n" + - " // the value, and NULL is NOT INSERTED — a key past the bound, or an\n" + - " // arena that could not carve another segment.\n" + - " void * ( * map_insert )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value );" + // the PLACE column (docs/SPEC-TABLES.md §8.1, §16), emitted only into a + // unit that declares a map or an unbounded array: the one resolver the ONE + // text walk cannot spell for itself, because TableMap and + // TableList are types it has no name for. The SHAPE of either is read + // from the array columns like every other array's, with array_bound = 0 + // the one tell that the offset names a reference and not the first element. + placeMember := "" + refuseReason := "" + if anyExtent { + placeMember = "\n // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and\n" + + " // hand it back at its defaults. A MAP places BY KEY, a string key comes\n" + + " // in as the bytes and the length, an integer key as the value, and NULL\n" + + " // is NOT INSERTED: a key past the bound, or an arena that could not carve\n" + + " // another segment. A LIST ignores the key and APPENDS, NULL at the arena\n" + + " // or the int32 cap. NULL on every field that is neither.\n" + + " void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value );" + refuseReason = ` + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +};` } pointerFieldMember, pointerTypeMember, pointerForward := "", "", "" if anyVariable { @@ -479,7 +497,7 @@ struct TableReport // must not look at then (docs/SPEC-TABLES.md §3.3). TableMessageReason reason = newer_form; }; - +` + refuseReason + ` // ---- reflection (tables only, docs/SPEC-TABLES.md) ---- // // Static field descriptors for every type in the table closure: name, wire @@ -580,7 +598,7 @@ struct TableWideRange // descriptor stays CONSTANT-INITIALISED (a captureless lambda converts to // a function pointer at compile time; the arms themselves are a static // inside it). NULL for every other kind. - const TableUnionInfo * (*arms)();` + mapFieldMember + ` + const TableUnionInfo * (*arms)();` + placeMember + ` const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded }; @@ -971,6 +989,8 @@ func Generate(u *ir.Unit) (map[string][]byte, error) { anyVariable := len(variable) > 0 anyKeyed := unitHasKeyedArray(u, closure) anyMap := unitHasMap(u, closure) + anyList := unitHasList(u, closure) + anyExtent := anyMap || anyList blocks := ir.Blocks(u) // The BLOCK FORM (docs/SPEC-TABLES.md §19) is emitted ON THE SIDE, into @@ -987,7 +1007,7 @@ func Generate(u *ir.Unit) (map[string][]byte, error) { slots[id] = uint64(i + 1) } for _, f := range u.Files { - g := &tableGen{unit: u, file: f, anyVariable: anyVariable, anyKeyed: anyKeyed, anyMap: anyMap, blocks: blocks, variable: variable, targets: targets, + g := &tableGen{unit: u, file: f, anyVariable: anyVariable, anyKeyed: anyKeyed, anyMap: anyMap, anyList: anyList, anyExtent: anyExtent, blocks: blocks, variable: variable, targets: targets, includes: map[string]bool{}, nativeIncludes: map[string]bool{}, slots: slots} var members []*ir.Struct members = append(members, orderTables(f.Tables)...) @@ -1047,6 +1067,7 @@ func Generate(u *ir.Unit) (map[string][]byte, error) { } g.pf("\n") g.emitCookLayoutAsserts(members) + g.emitListAlignAsserts(members) g.pf("// ---- reflection descriptors (tables only, docs/SPEC-TABLES.md) ----\n\n") for _, st := range members { g.pf("inline const TableTypeInfo * %sTableType();\n", st.Name) @@ -1097,11 +1118,13 @@ func Generate(u *ir.Unit) (map[string][]byte, error) { // The relocatability and standard-layout asserts read the COMPILER // INTRINSICS, so — 124 headers on its own — is not here. h.WriteString("#pragma once\n\n#include \n#include // the prefill's scalar-array fills\n#include // offsetof, for the reflection descriptors\n") - if anyKeyed { - // ENUM-KEYED arrays only: indexing one by None is a program error in - // EVERY configuration, and the accessor is where a runtime key can - // first be caught (docs/SPEC-TABLES.md §2.4). It is the runtime's - // only refusal, so it is the only reason these two hooks are here. + if anyKeyed || anyList { + // ENUM-KEYED arrays and UNBOUNDED arrays only: indexing a keyed array + // by None, or a list past its count, is a program error in EVERY + // configuration, and the accessor is where a runtime key or index can + // first be caught (docs/SPEC-TABLES.md §2.4, §2.9). Those are the + // runtime's only refusals, so they are the only reason these two + // hooks are here. h.WriteString(tableHooks) } if anyVariable { @@ -1137,19 +1160,32 @@ func Generate(u *ir.Unit) (map[string][]byte, error) { fmt.Fprintf(&h, "#include \"%s\"\n", n) } h.WriteString("\n") - h.WriteString(tablePrimitives(u.Package, anyVariable, anyKeyed, anyMap, ir.TableWireIdCapacity(u), u)) + h.WriteString(tablePrimitives(u.Package, anyVariable, anyKeyed, anyExtent, ir.TableWireIdCapacity(u), u)) if anyVariable { h.WriteString("\n") - h.WriteString(tableArenaRuntime(u.Package, anyMap)) + h.WriteString(tableArenaRuntime(u.Package, anyExtent)) + } + if anyExtent { + // the NODE EXTENT runtime (docs/SPEC-TABLES.md §2.8, §2.9): what a + // map and an unbounded array share once the key and the sort are + // taken out. Either makes its holder variable-length, so it always + // follows the arena runtime it is spelled in terms of. + h.WriteString("\n") + h.WriteString(tableExtentRuntime(u.Package)) } if anyMap { // the MAP runtime (docs/SPEC-TABLES.md §2.8): the storage type, the // order, the builder's head and segments, and the optional index. - // A map makes its holder variable-length, so it always follows the - // arena runtime it is spelled in terms of. h.WriteString("\n") h.WriteString(tableMapRuntime(u.Package)) } + if anyList { + // the LIST runtime (docs/SPEC-TABLES.md §2.9): the storage type and + // its const surface, the builder's head and segments, the index-order + // cursor and the load side's fill. + h.WriteString("\n") + h.WriteString(tableListRuntime(u.Package)) + } // the COOKED FORM's read side (docs/SPEC-TABLES.md §7) and the BUILD VERSION // it matches against, in EVERY unit that declares a table: every table // cooks and any table may be a cook's root, so there is no unit with @@ -1188,9 +1224,10 @@ func Generate(u *ir.Unit) (map[string][]byte, error) { c.WriteString("#include // the text form: number formatting\n") c.WriteString("#include // the text form: exact number conversion\n") c.WriteString("#include // the text form: the runtime's decimal point\n\n") - c.WriteString(tableJsonWalk(u.Package, anyVariable, anyMap)) + c.WriteString(tableJsonWalk(u.Package, anyVariable, anyMap, anyList)) fmt.Fprintf(&c, "\nnamespace %s {\n\n", u.Package) - cg := &tableGen{unit: u, file: f, anyVariable: anyVariable, anyMap: anyMap, blocks: blocks, variable: variable, targets: targets, + cg := &tableGen{unit: u, file: f, anyVariable: anyVariable, anyMap: anyMap, anyList: anyList, anyExtent: anyExtent, blocks: blocks, variable: variable, targets: targets, + includes: map[string]bool{}, nativeIncludes: map[string]bool{}} for _, st := range members { cg.emitJsonDefinitions(st) diff --git a/internal/codegen/cpptable/extent.go b/internal/codegen/cpptable/extent.go new file mode 100644 index 000000000..95098ea97 --- /dev/null +++ b/internal/codegen/cpptable/extent.go @@ -0,0 +1,847 @@ +// The NODE EXTENT's shared runtime (docs/SPEC-TABLES.md §2.8, §2.9, §6.3): +// what a map and an unbounded array have in common once the key and the sort +// are taken out of the map. Both put their arrays in the holder's node extent +// after the record's own storage, both are carved from that extent as the +// node's body decodes, and both make LoadMeasure walk the wire's framing for +// every N at every depth. So the cursor, the framing walkers over the by-value +// edges that hold them, and the one test an unreached slot takes live here, +// emitted into a unit that declares either construct and into no other. +package cpptable + +import ( + "fmt" + "slices" + "strings" + + "github.com/mas-bandwidth/schema/v2/ir" +) + +// tableExtentRuntime is the extent half of the variable-length runtime. It +// follows the arena runtime it is spelled in terms of and precedes the map +// and list runtimes that are spelled in terms of it. +func tableExtentRuntime(pkg string) string { + guard := strings.ToUpper(pkg) + "_SCHEMA_TABLE_EXTENT" + return `#ifndef ` + guard + ` +#define ` + guard + ` + +namespace ` + pkg + ` { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace ` + pkg + ` + +#endif // ` + guard + ` +` +} + +// ---- the NODE EXTENT's emitters (docs/SPEC-TABLES.md §2.8, §2.9, §6.3) ---- +// +// A map's entries and a list's elements are BY-VALUE RECORDS INSIDE THE +// HOLDER'S NODE EXTENT, laid after the record's own storage: count x sizeof +// at the element's alignment, zero slack, one array per container reachable +// BY VALUE from the record, which includes one inside a nested table and one +// inside an element or an entry, in depth-first field order. The placement +// is PRE-ORDER and it interleaves lists and maps on ONE rule: a container's +// whole array first, then, element by element in the container's own order, +// the arrays of any list or map that element holds by value. +// +// Two emitters walk that layout and they are ONE walk: the measure advances a +// running offset, and the pack advances the same one and copies. Nothing +// passes between them, which is what makes `used == total` a real check. + +// hasExtent reports a member with any list or map reachable by value: the +// members that carry an extent, and the ones whose extent walks are emitted. +func (g *tableGen) hasExtent(st *ir.Struct) bool { + for _, f := range st.Fields { + if f.IsMap() || f.IsList() { + return true + } + // A CONTAINER REACHABLE BY VALUE IS THIS RECORD'S EXTENT, whichever + // by-value edge reaches it: a nested table, an array of them, an + // enum-keyed array of them, or a union arm. + switch g.edgeOf(f) { + case edgeNested: + if ref, ok := f.Type.Ref.(*ir.Struct); ok && g.hasExtent(ref) { + return true + } + case edgeArm: + for _, v := range f.Type.Ref.(*ir.Union).Variants { + if ref := memberOf(g.unit, v.Type); ref != nil && g.hasExtent(ref) { + return true + } + } + } + // a nested table this walk does not call an edge can still hold a + // container, because a container makes its holder VARIABLE and every + // variable nesting is an edge, so there is nothing else to look at + } + return false +} + +// memberOf resolves one closure member by name. +func memberOf(u *ir.Unit, name string) *ir.Struct { + if st := u.Tables[name]; st != nil { + return st + } + return u.Structs[name] +} + +// alignOfEntry is the C ABI alignment of one generated entry record, the +// alignment its array is laid at, and the same model §20.3 commits the +// compiler to for every record in the closure. +func alignOfEntry(u *ir.Unit, entry *ir.Struct) int64 { + if ml := ir.RecordLayout(u, entry); ml != nil && ml.Align > 0 { + return ml.Align + } + return 8 +} + +// extentVisitor is what one extent emitter does at the two containers and at +// a by-value nesting that holds one. +type extentVisitor struct { + mapField func(f *ir.Field, expr, ind string) + listField func(f *ir.Field, expr, ind string) + descend func(table, expr, ind string) +} + +// emitExtentWalk is the ONE walk every extent emitter takes: the record's +// fields in declaration order, every container at its own position, +// descending each by-value edge in place. A pointer is NOT an edge here, a +// pointee is its own node with its own extent, and neither is a union arm +// that is one, for the same reason. +func (g *tableGen) emitExtentWalk(st *ir.Struct, subject string, v extentVisitor) { + ev := edgeVisitor{read: subject} + for _, f := range st.Fields { + if f.IsMap() { + v.mapField(f, subject+"."+f.Name, " ") + continue + } + if f.IsList() { + v.listField(f, subject+"."+f.Name, " ") + continue + } + switch g.edgeOf(f) { + case edgeNested: + ref, _ := f.Type.Ref.(*ir.Struct) + if ref == nil || !g.hasExtent(ref) { + continue + } + g.emitVariableByValueWalk(f, ev, func(expr edgeExpr) { v.descend(f.Type.Name, expr.Src, " ") }) + g.emitUnreachedExtentRefusal(f, ref, subject) + case edgeArm: + un := f.Type.Ref.(*ir.Union) + any := false + for _, arm := range un.Variants { + if ref := memberOf(g.unit, arm.Type); ref != nil && g.hasExtent(ref) { + any = true + } + } + if !any { + continue + } + // only the ARMS that hold a container are descended: a pointer arm + // and a byte buffer arm reach nodes, not this node's extent + armed := edgeVisitor{read: subject, + pointer: func(*ir.Field, edgeExpr) {}, + blob: func(*ir.Field, edgeExpr) {}, + descend: func(table string, expr edgeExpr, indent string) { + if ref := memberOf(g.unit, table); ref != nil && g.hasExtent(ref) { + v.descend(table, expr.Src, indent) + } + }, + } + g.emitVariableUnionWalk(f, armed) + } + } +} + +// emitExtent emits `ExtentAt`: the running offset every array reachable by +// value from one record takes, in the order the pack lays them, and +// `Extent`, the whole extent from a fresh offset. +func (g *tableGen) emitExtent(st *ir.Struct) { + g.pf("// %sExtentAt: the node extent %s's lists and maps take, PRE-ORDER, advancing\n", st.Name, st.Name) + g.pf("// the running offset exactly as %sExtentPack advances it (§2.8, §2.9).\n", st.Name) + g.pf("template \ninline bool %sExtentAt( const Ctx & ctx, const %s & value, int64_t & at )\n{\n", st.Name, st.Name) + if !g.hasExtent(st) { + g.pf(" (void) ctx; (void) value; (void) at; // no list or map below this record\n") + g.pf(" return true;\n}\n\n") + return + } + g.emitExtentWalk(st, "value", extentVisitor{ + mapField: func(f *ir.Field, expr, ind string) { + entry := mapEntryOf(f) + g.pf("%s{\n", ind) + g.pf("%s TableMapCursor<%s> cursor = TableMapOrder( ctx, %s );\n", ind, entry.Name, expr) + g.pf("%s if ( !cursor.ok ) { return false; }\n", ind) + g.pf("%s at = ( at + %d ) & ~(int64_t) %d; // at alignof( %s )\n", ind, alignOfEntry(g.unit, entry)-1, alignOfEntry(g.unit, entry)-1, entry.Name) + g.pf("%s at += (int64_t) cursor.count * (int64_t) sizeof( %s ); // the whole array FIRST\n", ind, entry.Name) + if g.isVar(entry.Name) { + g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order\n%s {\n", ind, ind) + g.pf("%s if ( !%sExtentAt( ctx, *cursor[i], at ) ) { TableMapRelease( cursor ); return false; }\n", ind, entry.Name) + g.pf("%s }\n", ind) + } + g.pf("%s TableMapRelease( cursor );\n", ind) + g.pf("%s}\n", ind) + }, + listField: func(f *ir.Field, expr, ind string) { + elem := g.listElementType(f) + g.pf("%s{\n", ind) + g.pf("%s TableListCursor<%s> cursor = TableListElements( ctx, %s );\n", ind, elem, expr) + g.pf("%s if ( !cursor.ok ) { return false; }\n", ind) + g.pf("%s at = ( at + (int64_t) alignof( %s ) - 1 ) & ~( (int64_t) alignof( %s ) - 1 );\n", ind, elem, elem) + g.pf("%s at += (int64_t) cursor.count * (int64_t) sizeof( %s ); // the whole array FIRST\n", ind, elem) + if ref := listElementStruct(f); ref != nil && g.hasExtent(ref) { + g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ ) // then, element by element in index order\n%s {\n", ind, ind) + g.pf("%s if ( !%sExtentAt( ctx, cursor[i], at ) ) { return false; }\n", ind, ref.Name) + g.pf("%s }\n", ind) + } + g.pf("%s}\n", ind) + }, + descend: func(table, expr, ind string) { + g.pf("%sif ( !%sExtentAt( ctx, %s, at ) ) { return false; }\n", ind, table, expr) + }, + }) + g.pf(" return true;\n}\n\n") + g.pf("// the whole extent of one node, from a fresh offset: what a pack reserves\n") + g.pf("// for it beside the record's own storage.\n") + g.pf("template \ninline int64_t %sExtent( const Ctx & ctx, const %s & value )\n{\n", st.Name, st.Name) + g.pf(" int64_t at = 0;\n") + g.pf(" if ( !%sExtentAt( ctx, value, at ) ) { return -1; }\n", st.Name) + g.pf(" return at;\n}\n\n") +} + +// emitExtentPack emits `ExtentPack`: the same walk, copying each map's +// entries in key order and each list's elements in index order into the +// node's extent and pointing the record's slot at them. +func (g *tableGen) emitExtentPack(st *ir.Struct) { + g.pf("// %sExtentPack: carve %s's arrays out of the node's extent and copy the\n", st.Name, st.Name) + g.pf("// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER,\n") + g.pf("// advancing the same running offset %sExtentAt advances (§2.8, §2.9).\n", st.Name) + g.pf("template \ninline bool %sExtentPack( const Ctx & ctx, const %s & src, %s & dst, uint8_t * extent, int64_t & at, int64_t capacity )\n{\n", st.Name, st.Name, st.Name) + if !g.hasExtent(st) { + g.pf(" (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no list or map below this record\n") + g.pf(" return true;\n}\n\n") + return + } + dstOf := func(expr string) string { return "dst" + expr[len("src"):] } + g.emitExtentWalk(st, "src", extentVisitor{ + mapField: func(f *ir.Field, expr, ind string) { + entry := mapEntryOf(f) + slot := dstOf(expr) + g.pf("%s{\n", ind) + g.pf("%s TableMapCursor<%s> cursor = TableMapOrder( ctx, %s );\n", ind, entry.Name, expr) + g.pf("%s if ( !cursor.ok ) { return false; }\n", ind) + g.pf("%s at = ( at + %d ) & ~(int64_t) %d;\n", ind, alignOfEntry(g.unit, entry)-1, alignOfEntry(g.unit, entry)-1) + g.pf("%s const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( %s );\n", ind, entry.Name) + g.pf("%s if ( at + bytes > capacity ) { TableMapRelease( cursor ); return false; }\n", ind) + g.pf("%s %s * placed = (%s *) ( extent + at );\n", ind, entry.Name, entry.Name) + g.pf("%s at += bytes;\n", ind) + g.pf("%s %s.count = cursor.count;\n", ind, slot) + g.pf("%s %s.padding = 0;\n", ind, slot) + g.pf("%s %s.entries.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &%s.entries ) : 0;\n", ind, slot, slot) + g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ )\n%s {\n", ind, ind) + g.pf("%s memcpy( (void *) ( placed + i ), (const void *) cursor[i], sizeof( %s ) ); // trivially copyable, by construction\n", ind, entry.Name) + g.pf("%s }\n", ind) + if g.isVar(entry.Name) { + g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ )\n%s {\n", ind, ind) + g.pf("%s if ( !%sExtentPack( ctx, *cursor[i], placed[i], extent, at, capacity ) ) { TableMapRelease( cursor ); return false; }\n", ind, entry.Name) + g.pf("%s }\n", ind) + } + g.pf("%s TableMapRelease( cursor );\n", ind) + g.pf("%s}\n", ind) + }, + listField: func(f *ir.Field, expr, ind string) { + elem := g.listElementType(f) + slot := dstOf(expr) + g.pf("%s{\n", ind) + g.pf("%s TableListCursor<%s> cursor = TableListElements( ctx, %s );\n", ind, elem, expr) + g.pf("%s if ( !cursor.ok ) { return false; }\n", ind) + g.pf("%s at = ( at + (int64_t) alignof( %s ) - 1 ) & ~( (int64_t) alignof( %s ) - 1 );\n", ind, elem, elem) + g.pf("%s const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( %s );\n", ind, elem) + g.pf("%s if ( at + bytes > capacity ) { return false; }\n", ind) + g.pf("%s %s * placed = (%s *) ( extent + at );\n", ind, elem, elem) + g.pf("%s at += bytes;\n", ind) + g.pf("%s %s.count = cursor.count;\n", ind, slot) + g.pf("%s %s.padding = 0;\n", ind, slot) + g.pf("%s %s.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &%s.elements ) : 0;\n", ind, slot, slot) + g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only\n%s {\n", ind, ind) + g.pf("%s memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( %s ) ); // trivially copyable, by construction\n", ind, elem) + g.pf("%s }\n", ind) + if ref := listElementStruct(f); ref != nil && g.hasExtent(ref) { + g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ )\n%s {\n", ind, ind) + g.pf("%s if ( !%sExtentPack( ctx, cursor[i], placed[i], extent, at, capacity ) ) { return false; }\n", ind, ref.Name) + g.pf("%s }\n", ind) + } + g.pf("%s}\n", ind) + }, + descend: func(table, expr, ind string) { + g.pf("%sif ( !%sExtentPack( ctx, %s, %s, extent, at, capacity ) ) { return false; }\n", ind, table, expr, dstOf(expr)) + }, + }) + g.pf(" return true;\n}\n\n") +} + +// emitExtentWalkSurface emits the framing walk and the two extent walks for +// every variable member of a unit that declares a list or a map. They are +// emitted for EVERY such member, because a walk that descends a by-value +// nesting has to be able to name the nested one's. +func (g *tableGen) emitExtentWalkSurface(members []*ir.Struct) { + if !g.anyExtent { + return + } + for _, st := range g.varMembers(members) { + g.emitWireExtent(st) + g.emitExtent(st) + g.emitExtentPack(st) + } +} + +// emitNodeBytes emits the bytes ONE NODE takes in a packed region: the +// record's own storage rounded to the arena's alignment, plus the extent its +// lists and maps take (docs/SPEC-TABLES.md §2.8, §2.9, §6.3), the sum rounded +// again so the next node starts aligned. A unit with neither construct emits +// exactly the term it always emitted. +func (g *tableGen) emitNodeBytes(table, expr, ind, onBad string, plain func(term string), extent func(term string)) { + target := memberOf(g.unit, table) + if !g.anyExtent || target == nil || !g.hasExtent(target) { + // a node with no container below it takes exactly the term it always + // took, so a unit without one emits what it emitted before either + // construct existed + plain(fmt.Sprintf("TableAlignUp64( (int64_t) sizeof( %s ) )", table)) + return + } + g.pf("%sint64_t node_extent = %sExtent( ctx, %s );\n", ind, table, expr) + g.pf("%sif ( node_extent < 0 ) { %s }\n", ind, onBad) + extent(fmt.Sprintf("TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + node_extent )", table)) +} + +// emitWireExtent emits `WireExtent`: the region bytes one record's lists +// and maps command, read from the wire FRAMING alone at every depth +// (docs/SPEC-TABLES.md §2.8, §2.9, §6.5). False is the refusal, carrying its +// reason, and it is what makes LoadMeasure answer -1. +func (g *tableGen) emitWireExtent(st *ir.Struct) { + g.pf("// %sWireExtent: the extent %s's lists and maps command, from the FRAMING alone.\n", st.Name, st.Name) + g.pf("// It reads no field value, so a caller can refuse a number it did not\n") + g.pf("// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5).\n") + g.pf("inline bool %sWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason )\n{\n", st.Name) + if !g.hasExtent(st) { + g.pf(" (void) body; (void) length; (void) at; (void) ids; (void) reason; // no list or map below this record\n") + g.pf(" return true;\n}\n\n") + return + } + g.pf(" TableReport scratch; // the scan's framing damage is the LOAD's to report\n") + g.pf(" TableReader r( body, length, &scratch, ids );\n") + g.pf(" for ( ;; )\n {\n") + g.pf(" uint64_t field_ref = 0;\n") + g.pf(" if ( !r.getleb( field_ref ) ) { return true; }\n") + g.pf(" if ( field_ref == 0 ) { return true; }\n") + g.pf(" if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; }\n") + g.pf(" const uint64_t field_id = ids->at( field_ref );\n") + g.pf(" if ( !r.has( 1 ) ) { return true; }\n") + g.pf(" uint8_t field_kind = r.get8();\n") + g.emitWireExtentCases(st) + g.pf(" if ( !r.skip( field_kind ) ) { return true; }\n") + g.pf(" }\n}\n\n") +} + +// emitWireExtentCases emits one arm per list field, one per map field and one +// per by-value nesting that holds either, in DECLARATION ORDER, so the framing +// scan advances the running offset in the same order the pack and the load +// carve it. +func (g *tableGen) emitWireExtentCases(st *ir.Struct) { + for _, f := range st.Fields { + if f.IsMap() { + entry := mapEntryOf(f) + inner := "NULL" + if g.hasExtent(entry) { + inner = "&" + entry.Name + "WireExtent" + } + g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s\n {\n", ir.TableFieldWireId(f), tkArray, f.Name) + g.pf(" uint64_t map_len = 0;\n") + g.pf(" if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; }\n") + g.pf(" const uint8_t * map_body = r.buffer + r.offset;\n") + g.pf(" r.offset += (int64_t) map_len;\n") + g.pf(" if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( %s ), (int64_t) alignof( %s ), %s, ids, reason ) ) { return false; }\n", + entry.Name, entry.Name, inner) + g.pf(" continue;\n }\n") + continue + } + if f.IsList() { + elem := g.listElementType(f) + inner := "NULL" + if ref := listElementStruct(f); ref != nil && g.hasExtent(ref) { + inner = "&" + ref.Name + "WireExtent" + } + g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s: an unbounded array\n {\n", ir.TableFieldWireId(f), tkArray, f.Name) + g.pf(" uint64_t list_len = 0;\n") + g.pf(" if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; }\n") + g.pf(" const uint8_t * list_body = r.buffer + r.offset;\n") + g.pf(" r.offset += (int64_t) list_len;\n") + g.pf(" if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( %s ), (int64_t) alignof( %s ), %d, %d, %s, ids, reason ) ) { return false; }\n", + elem, elem, listElementWireKind(f), listElementFloor(f), inner) + g.pf(" continue;\n }\n") + continue + } + switch g.edgeOf(f) { + case edgeNested: + ref, _ := f.Type.Ref.(*ir.Struct) + if ref == nil || !g.hasExtent(ref) { + continue + } + // a nested table's arrays are part of THIS node's extent, so its + // own scan runs over the nested body at the running offset + kind, walk := tkTable, "" + switch { + case f.KeyEnum != "": + kind, walk = tkKeyed, "TableWireExtentKeyed" + case f.Array != ir.ArrayNone: + kind, walk = tkArray, "TableWireExtentElements" + } + g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s: a nesting that holds a list or a map\n {\n", ir.TableFieldWireId(f), kind, f.Name) + g.pf(" uint64_t nested_len = 0;\n") + g.pf(" if ( !r.getleb( nested_len ) || !r.room( nested_len ) ) { return true; }\n") + g.pf(" const uint8_t * nested_body = r.buffer + r.offset;\n") + g.pf(" r.offset += (int64_t) nested_len;\n") + if walk == "" { + g.pf(" if ( !%sWireExtent( nested_body, (int64_t) nested_len, at, ids, reason ) ) { return false; }\n", f.Type.Name) + } else { + g.pf(" if ( !%s( nested_body, (int64_t) nested_len, at, &%sWireExtent, ids, reason ) ) { return false; }\n", walk, f.Type.Name) + } + g.pf(" continue;\n }\n") + case edgeArm: + un := f.Type.Ref.(*ir.Union) + any := false + for _, v := range un.Variants { + if ref := memberOf(g.unit, v.Type); ref != nil && g.hasExtent(ref) { + any = true + } + } + if !any { + continue + } + g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s: a union arm that holds a list or a map\n {\n", ir.TableFieldWireId(f), tkUnion, f.Name) + g.pf(" uint64_t arm_ref = 0;\n") + g.pf(" if ( !r.getleb( arm_ref ) ) { return true; }\n") + g.pf(" if ( arm_ref == 0 ) { continue; } // None: the reference is the whole payload\n") + g.pf(" if ( arm_ref > (uint64_t) ids->count ) { return true; }\n") + g.pf(" const uint64_t arm_id = ids->at( arm_ref );\n") + g.pf(" if ( !r.has( 1 ) ) { return true; }\n") + g.pf(" r.offset += 1; // the arm's kind byte\n") + g.pf(" uint64_t arm_len = 0;\n") + g.pf(" if ( !r.getleb( arm_len ) || !r.room( arm_len ) ) { return true; }\n") + g.pf(" const uint8_t * arm_body = r.buffer + r.offset;\n") + g.pf(" r.offset += (int64_t) arm_len;\n") + g.pf(" switch ( arm_id )\n {\n") + for _, v := range un.Variants { + ref := memberOf(g.unit, v.Type) + if ref == nil || !g.hasExtent(ref) { + continue + } + g.pf(" case 0x%016xull: if ( !%sWireExtent( arm_body, (int64_t) arm_len, at, ids, reason ) ) { return false; } break; // %s\n", + ir.TableWireId(v.Name), v.Type, v.Name) + } + g.pf(" default: break; // an arm this reader cannot name reads None\n") + g.pf(" }\n") + g.pf(" continue;\n }\n") + } + } +} + +// emitRootDataBytes emits a load's DATA term for the root itself: its record, +// plus the extent its own lists and maps take, read from the wire framing +// (§2.8, §2.9, §6.5). `reason` is the refusal's carrier, declared by the +// caller where a unit has an extent. +func (g *tableGen) emitRootDataBytes(st *ir.Struct, ind, onBad string) { + if !g.anyExtent { + g.pf("%sint64_t data = TableAlignUp64( (int64_t) sizeof( %s ) );\n", ind, st.Name) + return + } + g.pf("%sint64_t root_extent = 0;\n", ind) + g.pf("%sif ( !%sWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { %s }\n", ind, st.Name, onBad) + g.pf("%sint64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + root_extent );\n", ind, st.Name) +} + +// ---- the COOK's write side at the extent (docs/SPEC-TABLES.md §2.8, §2.9, §7.6) ---- +// +// A cook is a region written verbatim, so a cooked map is its SORTED entry +// array and a cooked list its element array in INDEX order, where the cook +// put them: the node's extent, laid after the record's own storage by the +// same PRE-ORDER rule the pack lays it by. A map's Find is then a binary +// search over the mapped bytes and a list's indexing one multiply, in place, +// with nothing to parse. + +// cookExtentSignature is one record's extent writer. +func (g *tableGen) cookExtentSignature(st *ir.Struct) string { + return fmt.Sprintf("template inline bool %sCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const %s & value, TableByteOrder order )", st.Name, st.Name) +} + +// emitCookExtent emits one record's extent writer: every list and map +// reachable by value, PRE-ORDER, each element or entry through its own cook +// writer. +func (g *tableGen) emitCookExtent(st *ir.Struct) { + g.pf("// %sCookExtent: %s's arrays into the node's extent, PRE-ORDER, a map's entries\n", st.Name, st.Name) + g.pf("// in ASCENDING key order and a list's elements in INDEX order, each through its\n") + g.pf("// own cook writer (§2.8, §2.9, §7.6).\n") + g.pf("%s\n{\n", g.cookExtentSignature(st)) + if !g.hasExtent(st) { + g.pf(" (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order;\n") + g.pf(" return true; // no list or map below this record\n}\n\n") + return + } + ml := ir.RecordLayout(g.unit, st) + offsetOf := func(name string) int64 { + for i := range ml.Fields { + if ml.Fields[i].Field.Name == name { + return ml.Fields[i].Offset + } + } + return 0 + } + usesRegion := false + for _, f := range st.Fields { + if f.IsList() && listElementIsPointer(f) { + usesRegion = true + } + } + if !usesRegion { + g.pf(" (void) region; // a table element's and an entry's references resolve through their own bodies\n") + } + for _, f := range st.Fields { + if f.IsMap() { + entry := mapEntryOf(f) + el := ir.RecordLayout(g.unit, entry) + slot := offsetOf(f.Name) + g.pf(" { // %s\n", f.Name) + g.pf(" TableMapCursor<%s> cursor = TableMapOrder( ctx, value.%s );\n", entry.Name, f.Name) + g.pf(" if ( !cursor.ok ) { return false; }\n") + g.pf(" at = ( at + %d ) & ~(int64_t) %d; // at alignof( %s )\n", el.Align-1, el.Align-1, entry.Name) + g.pf(" uint8_t * array = extent + at;\n") + g.pf(" at += (int64_t) cursor.count * %d; // the whole array FIRST\n", el.Size) + g.pf(" // the SIXTEEN BYTES of the slot: the self-relative delta, then the count\n") + g.pf(" table_cook_put( record + %d, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + %d ) ) : 0, 8, order );\n", slot, slot) + g.pf(" table_cook_put( record + %d, (uint64_t) (uint32_t) cursor.count, 4, order );\n", slot+8) + g.pf(" for ( int32_t i = 0; i < cursor.count; i++ )\n {\n") + g.pf(" %s\n", g.cookBodyCall(entry, fmt.Sprintf("array + i * %d", el.Size), "*cursor[i]")) + g.pf(" }\n") + if g.hasExtent(entry) { + g.pf(" for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order\n {\n") + g.pf(" if ( !%sCookExtent( ctx, region, extent, at, array + i * %d, *cursor[i], order ) ) { TableMapRelease( cursor ); return false; }\n", entry.Name, el.Size) + g.pf(" }\n") + } + g.pf(" TableMapRelease( cursor );\n }\n") + continue + } + if f.IsList() { + elem := g.listElementType(f) + size, align := ir.ListElementLayout(g.unit, f) + slot := offsetOf(f.Name) + g.pf(" { // %s: an unbounded array\n", f.Name) + g.pf(" TableListCursor<%s> cursor = TableListElements( ctx, value.%s );\n", elem, f.Name) + g.pf(" if ( !cursor.ok ) { return false; }\n") + g.pf(" at = ( at + %d ) & ~(int64_t) %d; // at alignof( %s )\n", align-1, align-1, elem) + g.pf(" uint8_t * array = extent + at;\n") + g.pf(" at += (int64_t) cursor.count * %d; // the whole array FIRST\n", size) + g.pf(" // the SIXTEEN BYTES of the slot: the self-relative delta, then the count\n") + g.pf(" table_cook_put( record + %d, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + %d ) ) : 0, 8, order );\n", slot, slot) + g.pf(" table_cook_put( record + %d, (uint64_t) (uint32_t) cursor.count, 4, order );\n", slot+8) + g.pf(" for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only\n {\n") + if listElementIsPointer(f) { + // the self-relative delta of §6.3 to the node the numbering reached, + // or a refusal for one it did not, exactly as a pointer field's slot + g.pf(" if ( !table_cook_ref( region, array + i * %d, (const void *) %sAt( ctx, cursor[i] ), order ) ) { return false; }\n", size, f.Type.Name) + } else { + g.emitCookWriteElement(f, fmt.Sprintf("array + i * %d", size), "cursor[i]", " ", "_"+f.Name) + } + g.pf(" }\n") + if ref := listElementStruct(f); ref != nil && g.hasExtent(ref) { + g.pf(" for ( int32_t i = 0; i < cursor.count; i++ ) // then, element by element in index order\n {\n") + g.pf(" if ( !%sCookExtent( ctx, region, extent, at, array + i * %d, cursor[i], order ) ) { return false; }\n", ref.Name, size) + g.pf(" }\n") + } + g.pf(" }\n") + continue + } + if g.edgeOf(f) != edgeNested { + continue + } + ref, _ := f.Type.Ref.(*ir.Struct) + if ref == nil || !g.hasExtent(ref) { + continue + } + nested := offsetOf(f.Name) + if f.Array == ir.ArrayNone { + g.pf(" if ( !%sCookExtent( ctx, region, extent, at, record + %d, value.%s, order ) ) { return false; } // %s\n", ref.Name, nested, f.Name, f.Name) + continue + } + stride := cookElementBytes(g.unit, f) + base := "value." + f.Name + bound := fmt.Sprintf("%d", f.ArrayBound) + if f.KeyEnum != "" && st.IsTable { + base += ".slots" + } + if f.Array == ir.ArrayCounted { + // THE LIVE COUNT, as the extent walk counts it: a slot past the + // count is storage the walk does not reach, and a non-empty + // container in one was already refused there (§7.6) + bound = fmt.Sprintf("( value.%s_count < %d ? value.%s_count : %d )", f.Name, f.ArrayBound, f.Name, f.ArrayBound) + } + g.pf(" for ( int32_t i = 0; i < %s; i++ ) // %s\n {\n", bound, f.Name) + g.pf(" if ( !%sCookExtent( ctx, region, extent, at, record + %d + i * %d, %s[i], order ) ) { return false; }\n", ref.Name, nested, stride, base) + g.pf(" }\n") + } + g.pf(" return true;\n}\n\n") +} + +// emitCookNode emits `CookNode`: one NODE's record and then its own extent. +// A nested record's writer is the body alone, because a nesting's arrays are +// part of the HOLDER's extent and this walk already reached them. +func (g *tableGen) emitCookNode(st *ir.Struct) { + ml := ir.RecordLayout(g.unit, st) + record := cookAlignUp(ml.Size, ir.RegionAlignFloor) + g.pf("// %sCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9).\n", st.Name) + g.pf("template inline bool %sCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const %s & value, TableByteOrder order )\n{\n", st.Name, st.Name) + if g.isVar(st.Name) { + g.pf(" if ( !%sCookBody( ctx, region, at, value, order ) ) { return false; }\n", st.Name) + } else { + g.pf(" %sCookBody( at, value, order );\n", st.Name) + } + g.pf(" int64_t extent_at = 0;\n") + g.pf(" return %sCookExtent( ctx, region, at + %d, extent_at, at, value, order );\n}\n\n", st.Name, record) +} + +// emitCookExtentSurface emits the extent writer and the node writer for every +// closure member of a unit that declares a list or a map. +func (g *tableGen) emitCookExtentSurface(members []*ir.Struct) { + if !g.anyExtent { + return + } + var bodies []*ir.Struct + for _, st := range members { + if ir.RecordLayout(g.unit, st) != nil { + bodies = append(bodies, st) + } + } + for _, st := range bodies { + g.pf("%s;\n", g.cookExtentSignature(st)) + } + g.pf("\n") + for _, st := range bodies { + g.emitCookExtent(st) + } + for _, st := range bodies { + g.emitCookNode(st) + } +} + +// emitCookNodeBytes emits one node's whole span in a cooked region: its +// record at the region's alignment floor, plus the extent its lists and maps +// take (docs/SPEC-TABLES.md §2.8, §2.9, §7.2). +func (g *tableGen) emitCookNodeBytes(st *ir.Struct, ind, expr, onBad string) { + ml := ir.RecordLayout(g.unit, st) + if !g.anyExtent || !g.hasExtent(st) { + g.pf("%ssize = %d; node_align = %d;\n", ind, ml.Size, ml.Align) + return + } + g.pf("%s{\n", ind) + g.pf("%s const int64_t extent = %sExtent( ctx, %s );\n", ind, st.Name, expr) + g.pf("%s if ( extent < 0 ) { %s }\n", ind, onBad) + g.pf("%s size = %d + extent; node_align = %d;\n", ind, cookAlignUp(ml.Size, ir.RegionAlignFloor), ml.Align) + g.pf("%s}\n", ind) +} + +// onlyExtentFields reports a record whose every field is a list or a map, a +// cook body that writes the empty slots and reads nothing off the value, +// because the extent writer fills them. +func onlyExtentFields(st *ir.Struct) bool { + for _, f := range st.Fields { + if !f.IsMap() && !f.IsList() { + return false + } + } + return len(st.Fields) > 0 +} + +// placeColumn is the ONE function column an out-of-line array carries +// (docs/SPEC-TABLES.md §8.1, §16): the resolver the ONE text walk cannot spell +// for itself, because TableMap and TableList are types it has no +// name for. A map places one entry BY KEY and hands it back at its defaults, and a +// list ignores the key and appends. Empty in a unit that declares neither, so +// such a unit's descriptors are what they always were. +func (g *tableGen) placeColumn(f *ir.Field) string { + if !g.anyExtent { + return "" + } + if f.IsList() { + return g.listPlaceThunk(f) + ", " + } + if !f.IsMap() { + return "NULL, " + } + entry := mapEntryOf(f) + n := entry.Name + hold := fmt.Sprintf("TableMap<%s>", n) + var insert string + if mapKeyIsString(f) { + insert = fmt.Sprintf("[]( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * "+ + "{ if ( key == NULL || key_length > k%sKeyBound ) { return NULL; } "+ // KEYS NEVER CLAMP + "%s * placed = TableMapPlace( worker, *(%s *) slot, key ); "+ + "if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }", n, n, hold) + } else { + typ, _ := g.cppFieldType(ir.MapKeyField(f).Type) + insert = fmt.Sprintf("[]( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * "+ + "{ %s * placed = TableMapPlace( worker, *(%s *) slot, (%s) key_value ); "+ + "if ( placed != NULL ) { TableEntrySetKey( *placed, (%s) key_value ); } return (void *) placed; }", n, hold, typ, typ) + } + return insert + ", " +} + +// nodeStorageBody, nodeStorageArg and nodeStorageReader are the EXTRA +// parameters a node's storage takes where a list or a map rides in an extent +// (docs/SPEC-TABLES.md §2.8, §2.9): the record's body and the id table, from +// which the framing scan sums the arrays, and the refusal's reason. A root +// that can name no such record does not take them, so a unit without either +// construct emits the dispatch it always emitted. +func (g *tableGen) nodeStorageBody(anyExtent bool) string { + if anyExtent { + return "const uint8_t * body, " + } + return "" +} + +func (g *tableGen) nodeStorageTail(anyExtent bool) string { + if anyExtent { + return ", const TableIdTable * ids, TableRefuseReason & reason" + } + return "" +} + +func (g *tableGen) nodeStorageArg(root *ir.Struct) string { + if g.rootHasExtent(root) { + return "body, " + } + return "" +} + +func (g *tableGen) nodeStorageArgTail(root *ir.Struct) string { + if g.rootHasExtent(root) { + return ", &ids_table, reason" + } + return "" +} + +func (g *tableGen) nodeStorageReader(root *ir.Struct) string { + if g.rootHasExtent(root) { + return "r.buffer, " + } + return "" +} + +func (g *tableGen) nodeStorageReaderTail(root *ir.Struct) string { + if g.rootHasExtent(root) { + return ", r.ids, reason" + } + return "" +} + +// rootHasExtent reports whether any record one root's numbering can name holds +// a list or a map by value, which is what decides the signature above. +func (g *tableGen) rootHasExtent(root *ir.Struct) bool { + if !g.anyExtent { + return false + } + return slices.ContainsFunc(g.pointerReachable(root), g.hasExtent) +} + +// emitUnreachedExtentRefusal refuses an UNREACHED NON-EMPTY SLOT, the same +// refusal §7.6 gives a pointer in that position (docs/SPEC-TABLES.md §2.8, +// §2.9): a COUNTED array's slots past its live count are storage the walk does +// not reach, so a non-empty list or map in one names elements the region will +// not hold, and the write answers false with nothing partial written. +// +// The test is the extent itself: an empty container takes no bytes and +// advances the running offset by none, so a record whose extent measures ZERO +// is a record whose every by-value list and map is empty. A measure that +// refuses answers non-zero here too, and refusing on it is the same answer one +// level up. +func (g *tableGen) emitUnreachedExtentRefusal(f *ir.Field, ref *ir.Struct, subject string) { + if f.Array != ir.ArrayCounted { + return // every other array shape is reached whole + } + g.pf(" for ( int32_t i = %s.%s_count; i < %d; i++ ) // %s: the slots the walk does not reach (§7.6)\n {\n", + subject, f.Name, f.ArrayBound, f.Name) + g.pf(" if ( !TableExtentUnreachedEmpty( %sExtent( ctx, %s.%s[i] ) ) ) { return false; }\n", ref.Name, subject, f.Name) + g.pf(" }\n") +} diff --git a/internal/codegen/cpptable/json.go b/internal/codegen/cpptable/json.go index ef2dcd035..6cb8afcbf 100644 --- a/internal/codegen/cpptable/json.go +++ b/internal/codegen/cpptable/json.go @@ -26,19 +26,29 @@ import ( // with three stubs no field ever reaches, and a pointered unit answers with the // graph half — the builder's reader, the region's writer and the `&node` map — // which carries its own gate across the pointered units of the corpus. -func tableJsonWalk(pkg string, variable bool, anyMap bool) string { +func tableJsonWalk(pkg string, variable bool, anyMap bool, anyList bool) string { guard := strings.ToUpper(pkg) + "_SCHEMA_TABLE_JSON" adapters := tableJsonFixedAdapters if variable { adapters = tableJsonGraphSource } - // the MAP half, on the pointer half's own terms (docs/SPEC-TABLES.md §2.8): - // a map makes its holder variable-length, so the real half always follows - // the graph half it names, and a map-free unit carries the stub. + // the EXTENT accessors both out-of-line halves read, then the MAP half and + // the LIST half, each on the pointer half's own terms (docs/SPEC-TABLES.md + // §2.8, §2.9): either construct makes its holder variable-length, so a real + // half always follows the graph half it names, and a unit without the + // construct carries the stub. + extentAdapters := "" + if anyMap || anyList { + extentAdapters = tableJsonExtentAdapters + } mapAdapters := tableJsonNoMapAdapters if anyMap { mapAdapters = tableJsonMapAdapters } + listAdapters := tableJsonNoListAdapters + if anyList { + listAdapters = tableJsonListAdapters + } // The include guard is LOAD-BEARING in a .cpp, which is not where a reader // expects to find one — hence the comment riding with it. It is what lets // several same-package .cpp files be concatenated into one translation @@ -56,7 +66,9 @@ func tableJsonWalk(pkg string, variable bool, anyMap bool) string { tableJsonAdapterDeclarations + tableJsonWalkSource + "\n" + adapters + + extentAdapters + "\n" + mapAdapters + + "\n" + listAdapters + "\n} // namespace " + pkg + "\n\n#endif // " + guard + "\n" } @@ -86,21 +98,52 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); + +` + +// tableJsonExtentAdapters is what both out-of-line halves read off a slot +// (docs/SPEC-TABLES.md §7.2, §8.1): the sixteen bytes are an int64 +// self-relative reference to the array and the int32 count, the same two +// facts for a map and a list, so one pair of accessors serves both. Emitted +// only into a unit that declares either. +const tableJsonExtentAdapters = ` +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} ` // tableJsonFixedAdapters answers for a unit that declares no pointer: no field @@ -139,12 +182,15 @@ inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char *, int32_t d // zero-cost property (§2.2) holding for the text form. const tableJsonMapAdapters = `// ---- json map walk: begin ---- -inline bool TableJsonIsMap( const TableFieldInfo * f ) { return f->entry != NULL; } +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} // the entry's two rows: fields[0] IS the key and fields[1] IS the value, which // is what makes a user's own table of pairs the same bytes (§2.8) -inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->entry->fields[0]; } -inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->entry->fields[1]; } +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } @@ -204,16 +250,17 @@ inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const // A region holds them in that order already, so this is the array in place. inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) { - const int32_t count = f->map_count( slot ); + const int32_t count = TableJsonExtentCount( slot ); if ( count == 0 ) { out.raw( "{}", 2 ); return true; } const TableFieldInfo * key = TableJsonMapKeyField( f ); const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); out.put( '{' ); for ( int32_t i = 0; i < count; i++ ) { if ( i > 0 ) { out.put( ',' ); } out.line( depth + 1 ); - const void * entry = f->map_at( slot, i ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); TableJsonWriteMapKey( out, entry, key ); out.raw( ": ", 2 ); if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } @@ -302,15 +349,15 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf } if ( !fits ) { in.report->kind_mismatch++; place = false; } } - const int32_t before = f->map_count( (const void *) slot ); - void * entry = place ? f->map_insert( *graph->worker, slot, token, token_length, key_value ) : NULL; + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; if ( place && entry == NULL ) { // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the // wire's rule, because a clamped key is a merged entry (§2.8). in.report->clamped++; } - else if ( entry != NULL && f->map_count( (const void *) slot ) == before ) + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) { in.report->duplicate++; // last-wins, the object rule inside the map } @@ -385,7 +432,139 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, } ` +// tableJsonListAdapters is the list half (docs/SPEC-TABLES.md §2.9, §16): the +// JSON array a bounded array already takes, with every element the text +// carries read, because there is no bound to drop a tail against. It is +// emitted ONLY into a unit that declares an unbounded array, and it is one +// half, the same bytes in every list-bearing .cpp, which is the generic-walk +// gate's property holding for the construct. +const tableJsonListAdapters = `// ---- json list walk: begin ---- + +// an unbounded array is the out-of-line array that is not a map (§8.1) +inline bool TableJsonIsList( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && !TableJsonIsMap( f ); +} + +// ToJson WRITES THE ELEMENTS IN INDEX ORDER, which is the only order there is, +// so unpack then pack is byte-stable without a rule of its own (§2.9, §17.2). +// A region holds the array in place, so this steps it at the descriptor's pitch. +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + const uint8_t * elements = TableJsonExtentElements( slot ); + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const uint8_t * element = elements + (int64_t) i * f->elem_size; + if ( f->kind == 17 ) + { + // a []*T's elements take the pointer row (§16.7): the pointee's + // object in place, null, or ` + "`&node`" + ` for a shared one + if ( !TableJsonWritePointer( out, element, f, depth + 1 ) ) { return false; } + } + else if ( !TableJsonWriteScalar( out, element, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; +} + +// FromJson READS EVERY ELEMENT THE TEXT CARRIES, appending each through the +// descriptor's place resolver: ` + "`[]`" + ` is an empty list, and null is +// kind_mismatch, the array row's own rule (§16.2). LAST WINS holds for a +// repeated key: the list goes back to EMPTY before this occurrence's elements +// land, the builder's storage being reclaimed at reset (§2.9). +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) slot, 8, 0 ); + TableJsonSetRaw( (uint8_t *) slot + 8, 4, 0 ); + const char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + void * element = f->place( *graph->worker, slot, NULL, 0, 0 ); + if ( element == NULL ) + { + // NOT ADDED: the arena could not carve another segment, or the + // count met the int32 cap. The text cannot be placed whole, and + // the read stops where §16.1's rule stops it. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( f->kind == 17 ) + { + // an element of a []*T (§2.9): null is a null slot, an object is the + // pointee in place or an ` + "`&node`" + ` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) element, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, element, f, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + // the wrong shape for the element kind: the slot keeps its + // defaults and the event counts, the array row's rule (§16.2) + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, element, f, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json list walk: end ---- +` + +// tableJsonNoListAdapters answers for a unit that declares no UNBOUNDED ARRAY: +// no field is one, so the two slot adapters are never reached, and a unit +// with no list carries no list machinery (§2.2, §2.9). +const tableJsonNoListAdapters = `// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} +` + // emitJsonDeclarations puts one closure member's text-form surface in the + // HEADER: three declarations and nothing else. The definitions, and the walker // they call, live in the generated Table.cpp — so a translation unit // that includes the header to use the wire codecs or the descriptors pays @@ -1211,6 +1390,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -2218,6 +2401,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; diff --git a/internal/codegen/cpptable/lists.go b/internal/codegen/cpptable/lists.go new file mode 100644 index 000000000..fc8bf25d8 --- /dev/null +++ b/internal/codegen/cpptable/lists.go @@ -0,0 +1,852 @@ +// Unbounded arrays in the C++ reference (docs/SPEC-TABLES.md §2.9): the +// runtime a `[]T` field's storage, its builder and its const surface are +// spelled in, and the per-field codecs the generated code hangs off it. +// +// An unbounded array is §2.8's map with the KEY and the SORT taken out: a +// counted array whose count the data decides, its elements by-value records +// inside the holder's node extent. So nothing here is a new wire construct, +// and most of what a map needed is not needed: no entry table, no order, no +// key compare, no ascending check and no duplicate rule. What the runtime +// adds is the reference-and-count slot, a builder that appends into segments +// that never move, and a cursor the four writing walks read in INDEX order +// without allocating. +package cpptable + +import ( + "fmt" + "strings" + + "github.com/mas-bandwidth/schema/v2/ir" +) + +// unitHasList reports whether any closure member declares a `[]T`. It gates +// the list runtime: not one symbol of it appears in a list-free unit's +// generated header (docs/SPEC-TABLES.md §2.2, §2.9). +func unitHasList(u *ir.Unit, closure map[string]bool) bool { + for name := range closure { + st := memberOf(u, name) + if st == nil { + continue + } + for _, f := range st.Fields { + if f.IsList() { + return true + } + } + } + return false +} + +// listFieldsOf lists one member's unbounded arrays in declaration order. +func listFieldsOf(st *ir.Struct) []*ir.Field { + var out []*ir.Field + for _, f := range st.Fields { + if f.IsList() { + out = append(out, f) + } + } + return out +} + +// listVerb spells one of a list field's claimed surface names on its holder: +// `` followed by the verb (docs/SPEC-TABLES.md §2.9, §11). +func listVerb(owner string, f *ir.Field, verb string) string { + return owner + ir.GoExportName(f.Name) + verb +} + +// listElementIsPointer reports a `[]*T`: the elements are pointer SLOTS, so +// the builder's Add hands back the slot at null and the const form answers the +// resolved `const T *` (docs/SPEC-TABLES.md §2.9). +func listElementIsPointer(f *ir.Field) bool { return f.Type.Pointer } + +// listTypeArg is the type argument the storage names: the element type for a +// `[]T`, and `T *` for a `[]*T`, which selects the pointer specialization. +func (g *tableGen) listTypeArg(f *ir.Field) string { + if listElementIsPointer(f) { + return f.Type.Name + " *" + } + typ, _ := g.cppFieldType(f.Type) + return typ +} + +// listStorageType is the C++ spelling of a list field's storage. +func (g *tableGen) listStorageType(f *ir.Field) string { + return fmt.Sprintf("TableList<%s>", g.listTypeArg(f)) +} + +// listElementType is the C++ type ONE ELEMENT is stored as: a TableRef for a +// `[]*T`, the element's own type otherwise. +func (g *tableGen) listElementType(f *ir.Field) string { + if listElementIsPointer(f) { + return "TableRef" + } + typ, _ := g.cppFieldType(f.Type) + return typ +} + +// listElementStruct is the element's table when the element is one, else nil. +func listElementStruct(f *ir.Field) *ir.Struct { + if listElementIsPointer(f) || f.Type.Kind != ir.TNamed { + return nil + } + ref, _ := f.Type.Ref.(*ir.Struct) + return ref +} + +// listElementWireKind is the element kind the list's kind 14 body carries: +// kind 17 for a pointer element, the element's own kind otherwise (§2.9, §3). +func listElementWireKind(f *ir.Field) int { + if listElementIsPointer(f) { + return tkNodeIndex + } + return ir.TableWireElemKind(f) +} + +// listElementFloor is the smallest wire footprint ONE element commands +// (docs/SPEC-TABLES.md §4.2, §6.5): a scalar its own width, a reference-shaped +// element (a pointer's node index, an enum's variant reference, a union's arm +// reference) one byte, a table element its own `L` and its terminator. It is +// what bounds the N a list's `L` can carry, and therefore what a LoadMeasure +// may be asked for. +func listElementFloor(f *ir.Field) int { + if listElementIsPointer(f) { + return 1 + } + switch kind := ir.TableWireElemKind(f); kind { + case tkTable: + return 2 + case tkEnum, tkUnion: + return 1 + default: + return tableKindWidth(kind) + } +} + +// ---- the runtime (docs/SPEC-TABLES.md §2.9) ---- + +// tableListRuntime is the list half of the variable-length runtime: the +// storage type and its const surface, the builder's head and segments, the +// index-order cursor the four writing walks read, and the load side's fill. +// It is emitted only into a unit that declares an unbounded array. +func tableListRuntime(pkg string) string { + guard := strings.ToUpper(pkg) + "_SCHEMA_TABLE_LIST" + return `#ifndef ` + guard + ` +#define ` + guard + ` + +namespace ` + pkg + ` { + +// ---- an UNBOUNDED ARRAY: a counted array whose count the data decides (§2.9) ---- +// +// On the wire, in a region and in a cook a list is the kind 14 body a [..N]T +// writes, its elements by-value records inside the holder's node extent. What +// this adds is the slot, a builder that appends into segments that never +// move, and a const surface that indexes and iterates in place. There is no +// sort, no key and no lookup: the order is INSERTION order, and it is +// identity the way position is identity in a fixed array. + +// elements carved from ONE call to the allocator pair. A new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableListSegmentElements = 32; + +// THE ELEMENT STORAGE: T itself, and a TableRef slot for a []*T, whose +// elements are references exactly as a pointer field's slot is (§2.1) +template struct TableListStorage { typedef T Element; }; +template struct TableListStorage { typedef TableRef Element; }; + +// WHAT THE CONST FORM ANSWERS: the element by reference, and on a []*T the +// RESOLVED pointer, one add on the self-relative delta, NULL for a null slot, +// exactly as At answers it (§6.2, §6.3) +template struct TableListConst +{ + typedef const T & Result; + static Result At( const T * element ) { return *element; } +}; +template struct TableListConst +{ + typedef const T * Result; + static Result At( const TableRef * element ) + { + return element->value != 0 ? (const T *) ( (const uint8_t *) element + element->value ) : NULL; + } +}; + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.9, §7.2) ---- +// +// An int64 self-relative reference to the element array and an int32 count, +// then padding to eight. The reference is a TableRef like a pointer's: in the +// arena it names the builder's HEAD, in a region it is the delta from the slot +// to the first element, and 0 is the empty list in both. It is the map's slot +// exactly, because it is the same two facts. +template struct TableList +{ + typedef typename TableListStorage::Element Element; + + TableRef elements; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Element * Elements() const + { + return elements.value != 0 ? (const Element *) ( (const uint8_t *) &elements + elements.value ) : NULL; + } + int32_t size() const { return count; } + + // INDEXING IS BOUNDS-CHECKED IN EVERY BUILD (§2.4, §2.9): the extent is a + // number that CAME FROM A FILE, so an index past it is not a mistake a + // release build gets to make cheaply. There is no undefined-behavior path + // here in any configuration. The assert carries the message where a + // debugger can read it and NDEBUG removes that. The fatal is what stands + // after it. Both go through the hooks: define schema_assert and + // schema_fatal and this refusal lands in your own handler. + void RefuseIndex( int32_t index ) const + { + if ( (uint32_t) index >= (uint32_t) count ) + { + schema_assert( false && "an unbounded array is indexed inside its count, which came from a file" ); + schema_fatal(); + } + } + typename TableListConst::Result operator[]( int32_t index ) const + { + RefuseIndex( index ); + return TableListConst::At( Elements() + index ); + } + + // ---- iteration: INDEX order, the element and no key ---- + // + // It carries no iterator_traits, for the reason TableKeyed's does not + // (§13.9). + struct ConstIterator + { + const Element * at; + typename TableListConst::Result operator*() const { return TableListConst::At( at ); } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Elements() }; } + ConstIterator end() const { return ConstIterator{ Elements() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.9, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first element is added. Each +// segment is a fixed number of elements carved from one call to the allocator +// pair. An element's address is stable for the arena's life, so a T * handed +// back by Add stays valid while other elements arrive. +struct TableListHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an Add appends into + int32_t live; + int32_t dead; +}; + +template struct TableListSegment +{ + TableRef next; + int32_t used; // elements carved from this segment + int32_t padding; + uint32_t dead[ ( kTableListSegmentElements + 31 ) / 32 ]; // Erase marks one bit, never the element + Element elements[ kTableListSegmentElements ]; +}; + +inline bool TableListSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// the head, allocated when the first element is added +template +inline TableListHead * TableListReach( TableWorker & worker, TableList & list ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( list.elements.value != 0 ) { return (TableListHead *) TableArenaAt( *worker.arena, (uint32_t) list.elements.value ); } + uint32_t at = 0; + TableListHead * head = (TableListHead *) worker.AllocRaw( (int64_t) sizeof( TableListHead ), (int64_t) alignof( TableListHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + list.elements.value = (int64_t) at; + return head; +} + +// one element's storage, appended: the current segment when it has room, a +// new one carved from one call to the pair when it does not. NULL means NOT +// ADDED: an arena that cannot carve another segment, or a count at the int32 +// cap (§2.2, §2.9). +template +inline typename TableList::Element * TableListAppend( TableWorker & worker, TableListHead * head, TableList & list ) +{ + typedef typename TableList::Element Element; + if ( list.count >= INT32_MAX ) { return NULL; } // the int32 storage cap + TableListSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableListSegmentElements ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableListSegment *) worker.AllocRaw( (int64_t) sizeof( TableListSegment ), (int64_t) alignof( TableListSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableListSegment * previous = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Element * element = segment->elements + segment->used; + segment->used++; + head->live++; + list.count++; + return element; +} + +// ADD, whole: the head, the append, and the element at its declared defaults +// (§2.9). The text form's placement is this same call, because a list has no +// key to place under (§16). +template +inline typename TableList::Element * TableListPlace( TableWorker & worker, TableList & list ) +{ + typedef typename TableList::Element Element; + TableListHead * head = TableListReach( worker, list ); + if ( head == NULL ) { return NULL; } + Element * element = TableListAppend( worker, head, list ); + if ( element == NULL ) { return NULL; } + new ( element ) Element(); // value-init: the declared defaults, and null for a slot + return element; +} + +// ERASE, ADDRESSED BY THE POINTER (§2.9): the element Add handed back is the +// handle, because a list has no key and the address is the one thing the +// builder promises never moves (§6.4). It marks the element DEAD, one bit in +// the segment's slot and not in the element storage, and decrements the live +// count. False when the pointer is not this list's. Its storage is reclaimed +// at RESET and never reused mid-build, the map's rule for the map's reason. +template +inline bool TableListErase( TableArena & arena, TableList & list, const typename TableList::Element * element ) +{ + typedef typename TableList::Element Element; + if ( list.elements.value == 0 || element == NULL ) { return false; } + TableListHead * head = (TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableListSegment * segment = (TableListSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + if ( element >= segment->elements && element < segment->elements + segment->used ) + { + const int32_t i = (int32_t) ( element - segment->elements ); + if ( TableListSegmentDead( segment->dead, i ) ) { return false; } // already erased + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + list.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INDEX order, live elements only (§2.9) ---- +template struct TableListEach +{ + typedef typename TableList::Element Element; + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableListSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableListSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + Element * operator*() const { return segment->elements + index; } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableListEach TableListEachOf( const TableArena & arena, const TableList & list ) +{ + TableListEach each = { &arena, TableRef() }; + if ( list.elements.value != 0 ) + { + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + each.first = head->first; + } + return each; +} + +// ---- the INDEX-ORDER CURSOR the four writing walks read (§2.9) ---- +// +// Measure, Save, Lock and Cook each visit a list's live elements in the order +// they were added, and they allocate nothing to do it: a region's cursor is +// the array in place, and the builder's walks the segment chain. Indexing the +// builder's form is SEQUENTIAL by construction, every walk steps i, i + 1, +// i + 2, so the cursor remembers where the last access landed and moves one +// live slot per step. An access behind the memo restarts from the first +// segment, which no walk here does. +template struct TableListCursor +{ + const Element * elements = NULL; // the region's form: the array in place + const TableArena * arena = NULL; // the builder's form: the segments + TableRef first; + int32_t count = 0; + bool ok = false; + // the memo: the segment and slot the last access landed on, and the live + // index that slot holds + mutable const TableListSegment * segment = NULL; + mutable int32_t within = -1; + mutable int32_t logical = -1; + + const Element * At( int32_t index ) const + { + if ( elements != NULL ) { return elements + index; } + if ( segment == NULL || index < logical ) + { + segment = first.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL; + within = -1; + logical = -1; + } + while ( logical < index ) + { + for ( ;; ) + { + within++; + while ( segment != NULL && within >= segment->used ) + { + segment = segment->next.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + within = 0; + } + if ( segment == NULL ) { return NULL; } // the slot and the head disagree + if ( !TableListSegmentDead( segment->dead, within ) ) { break; } + } + logical++; + } + return segment->elements + within; + } + const Element & operator[]( int32_t index ) const { return *At( index ); } +}; + +// the REGION form: the array is the cursor +template +inline TableListCursor::Element> TableListElements( const TableRegionCtx &, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.elements = list.Elements(); + cursor.count = list.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: the live elements out of the segment chain, in the +// order they were added. A dead element costs nothing on any wire (§2.9). +template +inline TableListCursor::Element> TableListElements( const TableArena & arena, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.arena = &arena; + cursor.count = list.count; + if ( list.elements.value == 0 || list.count <= 0 ) { cursor.ok = list.count == 0; cursor.count = 0; return cursor; } + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + if ( head->live != list.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + cursor.first = head->first; + cursor.ok = true; + return cursor; +} + +template +inline TableListCursor::Element> TableListElements( const TableArenaCtx & ctx, const TableList & list ) +{ + return TableListElements( *ctx.arena, list ); +} + +// ---- the LOAD side: where a decoded element lands (§2.9) ---- +// +// The same two shapes the map's fill takes, because the decoder above them +// cannot tell which it has: a REGION carves the element array out of the +// holder node's own extent, PRE-ORDER, and the TOOL's path appends into the +// builder's arena. +template struct TableListFill +{ + typedef typename TableList::Element Element; + TableList * list = NULL; + Element * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; + bool refused = false; // a count above the int32 cap on the tool's path: LoadBuilder answers NULL +}; + +template +inline TableListFill TableListFillBegin( const TableNodeMap & nodes, TableList & list, uint64_t n ) +{ + typedef typename TableList::Element Element; + TableListFill fill; + fill.list = &list; + list.elements.value = 0; + list.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( n > (uint64_t) INT32_MAX ) + { + // A COUNT ABOVE THE int32 STORAGE CAP (§2.2, §2.9): into a region it was + // refused by LoadMeasure before this ran, and into a builder it is the + // refusal LoadBuilder answers NULL for, moving no counter + fill.refused = nodes.carve->worker != NULL; + return fill; + } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Element ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Element ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Element *) base; + fill.capacity = (int32_t) n; + list.elements.value = (int64_t) ( base - (const uint8_t *) &list.elements ); + fill.ok = true; + return fill; +} + +// the next slot, at the element's declared defaults. NULL when the arena +// could not carve, which the decoder reports as framing damage +template inline typename TableList::Element * TableListFillNext( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count >= fill.capacity ) { return NULL; } + Element * element = fill.array + fill.list->count; + new ( element ) Element(); + fill.list->count++; + return element; + } + return TableListPlace( *fill.worker, *fill.list ); +} + +// A SLOT WHOSE ELEMENT NEVER LANDED is given back (§2.9, §4): the array keeps +// what it decoded, and an element whose own framing gave out before one byte +// of it decoded was not decoded. The region's form uncounts it, and the builder's +// marks it dead, which is what the storage rule allows mid-build. +template inline void TableListFillDrop( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count > 0 ) { fill.list->count--; } + return; + } + if ( fill.list->elements.value == 0 ) { return; } + TableListHead * head = (TableListHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.list->elements.value ); + if ( head->last.value == 0 ) { return; } + TableListSegment * segment = (TableListSegment *) TableArenaAt( *fill.worker->arena, (uint32_t) head->last.value ); + if ( segment->used <= 0 ) { return; } + const int32_t i = segment->used - 1; + if ( TableListSegmentDead( segment->dead, i ) ) { return; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + fill.list->count--; +} + +// an EMPTY list's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableListFillEnd( TableListFill & fill ) +{ + if ( fill.array != NULL && fill.list->count == 0 ) { fill.list->elements.value = 0; } +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.9, §6.5) ---- +// +// N x sizeof( T ) rounded to alignof( T ), AT EVERY DEPTH. N is framing and +// not a value, so this reads no field: it walks the list's own header and, +// where a table element holds a list or a map of its own, the elements' +// headers under it. Every -1 carries its REASON (§6.5): the int32 cap first, +// because a count past it cannot fit any body, and then the body's own L. +inline bool TableListWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t elem_size, int64_t elem_align, uint8_t elem_kind, int64_t elem_floor, + TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != elem_kind ) { return true; } // another element kind: §4's ordinary kind mismatch, the field reads empty + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; } // an N the list's L cannot carry + at = ( at + elem_align - 1 ) & ~( elem_align - 1 ); + at += (int64_t) n * elem_size; + if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace ` + pkg + ` + +#endif // ` + guard + ` +` +} + +// ---- the list's wire codecs (docs/SPEC-TABLES.md §2.9, §3) ---- +// +// A list rides as the kind 14 ARRAY a bounded array rides as, over the +// element's own kind, so the element measure, write and read are the bounded +// array's own emitters over a cursor rather than over inline storage. What +// is the list's own is the cursor and the fill. + +// emitListMeasureField adds one list field to a body's measured size. +func (g *tableGen) emitListMeasureField(f *ir.Field) { + id := ir.TableFieldWireId(f) + cursor := "cursor_" + f.Name + g.pf(" {\n") + g.pf(" // %s: a kind %d array of kind %d elements, INDEX order (§2.9)\n", f.Name, tkArray, listElementWireKind(f)) + g.pf(" TableListCursor<%s> %s = TableListElements( ctx, value.%s );\n", g.listElementType(f), cursor, f.Name) + g.pf(" if ( !%s.ok ) { return -1; } // the slot and the head disagree\n", cursor) + g.pf(" if ( %s.count > 0 ) // an EMPTY list elides, the by-value rule (§3)\n {\n", cursor) + g.pf(" const uint64_t ref_%s = %s;\n", f.Name, g.wireRef(id)) + g.pf(" int64_t body_%s = 0;\n", f.Name) + g.emitArrayBodyMeasure(f, listElementWireKind(f), "body_"+f.Name, cursor+".count", cursor+"[%s]", " ", "return -1;", "_"+f.Name) + g.pf(" bytes += TableLebBytes( ref_%s ) + 1 + %s;\n", f.Name, framed("body_"+f.Name)) + g.pf(" }\n") + g.pf(" }\n") +} + +// emitListWriteField writes one list field: the array framing, then the live +// elements in INDEX order, dead elements dropped. +func (g *tableGen) emitListWriteField(f *ir.Field) { + id := ir.TableFieldWireId(f) + cursor := "cursor_" + f.Name + g.pf(" {\n") + g.pf(" TableListCursor<%s> %s = TableListElements( ctx, value.%s ); // %s\n", g.listElementType(f), cursor, f.Name, f.Name) + g.pf(" if ( !%s.ok ) { return false; }\n", cursor) + g.pf(" if ( %s.count > 0 ) // an EMPTY list elides, the by-value rule (§3)\n {\n", cursor) + g.pf(" const uint64_t ref_%s = %s;\n", f.Name, g.wireRef(id)) + g.pf(" int64_t body_%s = 0;\n", f.Name) + g.emitArrayBodyMeasure(f, listElementWireKind(f), "body_"+f.Name, cursor+".count", cursor+"[%s]", " ", "return false;", "_"+f.Name) + g.pf(" w.putleb( ref_%s ); w.put8( %d ); w.putleb( (uint64_t) body_%s ); // %s\n", f.Name, tkArray, f.Name, f.Name) + g.emitArrayBodyWrite(f, listElementWireKind(f), cursor+".count", cursor+"[%s]", " ", "_"+f.Name) + g.pf(" }\n") + g.pf(" }\n") +} + +// emitListReadField decodes one list field, and every reader rule §2.9 +// states lands here: the element kind checked against the reader's +// declaration, the count taken as the data's with nothing to clamp it +// against, the elements bounded by the body's own L, and a slot whose element +// never landed given back. +func (g *tableGen) emitListReadField(f *ir.Field) { + ind := " " + elemKind := listElementWireKind(f) + g.pf("%suint64_t body_len = 0;\n", ind) + g.pf("%sif ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; }\n", ind) + g.pf("%sint64_t body_end = r.offset + (int64_t) body_len;\n", ind) + g.pf("%s// A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps\n", ind) + g.pf("%s// the value it has, no counter is raised, and the walk continues past L.\n", ind) + g.pf("%sif ( body_len >= 2 )\n%s{\n", ind, ind) + g.pf("%s uint8_t elem_kind = r.get8();\n", ind) + g.pf("%s uint64_t count = 0;\n", ind) + g.pf("%s const bool counted_ok = r.getleb( count );\n", ind) + g.pf("%s if ( !counted_ok ) { r.report->malformed = true; }\n", ind) + g.pf("%s // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's\n", ind) + g.pf("%s // element-kind rule: the field reads EMPTY and one kind_mismatch counts\n", ind) + g.pf("%s else if ( elem_kind != %d ) { r.report->kind_mismatch++; r.offset = body_end; break; }\n", ind, elemKind) + g.pf("%s else\n%s {\n", ind, ind) + g.pf("%s // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped\n", ind) + g.pf("%s // cannot fire on it. A count above the int32 storage cap is the\n", ind) + g.pf("%s // fill's refusal, and it moves no counter.\n", ind) + g.pf("%s TableListFill<%s> fill = TableListFillBegin( nodes, value.%s, count );\n", ind, g.listTypeArg(f), f.Name) + g.pf("%s if ( fill.refused ) { nodes.refused = true; return false; }\n", ind) + g.pf("%s if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; }\n", ind) + g.pf("%s // elements are BOUNDED by the field body: a count the length cannot\n", ind) + g.pf("%s // cover keeps the decoded prefix, flags malformed, and the parent\n", ind) + g.pf("%s // continues at the next field\n", ind) + g.pf("%s TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids );\n", ind) + g.pf("%s for ( uint64_t i = 0; i < count; i++ )\n%s {\n", ind, ind) + g.pf("%s %s * slot = TableListFillNext( fill );\n", ind, g.listElementType(f)) + g.pf("%s if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve\n", ind) + g.pf("%s bool landed = false;\n", ind) + g.pf("%s do\n%s {\n", ind, ind) + g.emitTableReadElementInto(f, elemKind, "( *slot )", ind+" ", "sub", "_"+f.Name) + g.pf("%s landed = true;\n", ind) + g.pf("%s } while ( 0 );\n", ind) + g.pf("%s if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded\n", ind) + g.pf("%s }\n", ind) + g.pf("%s TableListFillEnd( fill );\n", ind) + g.pf("%s }\n", ind) + g.pf("%s}\n", ind) + g.pf("%sr.offset = body_end; // excess bytes and slack skip via the length\n", ind) +} + +// ---- the three walks at a list (docs/SPEC-TABLES.md §2.9, §3.1) ---- +// +// A list is a BY-VALUE EDGE of the ONE declaration-order walk: it is reached +// at its field's position, its elements are visited in INDEX ORDER, and each +// element is descended for the pointer slots inside it before the next +// element is reached. A []*T declared before a pointer field therefore +// reaches a shared node FIRST and numbers it first. The rule is the walk's, +// not the list's. + +// emitListEdge opens one list's cursor under the walk's subjects and visits +// each element through the emitter's own pointer or descend visitor: the +// element IS the pointer slot on a []*T, and a variable table to descend on a +// []T. The pack's twin is the array ExtentPack already placed in the node's +// extent, reached from the write subject's slot. +func (g *tableGen) emitListEdge(f *ir.Field, v edgeVisitor, onBad string) { + elem := g.listElementType(f) + cursor := "cursor_" + f.Name + g.pf(" { // %s: a by-value edge, elements in INDEX order (§2.9, §3.1)\n", f.Name) + g.pf(" TableListCursor<%s> %s = TableListElements( ctx, %s.%s );\n", elem, cursor, v.read, f.Name) + g.pf(" if ( !%s.ok ) { %s }\n", cursor, onBad) + placed := "" + if v.write != "" { + placed = "placed_" + f.Name + slot := v.write + "." + f.Name + g.pf(" %s * %s = (%s *) ( %s.elements.value != 0 ? ( (uint8_t *) &%s.elements + %s.elements.value ) : NULL );\n", + elem, placed, elem, slot, slot, slot) + } + g.pf(" for ( int32_t i = 0; i < %s.count; i++ )\n {\n", cursor) + expr := edgeExpr{Src: cursor + "[i]"} + if placed != "" { + expr.Dst = placed + "[i]" + } + saved := g.indent + g.indent = saved + " " + if listElementIsPointer(f) { + v.pointer(f, expr) + } else { + v.descend(f.Type.Name, expr, " ") + } + g.indent = saved + g.pf(" }\n }\n") +} + +// ---- the builder's three (docs/SPEC-TABLES.md §2.9) ---- +// +// FREE FUNCTIONS taking the worker or the arena, as Emplace and the map's +// five are. Add takes the WORKER because it may allocate a segment. Each and +// Erase take the arena because neither ever does. +func (g *tableGen) emitListBuilderSurface(owner *ir.Struct, f *ir.Field) { + hold := fmt.Sprintf("TableList<%s> & list", g.listTypeArg(f)) + elem := g.listElementType(f) + g.pf("// ---- %s.%s: the builder's three (§2.9) ----\n\n", owner.Name, f.Name) + if listElementIsPointer(f) { + g.pf("// ADD: the element is appended and handed back to fill. On a []*T that is\n") + g.pf("// the SLOT at null, which %sEmplace fills as it fills any pointer slot,\n", f.Type.Name) + g.pf("// and a second slot may hold the same reference: two slots, one node.\n") + } else { + g.pf("// ADD: the element is appended at its declared defaults and handed back to\n") + g.pf("// fill. Nothing ever moves (§6.4), so the pointer stays valid while other\n") + g.pf("// elements arrive.\n") + } + g.pf("// NULL means NOT ADDED: an arena that cannot carve another segment, or a\n") + g.pf("// count at the int32 cap. A caller that needs the reason checks size().\n") + g.pf("inline %s * %s( TableWorker & worker, %s )\n{\n", elem, listVerb(owner.Name, f, "Add"), hold) + g.pf(" return TableListPlace( worker, list );\n}\n\n") + g.pf("// ERASE, by the element's own pointer: marks it DEAD, one bit in the\n") + g.pf("// segment's slot and not in the element storage. False when the pointer is\n") + g.pf("// not this list's. Storage is held until the builder resets. INDICES ARE\n") + g.pf("// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save.\n") + g.pf("inline bool %s( TableArena & arena, %s, const %s * element )\n{\n", listVerb(owner.Name, f, "Erase"), hold, elem) + g.pf(" return TableListErase( arena, list, element );\n}\n\n") + g.pf("// EACH on the builder: INDEX order, live elements only, yielding the\n") + g.pf("// element Add handed back.\n") + g.pf("inline TableListEach<%s> %s( const TableArena & arena, const TableList<%s> & list )\n{\n", g.listTypeArg(f), listVerb(owner.Name, f, "Each"), g.listTypeArg(f)) + g.pf(" return TableListEachOf( arena, list );\n}\n\n") +} + +// emitListBuilderSurfaces emits every list field's builder three, after the +// arena and list runtimes they are spelled in terms of. +func (g *tableGen) emitListBuilderSurfaces(members []*ir.Struct) { + if !g.anyList { + return + } + for _, st := range members { + for _, f := range listFieldsOf(st) { + g.emitListBuilderSurface(st, f) + } + } +} + +// emitListAlignAsserts holds every list element to the arena's alignment +// (docs/SPEC-TABLES.md §2.9): a node's extent begins at the arena's alignment +// and nothing inside it can ask for more. +func (g *tableGen) emitListAlignAsserts(members []*ir.Struct) { + if !g.anyList { + return + } + for _, st := range members { + for _, f := range listFieldsOf(st) { + g.pf("static_assert( alignof( %s ) <= kTableAlign, \"%s.%s: an unbounded array's element alignment must fit the arena's\" );\n", g.listElementType(f), st.Name, f.Name) + } + } + g.pf("\n") +} + +// listPlaceThunk is the descriptor's PLACE column for a list field +// (docs/SPEC-TABLES.md §8.1, §16): the one resolver the text walk cannot spell +// for itself, because TableList is a type it has no name for. The key +// arguments are the map's and a list ignores them: it appends. +func (g *tableGen) listPlaceThunk(f *ir.Field) string { + return fmt.Sprintf("[]( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(%s *) slot ); }", g.listStorageType(f)) +} diff --git a/internal/codegen/cpptable/maps.go b/internal/codegen/cpptable/maps.go index 4c258e8f9..9e1809a19 100644 --- a/internal/codegen/cpptable/maps.go +++ b/internal/codegen/cpptable/maps.go @@ -10,7 +10,6 @@ package cpptable import ( "fmt" - "slices" "strings" "github.com/mas-bandwidth/schema/v2/ir" @@ -525,12 +524,6 @@ inline TableMapEach TableMapEachOf( const TableArena & arena, const Table return each; } -// AN UNREACHED SLOT MUST HOLD NO MAP WITH ENTRIES IN IT (§2.8, §7.6). An empty -// map takes no bytes, so a record whose extent measures ZERO is a record whose -// every by-value map is empty; a measure that REFUSED answers non-zero here -// too, and refusing on it is the same answer one level up. -inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } - // ---- the LOAD side: where a decoded entry lands (§2.8) ---- // // THE READER TRUSTS NOTHING and spends one compare per entry. Every load path @@ -540,16 +533,9 @@ inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } // out of the holder node's own extent, and the TOOL's path appends into the // builder's arena, and the decoder above them cannot tell which it has. -// TableMapCarve is a node's extent cursor, PRE-ORDER: a map's whole entry -// array first, then, entry by entry in key order, the arrays of any map an -// entry's value holds by value. The cursor is the node map's, because the -// generated decoder is threaded with that and not with a region. -struct TableMapCarve -{ - uint8_t * at = NULL; // the region path: the node's extent, unspent - int64_t left = 0; - TableWorker * worker = NULL; // the TOOL's path: entries come from the arena -}; +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. // TableMapFill is one map field being decoded: where the next entry lands, and // the entry that last LANDED, which is what the ascending check compares @@ -674,10 +660,11 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // LoadMeasure's term for a map is N x sizeof( Entry ) rounded to // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value -// holds a map of its own, the entries' headers under it. The caller owns the -// allocation precisely so it can refuse a number it did not expect. -typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ); - +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -685,8 +672,8 @@ typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, i static const int64_t kTableMapEntryFloor = 2; inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, - int64_t entry_size, int64_t entry_align, TableMapWireExtentFn inner, - const TableIdTable * ids ) + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; TableReader r( body, length, &scratch, ids ); @@ -694,8 +681,9 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; - if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { return false; } // an N the map's L cannot carry + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); at += (int64_t) n * entry_size; if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term @@ -703,49 +691,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & { uint64_t elem = 0; if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// the same framing walk over an ARRAY OF TABLES that is not a map: its -// elements' own maps are part of this node's extent too -inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each -// length-prefixed element (docs/SPEC-TABLES.md §3.2) -inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t key = 0; - if ( !r.getleb( key ) ) { return true; } - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } r.offset += (int64_t) elem; } return true; @@ -1149,410 +1095,6 @@ func (g *tableGen) emitMapReadField(f *ir.Field) { g.pf("%sr.offset = body_end; // the remaining entries skip by the map's L\n", ind) } -// ---- the NODE EXTENT (docs/SPEC-TABLES.md §2.8, §6.3) ---- -// -// A map's entries are BY-VALUE RECORDS INSIDE THE HOLDER'S NODE EXTENT, laid -// after the record's own storage: count x sizeof( Entry ) at alignof( Entry ), -// zero slack, one array per map reachable BY VALUE from the record — which -// includes a map inside a nested table and a map inside an entry — in -// depth-first field order. The placement is PRE-ORDER: a map's whole entry -// array first, then, entry by entry in key order, the arrays of any map an -// entry's value holds by value. -// -// Two emitters walk that layout and they are ONE walk: the measure advances a -// running offset, and the pack advances the same one and copies. Nothing -// passes between them, which is what makes `used == total` a real check. - -// hasMapExtent reports a member with any map reachable by value — the members -// that carry an extent, and the ones whose two extent walks are emitted. -func (g *tableGen) hasMapExtent(st *ir.Struct) bool { - for _, f := range st.Fields { - if f.IsMap() { - return true - } - // A MAP REACHABLE BY VALUE IS THIS RECORD'S EXTENT (docs/SPEC-TABLES.md - // §2.8), whichever by-value edge reaches it: a nested table, an array - // of them, an enum-keyed array of them, or a union arm. - switch g.edgeOf(f) { - case edgeNested: - if ref, ok := f.Type.Ref.(*ir.Struct); ok && g.hasMapExtent(ref) { - return true - } - case edgeArm: - for _, v := range f.Type.Ref.(*ir.Union).Variants { - if ref := memberOf(g.unit, v.Type); ref != nil && g.hasMapExtent(ref) { - return true - } - } - } - // a nested table this walk does not call an edge can still hold a map, - // because a map makes its holder VARIABLE and every variable nesting is - // an edge — so there is nothing else to look at here - } - return false -} - -// memberOf resolves one closure member by name. -func memberOf(u *ir.Unit, name string) *ir.Struct { - if st := u.Tables[name]; st != nil { - return st - } - return u.Structs[name] -} - -// emitMapExtent emits `MapExtentAt`: the running offset every map array -// reachable by value from one record takes, in the order the pack lays them. -func (g *tableGen) emitMapExtent(st *ir.Struct) { - g.pf("// %sMapExtentAt: the node extent %s's maps take, PRE-ORDER, advancing the\n", st.Name, st.Name) - g.pf("// running offset exactly as %sMapPack advances it (docs/SPEC-TABLES.md §2.8).\n", st.Name) - g.pf("template \ninline bool %sMapExtentAt( const Ctx & ctx, const %s & value, int64_t & at )\n{\n", st.Name, st.Name) - if !g.hasMapExtent(st) { - g.pf(" (void) ctx; (void) value; (void) at; // no map below this record\n") - g.pf(" return true;\n}\n\n") - return - } - g.emitMapExtentWalk(st, "value", func(f *ir.Field, expr, ind string) { - entry := mapEntryOf(f) - g.pf("%s{\n", ind) - g.pf("%s TableMapCursor<%s> cursor = TableMapOrder( ctx, %s );\n", ind, entry.Name, expr) - g.pf("%s if ( !cursor.ok ) { return false; }\n", ind) - g.pf("%s at = ( at + %d ) & ~(int64_t) %d; // at alignof( %s )\n", ind, alignOfEntry(g.unit, entry)-1, alignOfEntry(g.unit, entry)-1, entry.Name) - g.pf("%s at += (int64_t) cursor.count * (int64_t) sizeof( %s ); // the whole array FIRST\n", ind, entry.Name) - if g.isVar(entry.Name) { - g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order\n%s {\n", ind, ind) - g.pf("%s if ( !%sMapExtentAt( ctx, *cursor[i], at ) ) { TableMapRelease( cursor ); return false; }\n", ind, entry.Name) - g.pf("%s }\n", ind) - } - g.pf("%s TableMapRelease( cursor );\n", ind) - g.pf("%s}\n", ind) - }, func(table, expr, ind string) { - g.pf("%sif ( !%sMapExtentAt( ctx, %s, at ) ) { return false; }\n", ind, table, expr) - }) - g.pf(" return true;\n}\n\n") - g.pf("// the whole extent of one node, from a fresh offset: what a pack reserves\n") - g.pf("// for it beside the record's own storage.\n") - g.pf("template \ninline int64_t %sMapExtent( const Ctx & ctx, const %s & value )\n{\n", st.Name, st.Name) - g.pf(" int64_t at = 0;\n") - g.pf(" if ( !%sMapExtentAt( ctx, value, at ) ) { return -1; }\n", st.Name) - g.pf(" return at;\n}\n\n") -} - -// emitMapPack emits `MapPack`: the same walk, copying each map's entries in -// key order into the node's extent and pointing the record's slot at them. -func (g *tableGen) emitMapPack(st *ir.Struct) { - g.pf("// %sMapPack: carve %s's map arrays out of the node's extent and copy the\n", st.Name, st.Name) - g.pf("// entries in ASCENDING key order, PRE-ORDER, advancing the same running\n") - g.pf("// offset %sMapExtentAt advances (docs/SPEC-TABLES.md §2.8).\n", st.Name) - g.pf("template \ninline bool %sMapPack( const Ctx & ctx, const %s & src, %s & dst, uint8_t * extent, int64_t & at, int64_t capacity )\n{\n", st.Name, st.Name, st.Name) - if !g.hasMapExtent(st) { - g.pf(" (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no map below this record\n") - g.pf(" return true;\n}\n\n") - return - } - dstOf := func(expr string) string { return "dst" + expr[len("src"):] } - g.emitMapExtentWalk(st, "src", func(f *ir.Field, expr, ind string) { - entry := mapEntryOf(f) - slot := dstOf(expr) - g.pf("%s{\n", ind) - g.pf("%s TableMapCursor<%s> cursor = TableMapOrder( ctx, %s );\n", ind, entry.Name, expr) - g.pf("%s if ( !cursor.ok ) { return false; }\n", ind) - g.pf("%s at = ( at + %d ) & ~(int64_t) %d;\n", ind, alignOfEntry(g.unit, entry)-1, alignOfEntry(g.unit, entry)-1) - g.pf("%s const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( %s );\n", ind, entry.Name) - g.pf("%s if ( at + bytes > capacity ) { TableMapRelease( cursor ); return false; }\n", ind) - g.pf("%s %s * placed = (%s *) ( extent + at );\n", ind, entry.Name, entry.Name) - g.pf("%s at += bytes;\n", ind) - g.pf("%s %s.count = cursor.count;\n", ind, slot) - g.pf("%s %s.padding = 0;\n", ind, slot) - g.pf("%s %s.entries.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &%s.entries ) : 0;\n", ind, slot, slot) - g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ )\n%s {\n", ind, ind) - g.pf("%s memcpy( (void *) ( placed + i ), (const void *) cursor[i], sizeof( %s ) ); // trivially copyable, by construction\n", ind, entry.Name) - g.pf("%s }\n", ind) - if g.isVar(entry.Name) { - g.pf("%s for ( int32_t i = 0; i < cursor.count; i++ )\n%s {\n", ind, ind) - g.pf("%s if ( !%sMapPack( ctx, *cursor[i], placed[i], extent, at, capacity ) ) { TableMapRelease( cursor ); return false; }\n", ind, entry.Name) - g.pf("%s }\n", ind) - } - g.pf("%s TableMapRelease( cursor );\n", ind) - g.pf("%s}\n", ind) - }, func(table, expr, ind string) { - g.pf("%sif ( !%sMapPack( ctx, %s, %s, extent, at, capacity ) ) { return false; }\n", ind, table, expr, dstOf(expr)) - }) - g.pf(" return true;\n}\n\n") -} - -// emitMapExtentWalk is the ONE walk both extent emitters take: the record's -// fields in declaration order, every map at its own position, descending each -// by-value edge in place. A pointer is NOT an edge here — a pointee is its own -// node with its own extent — and neither is a union arm that is one, for the -// same reason. -func (g *tableGen) emitMapExtentWalk(st *ir.Struct, subject string, mapField func(f *ir.Field, expr, ind string), descend func(table, expr, ind string)) { - v := edgeVisitor{read: subject} - for _, f := range st.Fields { - if f.IsMap() { - mapField(f, subject+"."+f.Name, " ") - continue - } - switch g.edgeOf(f) { - case edgeNested: - ref, _ := f.Type.Ref.(*ir.Struct) - if ref == nil || !g.hasMapExtent(ref) { - continue - } - g.emitVariableByValueWalk(f, v, func(expr edgeExpr) { descend(f.Type.Name, expr.Src, " ") }) - g.emitUnreachedMapRefusal(f, ref, subject) - case edgeArm: - un := f.Type.Ref.(*ir.Union) - any := false - for _, arm := range un.Variants { - if ref := memberOf(g.unit, arm.Type); ref != nil && g.hasMapExtent(ref) { - any = true - } - } - if !any { - continue - } - // only the ARMS that hold a map are descended: a pointer arm and a - // byte buffer arm reach nodes, not this node's extent - armed := edgeVisitor{read: subject, - pointer: func(*ir.Field, edgeExpr) {}, - blob: func(*ir.Field, edgeExpr) {}, - descend: func(table string, expr edgeExpr, indent string) { - if ref := memberOf(g.unit, table); ref != nil && g.hasMapExtent(ref) { - descend(table, expr.Src, indent) - } - }, - } - g.emitVariableUnionWalk(f, armed) - } - } -} - -// alignOfEntry is the C ABI alignment of one generated entry record — the -// alignment its array is laid at, and the same model §20.3 commits the -// compiler to for every record in the closure. -func alignOfEntry(u *ir.Unit, entry *ir.Struct) int64 { - if ml := ir.RecordLayout(u, entry); ml != nil && ml.Align > 0 { - return ml.Align - } - return 8 -} - -// ---- the three walks at a map (docs/SPEC-TABLES.md §2.8, §3.1) ---- -// -// A map is a BY-VALUE EDGE of the ONE declaration-order walk: it is reached at -// its field's position, its entries are visited in ASCENDING KEY ORDER, and -// each entry's value is descended for the pointer slots inside it before the -// next entry is reached. A map declared before a pointer field therefore -// reaches a shared node FIRST and numbers it first, exactly as a union arm or -// a nested table declared there does. The rule is the walk's, not the map's. - -// mapNumberEdge descends one map's entries for the NUMBERING walk. -func (g *tableGen) mapNumberEdge(f *ir.Field) { - g.emitMapEntryLoop(f, "value", "return false;", func(entry, elem, ind string) { - g.pf("%sif ( !%sNumber( ctx, numbering, %s ) ) { TableMapRelease( cursor_%s ); return false; }\n", ind, entry, elem, f.Name) - }) -} - -// mapPackMeasureEdge descends one map's entries for the PACK MEASURE. -func (g *tableGen) mapPackMeasureEdge(f *ir.Field) { - g.emitMapEntryLoop(f, "value", "return -1;", func(entry, elem, ind string) { - g.pf("%sint64_t inner = %sPackMeasure( ctx, seen, %s );\n", ind, entry, elem) - g.pf("%sif ( inner < 0 ) { TableMapRelease( cursor_%s ); return -1; }\n", ind, f.Name) - g.pf("%sbytes += inner;\n", ind) - }) -} - -// mapPackEdge descends one map's entries for the PACK, against the array -// MapPack already placed in the node's extent. -func (g *tableGen) mapPackEdge(f *ir.Field) { - entry := mapEntryOf(f) - g.emitMapEntryLoopHead(f, "src", "return false;") - g.pf(" %s * placed_%s = (%s *) ( dst.%s.entries.value != 0 ? ( (uint8_t *) &dst.%s.entries + dst.%s.entries.value ) : NULL );\n", - entry.Name, f.Name, entry.Name, f.Name, f.Name, f.Name) - g.pf(" for ( int32_t i = 0; i < cursor_%s.count; i++ )\n {\n", f.Name) - g.pf(" if ( !%sPackEdges( ctx, seen, *cursor_%s[i], placed_%s[i], base, capacity, used ) ) { TableMapRelease( cursor_%s ); return false; }\n", - entry.Name, f.Name, f.Name, f.Name) - g.pf(" }\n") - g.pf(" TableMapRelease( cursor_%s );\n }\n", f.Name) -} - -// emitMapEntryLoopHead opens one map's sorted cursor over the given subject. -func (g *tableGen) emitMapEntryLoopHead(f *ir.Field, subject, onBad string) { - entry := mapEntryOf(f) - g.pf(" { // %s: a by-value edge, entries in ASCENDING key order (§2.8, §3.1)\n", f.Name) - g.pf(" TableMapCursor<%s> cursor_%s = TableMapOrder( ctx, %s.%s );\n", entry.Name, f.Name, subject, f.Name) - g.pf(" if ( !cursor_%s.ok ) { %s }\n", f.Name, onBad) -} - -// emitMapEntryLoop is the whole shape: the cursor, the loop, the release. -func (g *tableGen) emitMapEntryLoop(f *ir.Field, subject, onBad string, body func(entry, elem, ind string)) { - entry := mapEntryOf(f) - g.emitMapEntryLoopHead(f, subject, onBad) - g.pf(" for ( int32_t i = 0; i < cursor_%s.count; i++ )\n {\n", f.Name) - body(entry.Name, fmt.Sprintf("*cursor_%s[i]", f.Name), " ") - g.pf(" }\n") - g.pf(" TableMapRelease( cursor_%s );\n }\n", f.Name) -} - -// emitMapWalkSurface emits the two extent walks for every variable member of a -// map-bearing unit. They are emitted for EVERY such member, because a walk -// that descends a by-value nesting has to be able to name the nested one's. -func (g *tableGen) emitMapWalkSurface(members []*ir.Struct) { - if !g.anyMap { - return - } - for _, st := range g.varMembers(members) { - g.emitMapWireExtent(st) - g.emitMapExtent(st) - g.emitMapPack(st) - } -} - -// emitNodeBytes emits the bytes ONE NODE takes in a packed region: the -// record's own storage rounded to the arena's alignment, plus the extent its -// maps take (docs/SPEC-TABLES.md §2.8, §6.3), the sum rounded again so the -// next node starts aligned. A unit with no map emits exactly the term it -// always emitted. -func (g *tableGen) emitNodeBytes(table, expr, ind, onBad string, plain func(term string), extent func(term string)) { - target := memberOf(g.unit, table) - if !g.anyMap || target == nil || !g.hasMapExtent(target) { - // a node with no map below it takes exactly the term it always took, - // so a map-free unit emits what it emitted before the construct existed - plain(fmt.Sprintf("TableAlignUp64( (int64_t) sizeof( %s ) )", table)) - return - } - g.pf("%sint64_t node_extent = %sMapExtent( ctx, %s );\n", ind, table, expr) - g.pf("%sif ( node_extent < 0 ) { %s }\n", ind, onBad) - extent(fmt.Sprintf("TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + node_extent )", table)) -} - -// emitMapWireExtent emits `WireExtent`: the region bytes one record's maps -// command, read from the wire FRAMING alone at every depth -// (docs/SPEC-TABLES.md §2.8, §6.5). False is the refusal — an N the map's L -// cannot carry — and it is what makes LoadMeasure answer -1. -func (g *tableGen) emitMapWireExtent(st *ir.Struct) { - g.pf("// %sWireExtent: the extent %s's maps command, from the FRAMING alone.\n", st.Name, st.Name) - g.pf("// It reads no field value, so a caller can refuse a number it did not\n") - g.pf("// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5).\n") - g.pf("inline bool %sWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids )\n{\n", st.Name) - if !g.hasMapExtent(st) { - g.pf(" (void) body; (void) length; (void) at; (void) ids; // no map below this record\n") - g.pf(" return true;\n}\n\n") - return - } - g.pf(" TableReport scratch; // the scan's framing damage is the LOAD's to report\n") - g.pf(" TableReader r( body, length, &scratch, ids );\n") - g.pf(" for ( ;; )\n {\n") - g.pf(" uint64_t field_ref = 0;\n") - g.pf(" if ( !r.getleb( field_ref ) ) { return true; }\n") - g.pf(" if ( field_ref == 0 ) { return true; }\n") - g.pf(" if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; }\n") - g.pf(" const uint64_t field_id = ids->at( field_ref );\n") - g.pf(" if ( !r.has( 1 ) ) { return true; }\n") - g.pf(" uint8_t field_kind = r.get8();\n") - g.emitMapExtentWireCases(st) - g.pf(" if ( !r.skip( field_kind ) ) { return true; }\n") - g.pf(" }\n}\n\n") -} - -// emitMapExtentWireCases emits one arm per map field and one per by-value -// nesting that holds a map, in DECLARATION ORDER, so the framing scan advances -// the running offset in the same order the pack and the load carve it. -func (g *tableGen) emitMapExtentWireCases(st *ir.Struct) { - for _, f := range st.Fields { - if f.IsMap() { - entry := mapEntryOf(f) - inner := "NULL" - if g.hasMapExtent(entry) { - inner = "&" + entry.Name + "WireExtent" - } - g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s\n {\n", ir.TableFieldWireId(f), tkArray, f.Name) - g.pf(" uint64_t map_len = 0;\n") - g.pf(" if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; }\n") - g.pf(" const uint8_t * map_body = r.buffer + r.offset;\n") - g.pf(" r.offset += (int64_t) map_len;\n") - g.pf(" if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( %s ), (int64_t) alignof( %s ), %s, ids ) ) { return false; }\n", - entry.Name, entry.Name, inner) - g.pf(" continue;\n }\n") - continue - } - switch g.edgeOf(f) { - case edgeNested: - ref, _ := f.Type.Ref.(*ir.Struct) - if ref == nil || !g.hasMapExtent(ref) { - continue - } - // a nested table's maps are part of THIS node's extent, so its own - // scan runs over the nested body at the running offset - kind, walk := tkTable, "" - switch { - case f.KeyEnum != "": - kind, walk = tkKeyed, "TableWireExtentKeyed" - case f.Array != ir.ArrayNone: - kind, walk = tkArray, "TableWireExtentElements" - } - g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s: a nesting that holds a map\n {\n", ir.TableFieldWireId(f), kind, f.Name) - g.pf(" uint64_t nested_len = 0;\n") - g.pf(" if ( !r.getleb( nested_len ) || !r.room( nested_len ) ) { return true; }\n") - g.pf(" const uint8_t * nested_body = r.buffer + r.offset;\n") - g.pf(" r.offset += (int64_t) nested_len;\n") - if walk == "" { - g.pf(" if ( !%sWireExtent( nested_body, (int64_t) nested_len, at, ids ) ) { return false; }\n", f.Type.Name) - } else { - g.pf(" if ( !%s( nested_body, (int64_t) nested_len, at, &%sWireExtent, ids ) ) { return false; }\n", walk, f.Type.Name) - } - g.pf(" continue;\n }\n") - case edgeArm: - un := f.Type.Ref.(*ir.Union) - any := false - for _, v := range un.Variants { - if ref := memberOf(g.unit, v.Type); ref != nil && g.hasMapExtent(ref) { - any = true - } - } - if !any { - continue - } - g.pf(" if ( field_id == 0x%016xull && field_kind == %d ) // %s: a union arm that holds a map\n {\n", ir.TableFieldWireId(f), tkUnion, f.Name) - g.pf(" uint64_t arm_ref = 0;\n") - g.pf(" if ( !r.getleb( arm_ref ) ) { return true; }\n") - g.pf(" if ( arm_ref == 0 ) { continue; } // None: the reference is the whole payload\n") - g.pf(" if ( arm_ref > (uint64_t) ids->count ) { return true; }\n") - g.pf(" const uint64_t arm_id = ids->at( arm_ref );\n") - g.pf(" if ( !r.has( 1 ) ) { return true; }\n") - g.pf(" r.offset += 1; // the arm's kind byte\n") - g.pf(" uint64_t arm_len = 0;\n") - g.pf(" if ( !r.getleb( arm_len ) || !r.room( arm_len ) ) { return true; }\n") - g.pf(" const uint8_t * arm_body = r.buffer + r.offset;\n") - g.pf(" r.offset += (int64_t) arm_len;\n") - g.pf(" switch ( arm_id )\n {\n") - for _, v := range un.Variants { - ref := memberOf(g.unit, v.Type) - if ref == nil || !g.hasMapExtent(ref) { - continue - } - g.pf(" case 0x%016xull: if ( !%sWireExtent( arm_body, (int64_t) arm_len, at, ids ) ) { return false; } break; // %s\n", - ir.TableWireId(v.Name), v.Type, v.Name) - } - g.pf(" default: break; // an arm this reader cannot name reads None\n") - g.pf(" }\n") - g.pf(" continue;\n }\n") - } - } -} - -// emitRootDataBytes emits a load's DATA term for the root itself: its record, -// plus the extent its own maps take, read from the wire framing (§2.8, §6.5). -func (g *tableGen) emitRootDataBytes(st *ir.Struct, ind, onBad string) { - if !g.anyMap { - g.pf("%sint64_t data = TableAlignUp64( (int64_t) sizeof( %s ) );\n", ind, st.Name) - return - } - g.pf("%sint64_t root_extent = 0;\n", ind) - g.pf("%sif ( !%sWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { %s }\n", ind, st.Name, onBad) - g.pf("%sint64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + root_extent );\n", ind, st.Name) -} - // ---- the builder's five, and the optional index (docs/SPEC-TABLES.md §2.8) ---- // // FREE FUNCTIONS taking the worker or the arena, as Emplace and the arena At @@ -1707,246 +1249,59 @@ func (g *tableGen) emitMapBuilderSurfaces(members []*ir.Struct) { } } -// ---- the COOK's write side at a map (docs/SPEC-TABLES.md §2.8, §7.6) ---- +// ---- the three walks at a map (docs/SPEC-TABLES.md §2.8, §3.1) ---- // -// A cook is a region written verbatim, so a cooked map is its SORTED entry -// array where the cook put it: the node's extent, laid after the record's own -// storage by the same PRE-ORDER rule the pack lays it by. Find is then a -// binary search over the mapped bytes, in place, with nothing to parse. - -// cookMapsSignature is one record's extent writer. -func (g *tableGen) cookMapsSignature(st *ir.Struct) string { - return fmt.Sprintf("template inline bool %sCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const %s & value, TableByteOrder order )", st.Name, st.Name) -} - -// emitCookMaps emits one record's extent writer: every map reachable by value, -// PRE-ORDER, each entry through its own cook body. -func (g *tableGen) emitCookMaps(st *ir.Struct) { - g.pf("// %sCookMaps: %s's map arrays into the node's extent, PRE-ORDER, the entries\n", st.Name, st.Name) - g.pf("// in ASCENDING key order, each through its own cook body (§2.8, §7.6).\n") - g.pf("%s\n{\n", g.cookMapsSignature(st)) - if !g.hasMapExtent(st) { - g.pf(" (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order;\n") - g.pf(" return true; // no map below this record\n}\n\n") - return - } - g.pf(" (void) region; // a map's entries carry their own references through their own bodies\n") - ml := ir.RecordLayout(g.unit, st) - offsetOf := func(name string) int64 { - for i := range ml.Fields { - if ml.Fields[i].Field.Name == name { - return ml.Fields[i].Offset - } - } - return 0 - } - for _, f := range st.Fields { - if f.IsMap() { - entry := mapEntryOf(f) - el := ir.RecordLayout(g.unit, entry) - slot := offsetOf(f.Name) - g.pf(" { // %s\n", f.Name) - g.pf(" TableMapCursor<%s> cursor = TableMapOrder( ctx, value.%s );\n", entry.Name, f.Name) - g.pf(" if ( !cursor.ok ) { return false; }\n") - g.pf(" at = ( at + %d ) & ~(int64_t) %d; // at alignof( %s )\n", el.Align-1, el.Align-1, entry.Name) - g.pf(" uint8_t * array = extent + at;\n") - g.pf(" at += (int64_t) cursor.count * %d; // the whole array FIRST\n", el.Size) - g.pf(" // the SIXTEEN BYTES of the slot: the self-relative delta, then the count\n") - g.pf(" table_cook_put( record + %d, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + %d ) ) : 0, 8, order );\n", slot, slot) - g.pf(" table_cook_put( record + %d, (uint64_t) (uint32_t) cursor.count, 4, order );\n", slot+8) - g.pf(" for ( int32_t i = 0; i < cursor.count; i++ )\n {\n") - g.pf(" %s\n", g.cookBodyCall(entry, fmt.Sprintf("array + i * %d", el.Size), "*cursor[i]")) - g.pf(" }\n") - if g.hasMapExtent(entry) { - g.pf(" for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order\n {\n") - g.pf(" if ( !%sCookMaps( ctx, region, extent, at, array + i * %d, *cursor[i], order ) ) { TableMapRelease( cursor ); return false; }\n", entry.Name, el.Size) - g.pf(" }\n") - } - g.pf(" TableMapRelease( cursor );\n }\n") - continue - } - if g.edgeOf(f) != edgeNested { - continue - } - ref, _ := f.Type.Ref.(*ir.Struct) - if ref == nil || !g.hasMapExtent(ref) { - continue - } - nested := offsetOf(f.Name) - if f.Array == ir.ArrayNone { - g.pf(" if ( !%sCookMaps( ctx, region, extent, at, record + %d, value.%s, order ) ) { return false; } // %s\n", ref.Name, nested, f.Name, f.Name) - continue - } - stride := cookElementBytes(g.unit, f) - base := "value." + f.Name - bound := fmt.Sprintf("%d", f.ArrayBound) - if f.KeyEnum != "" && st.IsTable { - base += ".slots" - } - if f.Array == ir.ArrayCounted { - // THE LIVE COUNT, as the extent walk counts it: a slot past the - // count is storage the walk does not reach, and a non-empty map in - // one was already refused there (§7.6) - bound = fmt.Sprintf("( value.%s_count < %d ? value.%s_count : %d )", f.Name, f.ArrayBound, f.Name, f.ArrayBound) - } - g.pf(" for ( int32_t i = 0; i < %s; i++ ) // %s\n {\n", bound, f.Name) - g.pf(" if ( !%sCookMaps( ctx, region, extent, at, record + %d + i * %d, %s[i], order ) ) { return false; }\n", ref.Name, nested, stride, base) - g.pf(" }\n") - } - g.pf(" return true;\n}\n\n") -} - -// emitCookNode emits `CookNode`: one NODE's record and then its own extent. -// A nested record's writer is the body alone, because a nesting's maps are -// part of the HOLDER's extent and this walk already reached them. -func (g *tableGen) emitCookNode(st *ir.Struct) { - ml := ir.RecordLayout(g.unit, st) - record := cookAlignUp(ml.Size, ir.RegionAlignFloor) - g.pf("// %sCookNode: one node — the record, then the extent its maps take (§2.8).\n", st.Name) - g.pf("template inline bool %sCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const %s & value, TableByteOrder order )\n{\n", st.Name, st.Name) - if g.isVar(st.Name) { - g.pf(" if ( !%sCookBody( ctx, region, at, value, order ) ) { return false; }\n", st.Name) - } else { - g.pf(" %sCookBody( at, value, order );\n", st.Name) - } - g.pf(" int64_t extent_at = 0;\n") - g.pf(" return %sCookMaps( ctx, region, at + %d, extent_at, at, value, order );\n}\n\n", st.Name, record) -} - -// emitCookMapSurface emits the extent writer and the node writer for every -// closure member of a map-bearing unit. -func (g *tableGen) emitCookMapSurface(members []*ir.Struct) { - if !g.anyMap { - return - } - var bodies []*ir.Struct - for _, st := range members { - if ir.RecordLayout(g.unit, st) != nil { - bodies = append(bodies, st) - } - } - for _, st := range bodies { - g.pf("%s;\n", g.cookMapsSignature(st)) - } - g.pf("\n") - for _, st := range bodies { - g.emitCookMaps(st) - } - for _, st := range bodies { - g.emitCookNode(st) - } -} +// A map is a BY-VALUE EDGE of the ONE declaration-order walk: it is reached at +// its field's position, its entries are visited in ASCENDING KEY ORDER, and +// each entry's value is descended for the pointer slots inside it before the +// next entry is reached. A map declared before a pointer field therefore +// reaches a shared node FIRST and numbers it first, exactly as a union arm or +// a nested table declared there does. The rule is the walk's, not the map's. -// cookNodeBytes is one node's whole span in a cooked region: its record at the -// region's alignment floor, plus the extent its maps take (docs/SPEC-TABLES.md -// §2.8, §7.2). -func (g *tableGen) emitCookNodeBytes(st *ir.Struct, ind, expr, onBad string) { - ml := ir.RecordLayout(g.unit, st) - if !g.anyMap || !g.hasMapExtent(st) { - g.pf("%ssize = %d; node_align = %d;\n", ind, ml.Size, ml.Align) - return - } - g.pf("%s{\n", ind) - g.pf("%s const int64_t extent = %sMapExtent( ctx, %s );\n", ind, st.Name, expr) - g.pf("%s if ( extent < 0 ) { %s }\n", ind, onBad) - g.pf("%s size = %d + extent; node_align = %d;\n", ind, cookAlignUp(ml.Size, ir.RegionAlignFloor), ml.Align) - g.pf("%s}\n", ind) +// mapNumberEdge descends one map's entries for the NUMBERING walk. +func (g *tableGen) mapNumberEdge(f *ir.Field) { + g.emitMapEntryLoop(f, "value", "return false;", func(entry, elem, ind string) { + g.pf("%sif ( !%sNumber( ctx, numbering, %s ) ) { TableMapRelease( cursor_%s ); return false; }\n", ind, entry, elem, f.Name) + }) } -// onlyMapFields reports a record whose every field is a map — a cook body that -// writes the empty slots and reads nothing off the value, because the extent -// writer fills them. -func onlyMapFields(st *ir.Struct) bool { - for _, f := range st.Fields { - if !f.IsMap() { - return false - } - } - return len(st.Fields) > 0 +// mapPackMeasureEdge descends one map's entries for the PACK MEASURE. +func (g *tableGen) mapPackMeasureEdge(f *ir.Field) { + g.emitMapEntryLoop(f, "value", "return -1;", func(entry, elem, ind string) { + g.pf("%sint64_t inner = %sPackMeasure( ctx, seen, %s );\n", ind, entry, elem) + g.pf("%sif ( inner < 0 ) { TableMapRelease( cursor_%s ); return -1; }\n", ind, f.Name) + g.pf("%sbytes += inner;\n", ind) + }) } -// mapColumn is the four descriptor columns a MAP field carries -// (docs/SPEC-TABLES.md §2.8, §16): the generated entry's descriptor, and the -// three thunks the ONE text walk cannot spell for itself. Empty in a unit that -// declares no map, so a map-free unit's descriptors are what they always were. -func (g *tableGen) mapColumn(f *ir.Field) string { - if !g.anyMap { - return "" - } - if !f.IsMap() { - return "NULL, NULL, NULL, NULL, " - } +// mapPackEdge descends one map's entries for the PACK, against the array +// MapPack already placed in the node's extent. +func (g *tableGen) mapPackEdge(f *ir.Field) { entry := mapEntryOf(f) - n := entry.Name - hold := fmt.Sprintf("TableMap<%s>", n) - count := fmt.Sprintf("[]( const void * slot ) -> int32_t { return ( (const %s *) slot )->count; }", hold) - at := fmt.Sprintf("[]( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const %s *) slot )->Entries() + index ); }", hold) - var insert string - if mapKeyIsString(f) { - insert = fmt.Sprintf("[]( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * "+ - "{ if ( key == NULL || key_length > k%sKeyBound ) { return NULL; } "+ // KEYS NEVER CLAMP - "%s * placed = TableMapPlace( worker, *(%s *) slot, key ); "+ - "if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }", n, n, hold) - } else { - typ, _ := g.cppFieldType(ir.MapKeyField(f).Type) - insert = fmt.Sprintf("[]( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * "+ - "{ %s * placed = TableMapPlace( worker, *(%s *) slot, (%s) key_value ); "+ - "if ( placed != NULL ) { TableEntrySetKey( *placed, (%s) key_value ); } return (void *) placed; }", n, hold, typ, typ) - } - return fmt.Sprintf("&%sTableInfo, %s, %s, %s, ", n, count, at, insert) -} - -// nodeStorageBody, nodeStorageArg and nodeStorageReader are the ONE extra -// parameter a node's storage takes where a map rides in an extent -// (docs/SPEC-TABLES.md §2.8): the record's body, from which the framing scan -// sums the entry arrays. A root that can name no such record does not take it, -// so a map-free unit's dispatch is the one it always emitted. -func (g *tableGen) nodeStorageBody(anyExtent bool) string { - if anyExtent { - return "const uint8_t * body, " - } - return "" -} - -func (g *tableGen) nodeStorageArg(root *ir.Struct) string { - if g.rootHasExtent(root) { - return "body, " - } - return "" -} - -func (g *tableGen) nodeStorageReader(root *ir.Struct) string { - if g.rootHasExtent(root) { - return "r.buffer, " - } - return "" + g.emitMapEntryLoopHead(f, "src", "return false;") + g.pf(" %s * placed_%s = (%s *) ( dst.%s.entries.value != 0 ? ( (uint8_t *) &dst.%s.entries + dst.%s.entries.value ) : NULL );\n", + entry.Name, f.Name, entry.Name, f.Name, f.Name, f.Name) + g.pf(" for ( int32_t i = 0; i < cursor_%s.count; i++ )\n {\n", f.Name) + g.pf(" if ( !%sPackEdges( ctx, seen, *cursor_%s[i], placed_%s[i], base, capacity, used ) ) { TableMapRelease( cursor_%s ); return false; }\n", + entry.Name, f.Name, f.Name, f.Name) + g.pf(" }\n") + g.pf(" TableMapRelease( cursor_%s );\n }\n", f.Name) } -// rootHasExtent reports whether any record one root's numbering can name holds -// a map by value — which is what decides both halves of the signature above. -func (g *tableGen) rootHasExtent(root *ir.Struct) bool { - if !g.anyMap { - return false - } - return slices.ContainsFunc(g.pointerReachable(root), g.hasMapExtent) +// emitMapEntryLoopHead opens one map's sorted cursor over the given subject. +func (g *tableGen) emitMapEntryLoopHead(f *ir.Field, subject, onBad string) { + entry := mapEntryOf(f) + g.pf(" { // %s: a by-value edge, entries in ASCENDING key order (§2.8, §3.1)\n", f.Name) + g.pf(" TableMapCursor<%s> cursor_%s = TableMapOrder( ctx, %s.%s );\n", entry.Name, f.Name, subject, f.Name) + g.pf(" if ( !cursor_%s.ok ) { %s }\n", f.Name, onBad) } -// emitUnreachedMapRefusal refuses an UNREACHED NON-EMPTY MAP SLOT, the same -// refusal §7.6 gives a pointer in that position (docs/SPEC-TABLES.md §2.8): a -// COUNTED array's slots past its live count are storage the walk does not -// reach, so a non-empty map in one names entries the region will not hold, and -// the write answers false with nothing partial written. -// -// The test is the extent itself: an empty map takes no bytes and advances the -// running offset by none, so a record whose extent measures ZERO is a record -// whose every by-value map is empty. A measure that refuses answers non-zero -// here too, and refusing on it is the same answer one level up. -func (g *tableGen) emitUnreachedMapRefusal(f *ir.Field, ref *ir.Struct, subject string) { - if f.Array != ir.ArrayCounted { - return // every other array shape is reached whole - } - g.pf(" for ( int32_t i = %s.%s_count; i < %d; i++ ) // %s: the slots the walk does not reach (§7.6)\n {\n", - subject, f.Name, f.ArrayBound, f.Name) - g.pf(" if ( !TableMapUnreachedEmpty( %sMapExtent( ctx, %s.%s[i] ) ) ) { return false; }\n", ref.Name, subject, f.Name) - g.pf(" }\n") +// emitMapEntryLoop is the whole shape: the cursor, the loop, the release. +func (g *tableGen) emitMapEntryLoop(f *ir.Field, subject, onBad string, body func(entry, elem, ind string)) { + entry := mapEntryOf(f) + g.emitMapEntryLoopHead(f, subject, onBad) + g.pf(" for ( int32_t i = 0; i < cursor_%s.count; i++ )\n {\n", f.Name) + body(entry.Name, fmt.Sprintf("*cursor_%s[i]", f.Name), " ") + g.pf(" }\n") + g.pf(" TableMapRelease( cursor_%s );\n }\n", f.Name) } diff --git a/internal/codegen/cpptable/pointers.go b/internal/codegen/cpptable/pointers.go index 7954411f0..aa8086c50 100644 --- a/internal/codegen/cpptable/pointers.go +++ b/internal/codegen/cpptable/pointers.go @@ -163,6 +163,11 @@ type edgeVisitor struct { // and because a map's entries are not a path under the write subject: the // pack's twin is the array it already placed in the node's extent. mapField func(f *ir.Field) + // listField is one `[]T` whose elements the walk descends, in INDEX order + // (docs/SPEC-TABLES.md §2.9). It takes the visitor itself, because a list's + // element is the pointer slot or the nested table the visitor already + // knows how to reach. The list adds only the cursor around them. + listField func(f *ir.Field, v edgeVisitor) } // at spells one storage PATH — ".field", ".field[i]", ".body.chunk" — under @@ -190,9 +195,25 @@ const ( // A map whose entry reaches nothing is not an edge — its extent is the // extent walk's, not this one's. edgeMap + // edgeList is a `[]T` whose ELEMENTS reach a node: a `[]*T`, whose elements + // ARE the pointer slots, or a `[]T` over a variable table with pointers + // inside it. The walk descends the elements in INDEX ORDER, the order the + // wire carries and the order a region holds (docs/SPEC-TABLES.md §2.9). A + // list whose elements reach nothing is not an edge: its extent is the + // extent walk's, not this one's. + edgeList ) func (g *tableGen) edgeOf(f *ir.Field) edgeKind { + if f.IsList() { + if listElementIsPointer(f) { + return edgeList + } + if ref := listElementStruct(f); ref != nil && g.isVar(ref.Name) { + return edgeList + } + return edgeNone + } if f.IsMap() { if g.noVariableEdges(f.MapEntry) { return edgeNone @@ -304,6 +325,8 @@ func (g *tableGen) emitEdgeOf(f *ir.Field, v edgeVisitor) { g.emitVariableUnionWalk(f, v) case edgeMap: v.mapField(f) + case edgeList: + v.listField(f, v) } } @@ -456,8 +479,9 @@ func (g *tableGen) emitVariableSurface(members []*ir.Struct) { if !g.anyVariable { return } - g.emitMapWalkSurface(members) + g.emitExtentWalkSurface(members) g.emitMapBuilderSurfaces(members) + g.emitListBuilderSurfaces(members) for _, st := range g.varMembers(members) { g.owner = st g.emitNumber(st) @@ -551,7 +575,8 @@ func (g *tableGen) emitNumber(st *ir.Struct) { descend: func(table string, expr edgeExpr, indent string) { g.pf("%sif ( !%sNumber( ctx, numbering, %s ) ) { return false; }\n", indent, table, expr.Src) }, - mapField: g.mapNumberEdge, + mapField: g.mapNumberEdge, + listField: func(f *ir.Field, v edgeVisitor) { g.emitListEdge(f, v, "return false;") }, }) g.pf(" return true;\n}\n\n") } @@ -614,7 +639,8 @@ func (g *tableGen) emitPackMeasure(st *ir.Struct) { g.pf("%sif ( inner < 0 ) { return -1; }\n", indent) g.pf("%sbytes += inner;\n", indent) }, - mapField: g.mapPackMeasureEdge, + mapField: g.mapPackMeasureEdge, + listField: func(f *ir.Field, v edgeVisitor) { g.emitListEdge(f, v, "return -1;") }, }) g.pf(" return bytes;\n}\n\n") } @@ -632,7 +658,7 @@ func (g *tableGen) emitPack(st *ir.Struct) { g.pf("// required sign (§6.3), and sharing and a back-reference are one fact. A\n") g.pf("// reference to a node whose descent is still OPEN is a cycle, and this\n") g.pf("// refuses it rather than packing one.\n") - if g.anyMap { + if g.anyExtent { // THE NODE'S EXTENT IS CARVED BEFORE ANY POINTEE IS PLACED // (docs/SPEC-TABLES.md §2.8, §6.3): a node's extent runs to the next // directory entry, so a pointee laid between two of its map arrays @@ -645,7 +671,7 @@ func (g *tableGen) emitPack(st *ir.Struct) { g.pf(" int64_t at = 0;\n") g.pf(" uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( %s ) );\n", st.Name) g.pf(" const int64_t room = capacity - ( (int64_t) ( extent - base ) );\n") - g.pf(" if ( !%sMapPack( ctx, src, dst, extent, at, room ) ) { return false; }\n", st.Name) + g.pf(" if ( !%sExtentPack( ctx, src, dst, extent, at, room ) ) { return false; }\n", st.Name) g.pf(" return %sPackEdges( ctx, seen, src, dst, base, capacity, used );\n}\n\n", st.Name) g.pf("template \ninline bool %sPackEdges( const Ctx & ctx, TablePackMap & seen, const %s & src, %s & dst, uint8_t * base, int64_t capacity, int64_t & used )\n{\n", st.Name, st.Name, st.Name) if g.noVariableEdges(st) { @@ -697,12 +723,13 @@ func (g *tableGen) emitPack(st *ir.Struct) { blob: g.emitPackBlobField, descend: func(table string, expr edgeExpr, indent string) { call := "Pack" - if g.anyMap { + if g.anyExtent { call = "PackEdges" // the extent walk already reached this nesting } g.pf("%sif ( !%s%s( ctx, seen, %s, %s, base, capacity, used ) ) { return false; }\n", indent, table, call, expr.Src, expr.Dst) }, - mapField: g.mapPackEdge, + mapField: g.mapPackEdge, + listField: func(f *ir.Field, v edgeVisitor) { g.emitListEdge(f, v, "return false;") }, }) g.pf(" return true;\n}\n\n") } @@ -811,8 +838,8 @@ func (g *tableGen) emitBuilderAndPublicSurface(st *ir.Struct) { g.pf(" below = %sPackMeasure( ctx, seen, root );\n", n) g.pf(" }\n") g.pf(" if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it\n") - if g.anyMap { - g.pf(" int64_t root_extent = %sMapExtent( ctx, root );\n", n) + if g.anyExtent { + g.pf(" int64_t root_extent = %sExtent( ctx, root );\n", n) g.pf(" if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run\n") g.pf(" int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + root_extent ) + below;\n", n) } else { @@ -823,7 +850,7 @@ func (g *tableGen) emitBuilderAndPublicSurface(st *ir.Struct) { g.pf(" // allocator's contract: a packed region carries node padding.\n") g.pf(" uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total );\n") g.pf(" if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; }\n") - if g.anyMap { + if g.anyExtent { g.pf(" int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + root_extent );\n", n) } else { g.pf(" int64_t used = TableAlignUp64( (int64_t) sizeof( %s ) );\n", n) @@ -1007,8 +1034,8 @@ func (g *tableGen) emitBuilderAndPublicSurface(st *ir.Struct) { g.pf(" nodes.entries = directory;\n") g.pf(" nodes.count = records + 1;\n") g.pf(" nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here\n") - if g.anyMap { - g.pf(" nodes.worker = &builder.main; // and a map's entries are the arena's, not a node extent's (§2.8)\n") + if g.anyExtent { + g.pf(" nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9)\n") } g.pf(" {\n") g.pf(" TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table );\n") @@ -1040,12 +1067,18 @@ func (g *tableGen) emitBuilderAndPublicSurface(st *ir.Struct) { g.pf(" }\n }\n") g.pf(" TableReader r( wire, wire_bytes, out, &ids_table );\n") g.pf(" r.nested = false; // the ROOT body, the one that carries the node table\n") - if g.anyMap { - g.pf(" TableMapCarve root_carve;\n") + if g.anyExtent { + g.pf(" TableExtentCarve root_carve;\n") g.pf(" root_carve.worker = &builder.main;\n") g.pf(" nodes.carve = &root_carve;\n") } g.pf(" bool ok = %sLoadBody( r, nodes, *root );\n", n) + if g.anyExtent { + g.pf(" // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md\n") + g.pf(" // §2.9): the partial builder is the caller's to discard, and the report\n") + g.pf(" // holds what it held when the count was met\n") + g.pf(" ok = ok && !nodes.refused;\n") + } g.pf(" allocator.free( allocator.context, directory );\n") g.pf(" return ok;\n}\n\n") } @@ -1066,7 +1099,7 @@ func (g *tableGen) emitRootNodeDispatch(st *ir.Struct) { g.pf("// this build cannot name — which keeps its index and reads null. A BYTE\n") g.pf("// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5),\n") g.pf("// which is the one answer the record's LENGTH decides.\n") - if g.anyMap { + if g.anyExtent { g.pf("// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8),\n") g.pf("// so a record's storage is its type's PLUS N x sizeof( Entry ) at every\n") g.pf("// depth, summed from the FRAMING: N is framing and not a value, and this\n") @@ -1074,23 +1107,23 @@ func (g *tableGen) emitRootNodeDispatch(st *ir.Struct) { } anyExtent := false for _, t := range reachable { - if g.anyMap && g.hasMapExtent(t) { + if g.anyExtent && g.hasExtent(t) { anyExtent = true } } // THE BODY IS ONLY A PARAMETER WHERE A MAP RIDES IN AN EXTENT: a root that // can name no such record answers from the type id and the length, exactly // as it did before the construct existed. - g.pf("inline int64_t %sNodeStorage( uint64_t type_id, %sint64_t length )\n{\n", n, g.nodeStorageBody(anyExtent)) + g.pf("inline int64_t %sNodeStorage( uint64_t type_id, %sint64_t length%s )\n{\n", n, g.nodeStorageBody(anyExtent), g.nodeStorageTail(anyExtent)) if len(blobs) == 0 { g.pf(" (void) length; // no byte buffer below this root: every node's storage is its type's\n") } g.pf(" switch ( type_id )\n {\n") for _, t := range reachable { - if g.anyMap && g.hasMapExtent(t) { + if g.anyExtent && g.hasExtent(t) { g.pf(" case 0x%016xull: // %s\n {\n", ir.TableWireId(t.Name), t.Name) g.pf(" int64_t extent = 0;\n") - g.pf(" if ( !%sWireExtent( body, length, extent ) ) { return kTableNodeRefused; }\n", t.Name) + g.pf(" if ( !%sWireExtent( body, length, extent, ids, reason ) ) { return kTableNodeRefused; }\n", t.Name) g.pf(" return TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + extent );\n }\n", t.Name) continue } @@ -1124,7 +1157,7 @@ func (g *tableGen) emitRootNodeDispatch(st *ir.Struct) { } g.pf(" default: break;\n }\n}\n\n") - if g.anyMap { + if g.anyExtent { g.pf("// %sNodeRecordBytes: one record's OWN storage, before the extent its maps\n", n) g.pf("// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins.\n") g.pf("inline int64_t %sNodeRecordBytes( uint64_t type_id )\n{\n", n) @@ -1163,14 +1196,18 @@ func (g *tableGen) emitRootNodeDispatch(st *ir.Struct) { g.pf("// %sNodeBody: PASS TWO's half — decode one record's body into the storage it\n", n) g.pf("// already owns.\n") g.pf("inline void %sNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at )\n{\n", n) - if g.anyMap { - g.pf(" // the node's own EXTENT, where its maps' entry arrays are carved from,\n") - g.pf(" // PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8). The tool's\n") - g.pf(" // path carries a worker instead: there the entries are the arena's.\n") - g.pf(" TableMapCarve carve;\n") + if g.anyExtent { + g.pf(" // the node's own EXTENT, where its lists' and maps' arrays are carved\n") + g.pf(" // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9).\n") + g.pf(" // The tool's path carries a worker instead: there the arrays are the\n") + g.pf(" // arena's.\n") + g.pf(" TableExtentCarve carve;\n") g.pf(" carve.worker = nodes.worker;\n") g.pf(" if ( carve.worker == NULL )\n {\n") - g.pf(" const int64_t storage = %sNodeStorage( type_id, %sr.size );\n", n, g.nodeStorageReader(st)) + if g.rootHasExtent(st) { + g.pf(" TableRefuseReason reason = count_over_length; // pass one already refused what this could refuse\n") + } + g.pf(" const int64_t storage = %sNodeStorage( type_id, %sr.size%s );\n", n, g.nodeStorageReader(st), g.nodeStorageReaderTail(st)) g.pf(" const int64_t record = storage > 0 ? %sNodeRecordBytes( type_id ) : 0;\n", n) g.pf(" carve.at = at + record;\n") g.pf(" carve.left = storage > record ? storage - record : 0;\n") @@ -1199,7 +1236,7 @@ func (g *tableGen) emitRootNodeDispatch(st *ir.Struct) { g.pf(" case %s: if ( r.size > 0 ) { memcpy( at + kTableBlobHeader, r.buffer, (size_t) r.size ); } break; // *%s\n", b.constant, b.word) } g.pf(" default: break;\n }\n") - if g.anyMap { + if g.anyExtent { g.pf(" nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done\n") } g.pf("}\n\n") @@ -1452,14 +1489,14 @@ func (g *tableGen) emitVariableLoadMeasure(st *ir.Struct, message bool) { // the connection's rather than the wire's, so there is no trailer to // locate and no stray-byte rule between a terminator and a first // entry — the message's last byte IS the body's terminator. - g.pf("inline int64_t %sLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL )\n{\n", n) + g.pf("inline int64_t %sLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL%s )\n{\n", n, g.loadMeasureReasonParam()) g.pf(" TableReport ignored;\n") g.pf(" if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; }\n") g.pf(" const TableIdTable & ids_table = vocabulary.table;\n") g.pf(" const uint8_t * const wire = message + 1;\n") g.pf(" const int64_t wire_bytes = message_bytes - 1;\n") } else { - g.pf("inline int64_t %sLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL )\n{\n", n) + g.pf("inline int64_t %sLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL%s )\n{\n", n, g.loadMeasureReasonParam()) g.pf(" TableReport ignored;\n") g.pf(" TableIdTable ids_table;\n") g.pf(" int64_t body_bytes = 0;\n") @@ -1472,16 +1509,23 @@ func (g *tableGen) emitVariableLoadMeasure(st *ir.Struct, message bool) { g.pf(" const int64_t wire_bytes = body_bytes;\n") } g.pf(" TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table );\n") - g.emitRootDataBytes(st, " ", "return -1;") + refuse := "return -1;" + if g.anyExtent { + // A -1 CARRIES A REASON (docs/SPEC-TABLES.md §6.5), as an enum + // out-parameter, and a refusal moves no counter + g.pf(" TableRefuseReason reason = count_over_length;\n") + refuse = "if ( reason_out != NULL ) { *reason_out = reason; } return -1;" + } + g.emitRootDataBytes(st, " ", refuse) g.pf(" int64_t records = 0;\n") g.pf(" uint64_t type_id = 0;\n") g.pf(" const uint8_t * body = NULL;\n") g.pf(" int64_t length = 0;\n") g.pf(" while ( TableNodeScanNext( scan, type_id, body, length ) )\n {\n") g.pf(" records++;\n") - g.pf(" int64_t storage = %sNodeStorage( type_id, %slength );\n", n, g.nodeStorageArg(st)) - if g.anyMap { - g.pf(" if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8)\n") + g.pf(" int64_t storage = %sNodeStorage( type_id, %slength%s );\n", n, g.nodeStorageArg(st), g.nodeStorageArgTail(st)) + if g.anyExtent { + g.pf(" if ( storage == kTableNodeRefused ) { %s } // an N the record's framing cannot carry (§2.8, §2.9)\n", refuse) } g.pf(" if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none\n") g.pf(" }\n") @@ -1544,6 +1588,9 @@ func (g *tableGen) emitVariableLoad(st *ir.Struct, message bool) { g.pf(" const uint8_t * body = NULL;\n") g.pf(" int64_t length = 0;\n\n") g.pf(" // the record count and the data bytes, from the FRAMING alone\n") + if g.anyExtent { + g.pf(" TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed\n") + } g.emitRootDataBytes(st, " ", "out->malformed = true; return NULL;") g.pf(" int64_t records = 0;\n") g.pf(" {\n") @@ -1551,8 +1598,8 @@ func (g *tableGen) emitVariableLoad(st *ir.Struct, message bool) { g.pf(" TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table );\n") g.pf(" while ( TableNodeScanNext( scan, type_id, body, length ) )\n {\n") g.pf(" records++;\n") - g.pf(" int64_t storage = %sNodeStorage( type_id, %slength );\n", n, g.nodeStorageArg(st)) - if g.anyMap { + g.pf(" int64_t storage = %sNodeStorage( type_id, %slength%s );\n", n, g.nodeStorageArg(st), g.nodeStorageArgTail(st)) + if g.anyExtent { g.pf(" if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; }\n") } g.pf(" if ( storage > 0 ) { data += storage; }\n") @@ -1572,7 +1619,7 @@ func (g *tableGen) emitVariableLoad(st *ir.Struct, message bool) { g.pf(" // resolves whichever way it points. It reads no body.\n") g.pf(" {\n") g.pf(" TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table );\n") - if g.anyMap { + if g.anyExtent { g.pf(" int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( %s ) ) + root_extent );\n", n) } else { g.pf(" int64_t used = TableAlignUp64( (int64_t) sizeof( %s ) );\n", n) @@ -1580,8 +1627,9 @@ func (g *tableGen) emitVariableLoad(st *ir.Struct, message bool) { g.pf(" int64_t k = 0;\n") g.pf(" int32_t unknown_records = 0; // counted once the scan is known whole\n") g.pf(" while ( TableNodeScanNext( scan, type_id, body, length ) )\n {\n") - g.pf(" int64_t storage = %sNodeStorage( type_id, %slength );\n", n, g.nodeStorageArg(st)) + g.pf(" int64_t storage = %sNodeStorage( type_id, %slength%s );\n", n, g.nodeStorageArg(st), g.nodeStorageArgTail(st)) g.pf(" if ( storage <= 0 )\n {\n") + g.pf(" // a record whose type id this build cannot name KEEPS ITS\n") g.pf(" // INDEX, is counted once here and not once per pointer, and\n") g.pf(" // every reference to it reads null (§3.1)\n") @@ -1618,8 +1666,8 @@ func (g *tableGen) emitVariableLoad(st *ir.Struct, message bool) { g.pf(" // against a numbering already known good or already known bad\n") g.pf(" TableReader r( wire, wire_bytes, out, &ids_table );\n") g.pf(" r.nested = false; // the ROOT body, the one that carries the node table\n") - if g.anyMap { - g.pf(" TableMapCarve root_carve;\n") + if g.anyExtent { + g.pf(" TableExtentCarve root_carve;\n") g.pf(" root_carve.at = region + TableAlignUp64( (int64_t) sizeof( %s ) );\n", n) g.pf(" root_carve.left = root_extent;\n") g.pf(" nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's\n") @@ -1627,3 +1675,13 @@ func (g *tableGen) emitVariableLoad(st *ir.Struct, message bool) { g.pf(" %sLoadBody( r, nodes, *root );\n", n) g.pf(" return root;\n}\n\n") } + +// loadMeasureReasonParam is the out-parameter a LoadMeasure takes where a unit +// has an extent (docs/SPEC-TABLES.md §6.5): the reason a -1 carries. A unit +// with neither a list nor a map keeps the signature it always had. +func (g *tableGen) loadMeasureReasonParam() string { + if g.anyExtent { + return ", TableRefuseReason * reason_out = NULL" + } + return "" +} diff --git a/internal/tablecook/check.go b/internal/tablecook/check.go index 82a6e982e..a32bc66a4 100644 --- a/internal/tablecook/check.go +++ b/internal/tablecook/check.go @@ -142,6 +142,19 @@ type scan struct { buf []byte dir []DirectoryEntry pointers int + // THE NODE UNDER THE SCAN, for §7.4's element-array clause: an unbounded + // array's slot must point inside its holder's own extent, so the walk + // carries where that node begins and ends, and the arrays it has already + // placed there, so no two overlap. Both are reset per node. + base int64 + extent int64 + arrays []arrayRange +} + +// arrayRange is one element array a list slot placed inside the node under +// the scan, in region offsets. +type arrayRange struct { + start, end int64 } // node walks one directory entry. A BYTE BUFFER's node has no fields to walk @@ -160,6 +173,7 @@ func (s *scan) node(base, extent int64, typeId uint64, st *ir.Struct) error { } return nil } + s.base, s.extent, s.arrays = base, base+extent, s.arrays[:0] return s.record(base, st) } @@ -187,6 +201,15 @@ func (s *scan) field(at int64, f *ir.Field) error { } value := pieces[0] switch { + case f.IsMap(): + // §7.4's MAP-SLOT clause (docs/SPEC-TABLES.md §2.8, §7.4) is + // schema#380's next PR, beside the tool's cook half, so a cook that + // holds a map slot is refused HERE, by name, at the node that holds + // it, rather than walked past: a slot the scan cannot bound is a slot + // a forgery could steer through. The C++ reference reads it (--lang + // cpp). A cook of a unit that declares a map SOMEWHERE ELSE checks as + // any other does, because the scan meets no such slot. + return fmt.Errorf("a map slot, and `cook-check` carries no map-slot clause yet (docs/SPEC-TABLES.md §7.4): the C++ reference reads the cook (--lang cpp), and the tool's clause is schema#380's next PR") case f.Type.Pointer && f.Array == ir.ArrayNone: return s.ref(value.Offset, f) case f.Type.Kind == ir.TString, f.Type.Kind == ir.TBytes: @@ -198,10 +221,54 @@ func (s *scan) field(at int64, f *ir.Field) error { return err } return s.companion(pieces[1].Offset, f.ArrayBound, "used count") + case f.Array == ir.ArrayList: + return s.list(value.Offset, f) } return s.element(value.Offset, f) } +// list is §7.4's ELEMENT-ARRAY clause (docs/SPEC-TABLES.md §2.9, §7.4): an +// unbounded array's sixteen-byte slot holds an int64 self-relative delta to +// its element array and an int32 count, and the array must sit INSIDE THE +// HOLDER'S OWN EXTENT, meaning containment, alignment, fit, and no overlap with any +// other array already placed in that node, before the elements' own slots, +// companions and tags are walked as a bounded array's are. There is no fifth +// clause, because there are no keys and no order. The check reads those four +// facts and not the offset the layout rule computes, so the layout rule stays +// independent of the check exactly as the pack order does. +func (s *scan) list(at int64, f *ir.Field) error { + delta := int64(s.ord.Uint64(s.buf[at:])) + count := int64(int32(s.ord.Uint32(s.buf[at+8:]))) + if count < 0 { + return fmt.Errorf("the count is %d, and an extent is never negative", count) + } + if delta == RefNull { + if count != 0 { + return fmt.Errorf("the reference is null and the count is %d: an empty list is the only list a null names", count) + } + return nil + } + if count == 0 { + return fmt.Errorf("the count is 0 and the reference is not null: an empty list's reference is null in every encoding") + } + size, align := ir.ListElementLayout(s.m.Unit, f) + start := at + delta + end := start + count*size + if start < s.base || end > s.extent { + return fmt.Errorf("the element array runs [%d, %d) and its holder's extent is [%d, %d): the array leaves the node", start, end, s.base, s.extent) + } + if start%align != 0 { + return fmt.Errorf("the element array starts at %d, which is not aligned to %d", start, align) + } + for _, other := range s.arrays { + if start < other.end && other.start < end { + return fmt.Errorf("the element array [%d, %d) overlaps another array [%d, %d) in the same node", start, end, other.start, other.end) + } + } + s.arrays = append(s.arrays, arrayRange{start, end}) + return s.slots(start, f, count) +} + // companion checks one count companion against its DECLARED bound. A negative // one is refused too: a count is an extent and an extent is never negative, and // a walker handed one indexes backwards out of the region. diff --git a/internal/tablecook/list_test.go b/internal/tablecook/list_test.go new file mode 100644 index 000000000..99701342c --- /dev/null +++ b/internal/tablecook/list_test.go @@ -0,0 +1,131 @@ +package tablecook_test + +import ( + "encoding/binary" + "strings" + "testing" + + "github.com/mas-bandwidth/schema/v2/internal/tablecook" + "github.com/mas-bandwidth/schema/v2/internal/tabletext" + "github.com/mas-bandwidth/schema/v2/ir" +) + +// §7.4's ELEMENT-ARRAY CLAUSE (docs/SPEC-TABLES.md §2.9, §7.4): an unbounded +// array's slot must point its array inside the holder's own extent, aligned, +// fitting, and overlapping no other array in that node. The tool cannot cook a +// list yet, so the cook under test is assembled BY HAND from §7.1 and §7.2: +// one root node, `Ints` from tables/lists, whose sixteen-byte slot names a +// three-element array laid after the record. + +// intsCook writes a cook of one Ints root with the given slot and count. The +// record is 24 bytes (the slot, then `after` and its padding), the array of +// three int32 follows at 24, and the data part rounds to 40. +func intsCook(u *ir.Unit, delta int64, count int32) []byte { + const header, data, attrib = int64(64), int64(40), int64(16) + out := make([]byte, header+data+attrib) + le := binary.LittleEndian + le.PutUint64(out[0:], tablecook.Magic) + le.PutUint64(out[8:], ir.BuildVersion(u)) + le.PutUint64(out[16:], tablecook.ByteOrderLittle) + le.PutUint64(out[24:], uint64(data)) + le.PutUint64(out[32:], uint64(attrib)) + le.PutUint64(out[40:], 8) + record := out[header:] + le.PutUint64(record[0:], uint64(delta)) + le.PutUint32(record[8:], uint32(count)) + le.PutUint32(record[16:], 5) // after + for i := range 3 { + le.PutUint32(record[24+i*4:], uint32(10*(i+1))) + } + dir := out[header+data:] + le.PutUint64(dir[0:], 0) + le.PutUint64(dir[8:], ir.TableTypeId("Ints")) + return out +} + +// TestCookCheckListSlot: the clause reads CONTAINMENT, ALIGNMENT, FIT and NO +// OVERLAP, and nothing else, and a null reference is an empty list and only +// that. The Makefile's negative control drops the containment test through an +// overlay and requires this test to go red. +func TestCookCheckListSlot(t *testing.T) { + u := unit(t, "../../tables/lists") + m := tabletext.NewModel(u) + + res, err := tablecook.Check(m, intsCook(u, 24, 3)) + if err != nil { + t.Fatalf("a cook whose list slot names its array inside the node was refused: %v", err) + } + if res.Root != "Ints" || res.Nodes != 1 { + t.Fatalf("checked the wrong shape: %+v", res) + } + if _, err := tablecook.Check(m, intsCook(u, 0, 0)); err != nil { + t.Fatalf("an empty list, null reference and zero count, was refused: %v", err) + } + + cases := []struct { + name string + delta int64 + count int32 + want string + }{ + {"the array leaves the node", 40, 3, "leaves the node"}, + {"the array leaves the region", 4096, 3, "leaves the node"}, + {"the array is not aligned", 26, 3, "not aligned"}, + {"the count does not fit", 24, 5, "leaves the node"}, + {"the count is negative", 24, -1, "never negative"}, + {"a null reference with a count", 0, 3, "empty list"}, + {"a reference with no count", 24, 0, "reference is not null"}, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + _, err := tablecook.Check(m, intsCook(u, c.delta, c.count)) + if err == nil { + t.Fatalf("FAILED: cook-check accepted a list slot that %s", c.name) + } + if !strings.Contains(err.Error(), c.want) { + t.Fatalf("refused, but not on the element-array clause: %v", err) + } + }) + } +} + +// squadCook writes a cook of one Squad root from tables/lists, the holder of +// `roster map[uint8]Item`: a 24-byte record, the sixteen-byte map slot at +// null and `name` after it, and one directory entry. +func squadCook(u *ir.Unit) []byte { + const header, data, attrib = int64(64), int64(24), int64(16) + out := make([]byte, header+data+attrib) + le := binary.LittleEndian + le.PutUint64(out[0:], tablecook.Magic) + le.PutUint64(out[8:], ir.BuildVersion(u)) + le.PutUint64(out[16:], tablecook.ByteOrderLittle) + le.PutUint64(out[24:], uint64(data)) + le.PutUint64(out[32:], uint64(attrib)) + le.PutUint64(out[40:], 8) + le.PutUint32(out[header+16:], 7) // name + dir := out[header+data:] + le.PutUint64(dir[0:], 0) + le.PutUint64(dir[8:], ir.TableTypeId("Squad")) + return out +} + +// TestCookCheckMapSlotRefusedByName: §7.4's map-slot clause is schema#380's +// next PR, so the scan refuses a map slot BY NAME where it meets one, naming +// the field, the reference that reads it and the PR that lands the clause, +// and a cook of a map-free root in the same unit checks as any other does. +func TestCookCheckMapSlotRefusedByName(t *testing.T) { + u := unit(t, "../../tables/lists") + m := tabletext.NewModel(u) + _, err := tablecook.Check(m, squadCook(u)) + if err == nil { + t.Fatalf("FAILED: cook-check walked past a map slot it has no clause for") + } + for _, want := range []string{"Squad.roster", "schema#380", "cpp", "§7.4"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("the refusal does not name %q: %v", want, err) + } + } + if _, err := tablecook.Check(m, intsCook(u, 24, 3)); err != nil { + t.Fatalf("a map-free root in a unit that declares a map elsewhere was refused: %v", err) + } +} diff --git a/ir/tablelist.go b/ir/tablelist.go index 243cad97c..8dbbcd856 100644 --- a/ir/tablelist.go +++ b/ir/tablelist.go @@ -52,3 +52,18 @@ func ListFields(u *Unit) []string { func (f *Field) CountedOnWire() bool { return f != nil && (f.Array == ArrayCounted || f.Array == ArrayList) } + +// ListElementLayout is the storage one element of a `[]T` takes and the +// alignment its array is laid at (docs/SPEC-TABLES.md §2.9): a TableRef for a +// `[]*T`, and the element type's own size and alignment otherwise. The element +// array in a holder's node extent is `count × size` at `align`, and this is +// the one place both the C++ writer and the tool's cook-check take the two +// numbers from. +func ListElementLayout(u *Unit, f *Field) (size, align int64) { + single := *f + single.Array = ArrayNone + single.ArrayBound = 0 + single.Type.Optional = false + p := elementPiece(u, &single) + return p.size, p.align +} diff --git a/tables/lists/Holders.schema b/tables/lists/Holders.schema new file mode 100644 index 000000000..2d35058af --- /dev/null +++ b/tables/lists/Holders.schema @@ -0,0 +1,44 @@ +package listdemo + +// WHERE ELSE A LIST RIDES IN A HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.9): a +// list of tables that hold lists, a list whose element holds a map, and a +// pointed-at node that holds a list of its own, so the node's storage is its +// record plus the extent its list commands. Lists and maps are ONE population +// in the extent, in declaration order, pre-order. + +table Sample { v int32 } + +table Row +{ + items []Sample + label int32 +} + +table Sheet +{ + rows []Row // a list of tables that hold lists: LoadMeasure sums at every depth + pinned *Row // a pointed-at holder: the node's extent is its list's +} + +table Item { count int32 } + +table Squad +{ + roster map[uint8]Item + name int32 +} + +table Army +{ + squads []Squad // an element that holds a MAP: the element array first, then each map + after int32 +} + +// AN UNREACHED NON-EMPTY LIST SLOT IS REFUSED by Cook and by Lock (§7.6): a +// counted array's slots past its live count are storage the walk does not +// reach, so a list with elements in one names storage the region will not hold +table Deck +{ + hands [..3]Row + after int32 +} diff --git a/tables/lists/Migrate.schema b/tables/lists/Migrate.schema new file mode 100644 index 000000000..c0255f356 --- /dev/null +++ b/tables/lists/Migrate.schema @@ -0,0 +1,21 @@ +package listdemo + +// THE MIGRATION ITSELF (docs/SPEC-TABLES.md §2.9): "the same bytes" is a claim +// about two schemas, so ONE content is declared twice, once at a bound and +// once unbounded, and one pinned wire is what both write byte for byte and +// both read into equal values. The bound is ABOVE the instance's count, so the +// row proves the framing and not the clamp. + +table Unit { v int32 } + +table Bounded +{ + items [..8]Unit + tag int32 +} + +table Unbounded +{ + items []Unit + tag int32 +} diff --git a/tables/lists/Report.schema b/tables/lists/Report.schema new file mode 100644 index 000000000..95f863e63 --- /dev/null +++ b/tables/lists/Report.schema @@ -0,0 +1,25 @@ +package listdemo + +// THE REPORT ROWS' ROOTS (docs/SPEC-TABLES.md §2.9, §4): a []uint8 that +// carries 100,000 elements, past 2^16 and so past any bound a schema on the +// page declares, so a reader that CLAMPS the count goes red; and two roots +// that differ only in the ELEMENT's kind, read one against the other, with a +// field AFTER the list so the parent can be seen to read on. + +table Bytes +{ + data []uint8 + after int32 +} + +table Ints +{ + values []int32 + after int32 +} + +table Floats +{ + values []float32 + after int32 +} diff --git a/tables/lists/Save.schema b/tables/lists/Save.schema new file mode 100644 index 000000000..c8cc3bd5b --- /dev/null +++ b/tables/lists/Save.schema @@ -0,0 +1,48 @@ +package listdemo + +// The UNBOUNDED ARRAY corpus (docs/SPEC-TABLES.md §2.9): §2.9's own example, +// the five element classes, and the report rows' roots. A list is a counted +// array whose count the DATA decides, and every instance the harness pins over +// these crosses the wire, the text and the cook. + +table Placement +{ + x float32 + y float32 + model uint32 +} + +table LogEntry { tick uint32 } + +table Save +{ + placements []Placement // as many as the world has + log []*LogEntry // pointer elements, two slots may name one node + scores []int32 // a scalar element is an element like any other +} + +enum Grade { A, B, C } + +flags Perm { Read, Write, Own } + +table Point +{ + x int32 + y int32 +} + +union Hit +{ + point Point + damage int32 +} + +// the element set is [..N]T's exactly: an enum, a flags mask and a union are +// elements as they are in a bounded array +table Mixed +{ + grades []Grade + perms []Perm + hits []Hit + bounds []int32 | min = 0, max = 100 // a bar attribute qualifies the ELEMENT, as on a [..N]T +} diff --git a/tables/lists/Shared.schema b/tables/lists/Shared.schema new file mode 100644 index 000000000..674812f46 --- /dev/null +++ b/tables/lists/Shared.schema @@ -0,0 +1,19 @@ +package listdemo + +// SHARING AND THE WALK ORDER (docs/SPEC-TABLES.md §2.9, §3.1): a []*T's +// elements are pointer slots, so two slots may name one node beside a null +// one, and a []*T DECLARED BEFORE a pointer field reaches a shared node first +// and numbers it first. A walk that grouped lists after the pointer fields +// would number the two the other way round and the pinned wire says so. + +table Photo +{ + width uint32 + height uint32 +} + +table Album +{ + photos []*Photo // BEFORE cover: the walk-order pin + cover *Photo // the same node, through a pointer field +} diff --git a/tables/lists/tables.baseline b/tables/lists/tables.baseline new file mode 100644 index 000000000..fa2a9eba5 --- /dev/null +++ b/tables/lists/tables.baseline @@ -0,0 +1,107 @@ +schema-tables-baseline 7 +package listdemo + +table Album + field photos id=0x40b1d94aff3ab130 kind=14 elem=17 type=Photo array=unbounded + field cover id=0xaa19a78e404dea20 kind=17 type=Photo + +table Army + field squads id=0x7848019b0c02a926 kind=14 elem=13 type=Squad array=unbounded + field after id=0xbf82010f6f71eae9 kind=4 + +table Bounded + field items id=0x3e7884bf4f412c6f kind=14 elem=13 type=Unit array=bounded bound=8 + field tag id=0x56d7ab194448a4f3 kind=4 + +table Bytes + field data id=0x855b556730a34a05 kind=14 elem=6 array=unbounded + field after id=0xbf82010f6f71eae9 kind=4 + +table Deck + field hands id=0x81b46a69304ee2c9 kind=14 elem=13 type=Row array=bounded bound=3 + field after id=0xbf82010f6f71eae9 kind=4 + +table Floats + field values id=0x21277bcf1a4d67fb kind=14 elem=10 array=unbounded + field after id=0xbf82010f6f71eae9 kind=4 + +table Ints + field values id=0x21277bcf1a4d67fb kind=14 elem=4 array=unbounded + field after id=0xbf82010f6f71eae9 kind=4 + +table Item + field count id=0xb1e5e28e4479a274 kind=4 + +table LogEntry + field tick id=0x1e7683ef2ebc7684 kind=8 + +table Mixed + field grades id=0xd90a4e7682f799c5 kind=14 elem=7 enum=Grade array=unbounded + field perms id=0x4af2ed8470862ea8 kind=14 elem=9 flags=Perm array=unbounded + field hits id=0x732dfbcc9b0cf0bb kind=14 elem=15 union=Hit array=unbounded + field bounds id=0x52f60c4caef0b768 kind=14 elem=4 array=unbounded min=0 max=100 + +table Photo + field width id=0xdbdacd932fd1e9bf kind=8 + field height id=0x17720bf67d347222 kind=8 + +table Placement + field x id=0xaf63f54c86021707 kind=10 + field y id=0xaf63f44c86021554 kind=10 + field model id=0x9de543933e6e703a kind=8 + +table Point + field x id=0xaf63f54c86021707 kind=4 + field y id=0xaf63f44c86021554 kind=4 + +table Row + field items id=0x3e7884bf4f412c6f kind=14 elem=13 type=Sample array=unbounded + field label id=0x39f7fcec8fcb623d kind=4 + +table Sample + field v id=0xaf63eb4c86020609 kind=4 + +table Save + field placements id=0xd24733aa574d4b09 kind=14 elem=13 type=Placement array=unbounded + field log id=0x125073191daf5431 kind=14 elem=17 type=LogEntry array=unbounded + field scores id=0x01986b0b27400fb2 kind=14 elem=4 array=unbounded + +table Sheet + field rows id=0xa3a7061ff10a8138 kind=14 elem=13 type=Row array=unbounded + field pinned id=0x5f82477707ad620f kind=17 type=Row + +table Squad + field roster id=0x1c84390d304f4f42 kind=14 elem=13 array=map keykind=6 + field name id=0xc4bcadba8e631b86 kind=4 + +table Unbounded + field items id=0x3e7884bf4f412c6f kind=14 elem=13 type=Unit array=unbounded + field tag id=0x56d7ab194448a4f3 kind=4 + +table Unit + field v id=0xaf63eb4c86020609 kind=4 + +table ec07a2f760550a91.1c84390d304f4f42 + field key id=0x3dc94a19365b10ec kind=6 + field value id=0x7ce4fd9430e80cea kind=13 type=Item + +enum Grade + variant A id=0xaf63fc4c860222ec + variant B id=0xaf63ff4c86022805 + variant C id=0xaf63fe4c86022652 + +flags Perm + variant Read bit=0 + variant Write bit=1 + variant Own bit=2 + +union Hit + arm point id=0x73feab3544c345b1 payload=Point + arm damage id=0x7f6308be8ab37fc0 kind=4 + +## history +### 2026-09-05 (UTC) — first baseline: the unbounded array corpus +- baseline created over 20 tables — data written BEFORE this point is not covered by it + +### 2026-09-05 (UTC) — Deck: the unreached-slot control +- no compatibility-affecting edits; the wire absorbs the rest diff --git a/test/tables/lists_main.cpp b/test/tables/lists_main.cpp new file mode 100644 index 000000000..85039ab10 --- /dev/null +++ b/test/tables/lists_main.cpp @@ -0,0 +1,1484 @@ +// THE LIST GATE (docs/SPEC-TABLES.md §2.9). One binary over the `tables/lists` +// unit: the builder's three, the four writing walks in INDEX order, the node +// extent a region and a cook carry, every reader rule, the migration golden, +// and the negative controls §2.9 names. Each row here is one of them, and the +// comment says which sabotage it turns red. +// +// Compiled WITHOUT the serialize include path: the Table headers stand alone. +// +// schema_test_lists every battery +// schema_test_lists measure-refusals the six LoadMeasure refusals alone +// (make tables-list-measure-refusals) + +#include +#include +#include + +#include + +#include "SaveTable.h" +#include "SharedTable.h" +#include "HoldersTable.h" +#include "MigrateTable.h" +#include "ReportTable.h" +#include "wirebuilder.h" + +using namespace listdemo; + +static int failures = 0; + +// ---- THE ALLOCATION AUDIT (docs/SPEC-TABLES.md §2.9, §6.5) ---- +// +// The reading path allocates nothing of its own: LoadMeasure, Load into the +// caller's region, the const indexing and iteration, and Open. Every +// allocation the program makes through operator new is counted here, and the +// audit requires the count to stay where it was across all of them. CONTROL: +// an allocation is planted in Load or in the const indexing, and the audit +// goes red. +static long long allocations = 0; + +void * operator new( size_t bytes ) +{ + allocations++; + void * p = malloc( bytes != 0 ? bytes : 1 ); + if ( p == NULL ) { abort(); } + return p; +} +void operator delete( void * p ) noexcept { free( p ); } +void operator delete( void * p, size_t ) noexcept { free( p ); } + +#define CHECK( condition ) \ + do \ + { \ + if ( !( condition ) ) \ + { \ + printf( "FAIL %s:%d: %s\n", __FILE__, __LINE__, #condition ); \ + fflush( stdout ); \ + failures++; \ + } \ + } while ( 0 ) + +#define CHECK_EQ( actual, expected ) \ + do \ + { \ + const long long a_ = (long long) ( actual ); \ + const long long e_ = (long long) ( expected ); \ + if ( a_ != e_ ) \ + { \ + printf( "FAIL %s:%d: %s = %lld, want %lld\n", \ + __FILE__, __LINE__, #actual, a_, e_ ); \ + fflush( stdout ); \ + failures++; \ + } \ + } while ( 0 ) + +// ---- every allocation sized from a measure goes through here ---- +// +// A measure is an answer from the code under test, so a region sized from one +// is the single place a broken measure reaches the allocator. The ceiling is +// CHECKED first, and a measure past it is a red CHECK on every platform rather +// than a call to calloc (test/tables/maps_main.cpp says why). +// +// 256 MiB: the largest measure this corpus produces is the 100,000-element +// clamp control's, well under a megabyte. + +static const int64_t kMeasureCeiling = 256 * 1024 * 1024; + +static void * measured_calloc( int64_t measure, int64_t extra, const char * expr, const char * file, int line ) +{ + if ( measure < 0 || measure > kMeasureCeiling ) + { + printf( "FAIL %s:%d: %s = %lld, past the %lld byte measure ceiling\n", + file, line, expr, (long long) measure, (long long) kMeasureCeiling ); + failures++; + return NULL; + } + return calloc( 1, (size_t) ( measure + extra ) ); +} + +#define MEASURED_CALLOC( measure, extra ) \ + measured_calloc( ( measure ), ( extra ), #measure, __FILE__, __LINE__ ) + +// ---- the shared golden wire (docs/SPEC-TABLES.md §3) ---- +// +// The C++ reference is the writer: these instances' encodings are pinned into +// testdata/wire/tables/.bin. A break here under an unchanged schema is +// stop-the-line, never a quiet re-pin. SCHEMA_UPDATE_WIRE_GOLDENS=1 rewrites +// them deliberately (make update-goldens). It answers whether the bytes are +// the pinned ones, because a COOK the pin refused is a file no Open may trust: +// Open matches the header and points (§7), so a layout sabotage that reached +// it would crash rather than fail a CHECK. + +static bool pin_golden( const char * name, const uint8_t * data, int64_t bytes ) +{ + char path[256]; + snprintf( path, sizeof( path ), "testdata/wire/tables/%s.bin", name ); + if ( getenv( "SCHEMA_UPDATE_WIRE_GOLDENS" ) ) + { + FILE * f = fopen( path, "wb" ); + if ( f == NULL ) { printf( "FAIL cannot write %s\n", path ); fflush( stdout ); failures++; return false; } + fwrite( data, 1, (size_t) bytes, f ); + fclose( f ); + return true; + } + FILE * f = fopen( path, "rb" ); + if ( f == NULL ) + { + printf( "FAIL missing table wire golden %s (run: make update-goldens)\n", path ); + fflush( stdout ); + failures++; + return false; + } + static uint8_t expected[1u << 20]; + const size_t n = fread( expected, 1, sizeof( expected ), f ); + fclose( f ); + if ( (int64_t) n != bytes || memcmp( expected, data, n ) != 0 ) + { + printf( "FAIL table wire golden %s: %lld bytes written, %lld pinned\n", + name, (long long) bytes, (long long) n ); + fflush( stdout ); + failures++; + return false; + } + return true; +} + +// A COOK IS WRITTEN FOR THE BUILD THAT OPENS IT (docs/SPEC-TABLES.md §7): the +// host's own order, so the round trip below holds on the big-endian leg too. +static TableByteOrder host_byte_order() +{ + const uint16_t probe = 1; + return *(const uint8_t *) &probe == 1 ? TableByteOrder::Little : TableByteOrder::Big; +} + +// THE COOKS `schema cook-check` READS (docs/SPEC-TABLES.md §7.4): when the +// Makefile names a directory, the cooks this gate writes are saved there, and +// beside one of them a FORGERY whose list slot points its array past the +// holder's extent, which the tool must refuse. +static void save_cook( const char * name, const void * data, int64_t bytes ) +{ + const char * dir = getenv( "SCHEMA_LIST_COOK_DIR" ); + if ( dir == NULL ) { return; } + char path[512]; + snprintf( path, sizeof( path ), "%s/%s.cook", dir, name ); + FILE * f = fopen( path, "wb" ); + if ( f == NULL ) { printf( "FAIL cannot write %s\n", path ); failures++; return; } + fwrite( data, 1, (size_t) bytes, f ); + fclose( f ); +} + +static void report_silent( const TableReport & r, const char * where ) +{ + if ( r.unknown != 0 || r.kind_mismatch != 0 || r.clamped != 0 || r.duplicate != 0 || r.malformed ) + { + printf( "FAIL %s: the report is not silent (unknown %d, kind_mismatch %d, clamped %d, duplicate %d, malformed %d)\n", + where, r.unknown, r.kind_mismatch, r.clamped, r.duplicate, (int) r.malformed ); + failures++; + } +} + +static void reports_agree( const TableReport & a, const TableReport & b ) +{ + CHECK_EQ( a.unknown, b.unknown ); + CHECK_EQ( a.kind_mismatch, b.kind_mismatch ); + CHECK_EQ( a.clamped, b.clamped ); + CHECK_EQ( a.duplicate, b.duplicate ); + CHECK_EQ( (int) a.malformed, (int) b.malformed ); +} + +// ---- the instances (docs/SPEC-TABLES.md §2.9) ---- + +// §2.9's own example: three placements, a []*T whose two slots name one node +// beside a null slot, and three scalars: `list_tables`. +static void build_save( SaveBuilder & b ) +{ + Save * save = b.GetRoot(); + for ( int i = 0; i < 3; i++ ) + { + Placement * placement = SavePlacementsAdd( b.main, save->placements ); + CHECK( placement != NULL ); + placement->x = 1.0f + (float) i; + placement->y = 2.0f * (float) i; + placement->model = (uint32_t) ( 3 + i ); + } + // a pointer element: Add hands back the SLOT at null, Emplace fills it as it + // fills any pointer slot, and a second slot holds the same reference + TableRef * slot = SaveLogAdd( b.main, save->log ); + CHECK( slot != NULL ); + LogEntry * shared = LogEntryEmplace( b.main, *slot ); + CHECK( shared != NULL ); + shared->tick = 7; + *SaveLogAdd( b.main, save->log ) = *slot; // two slots, one node + SaveLogAdd( b.main, save->log ); // and a null slot + *SaveScoresAdd( b.main, save->scores ) = 10; + *SaveScoresAdd( b.main, save->scores ) = 20; + *SaveScoresAdd( b.main, save->scores ) = 30; +} + +static uint8_t wire_tables[1u << 16]; +static int64_t bytes_tables = 0; + +// ---- the writer (docs/SPEC-TABLES.md §2.9) ---- + +static void test_writer() +{ + { + SaveBuilder b; + build_save( b ); + const int64_t measured = SaveMeasure( b ); + bytes_tables = SaveSave( b, wire_tables, sizeof( wire_tables ) ); + CHECK_EQ( measured, bytes_tables ); // measure == save over a list is a check on the arithmetic alone (§2.9) + pin_golden( "list_tables", wire_tables, bytes_tables ); + // MEASURE EQUALS SAVE AT EXACT CAPACITY, and one short of it refuses + static uint8_t exact[1u << 16]; + CHECK_EQ( SaveSave( b, exact, measured ), measured ); + CHECK( memcmp( exact, wire_tables, (size_t) measured ) == 0 ); + CHECK_EQ( SaveSave( b, exact, measured - 1 ), -1 ); + } + { + // CONTROL: the writer emits the elements OUT OF ORDER. `list_scalars` + // meets it: the byte compare against its pinned wire goes red while + // measure == save still holds. + SaveBuilder b; + Save * save = b.GetRoot(); + *SaveScoresAdd( b.main, save->scores ) = 10; + *SaveScoresAdd( b.main, save->scores ) = 20; + *SaveScoresAdd( b.main, save->scores ) = 30; + static uint8_t wire[1u << 16]; + const int64_t measured = SaveMeasure( b ); + const int64_t n = SaveSave( b, wire, sizeof( wire ) ); + CHECK_EQ( measured, n ); + pin_golden( "list_scalars", wire, n ); + } + { + // `list_empty`: an EMPTY list beside a full one elides under §3's + // by-value rule, and a fresh Save is the empty wire + SaveBuilder b; + Save * save = b.GetRoot(); + SavePlacementsAdd( b.main, save->placements )->model = 1; + SavePlacementsAdd( b.main, save->placements )->model = 2; + static uint8_t wire[1u << 16]; + const int64_t n = SaveSave( b, wire, sizeof( wire ) ); + CHECK_EQ( SaveMeasure( b ), n ); + pin_golden( "list_empty", wire, n ); + SaveBuilder fresh; + static uint8_t empty[64]; + CHECK_EQ( SaveSave( fresh, empty, sizeof( empty ) ), empty_wire_bytes ); + } + { + // CONTROL: `Save` emits a DEAD element. `list_erased` meets it, an + // erase from the MIDDLE with an add after it, so a writer that merely + // truncates still goes red, and the byte compare against the same + // five elements added directly says the sabotage is the skip and not + // the arithmetic. + SaveBuilder b; + Save * save = b.GetRoot(); + Placement * held[5] = { NULL, NULL, NULL, NULL, NULL }; + for ( int i = 0; i < 5; i++ ) + { + held[i] = SavePlacementsAdd( b.main, save->placements ); + held[i]->model = (uint32_t) ( 100 + i ); + } + CHECK( SavePlacementsErase( b.arena, save->placements, held[2] ) ); + CHECK( !SavePlacementsErase( b.arena, save->placements, held[2] ) ); // already erased: false + Placement foreign; + CHECK( !SavePlacementsErase( b.arena, save->placements, &foreign ) ); // not this list's: false + SavePlacementsAdd( b.main, save->placements )->model = 105; + CHECK_EQ( save->placements.count, 5 ); + static uint8_t erased[1u << 16]; + const int64_t measured = SaveMeasure( b ); + const int64_t n = SaveSave( b, erased, sizeof( erased ) ); + CHECK_EQ( measured, n ); + pin_golden( "list_erased", erased, n ); + + SaveBuilder direct; + const uint32_t models[5] = { 100, 101, 103, 104, 105 }; + for ( int i = 0; i < 5; i++ ) { SavePlacementsAdd( direct.main, direct.GetRoot()->placements )->model = models[i]; } + static uint8_t straight[1u << 16]; + const int64_t m = SaveSave( direct, straight, sizeof( straight ) ); + CHECK_EQ( m, n ); + CHECK( memcmp( straight, erased, (size_t) n ) == 0 ); + } + { + // the five element classes are [..N]T's exactly: an ENUM, a FLAGS mask + // and a UNION are elements as they are in a bounded array, and a bar + // attribute qualifies the ELEMENT + MixedBuilder b; + Mixed * mixed = b.GetRoot(); + *MixedGradesAdd( b.main, mixed->grades ) = Grade::A; + *MixedGradesAdd( b.main, mixed->grades ) = Grade::C; + *MixedGradesAdd( b.main, mixed->grades ) = Grade::B; + *MixedPermsAdd( b.main, mixed->perms ) = Perm_Read | Perm_Write; + *MixedPermsAdd( b.main, mixed->perms ) = Perm_Own; + Hit * point = MixedHitsAdd( b.main, mixed->hits ); + point->type = HitType::Point; + PointReset( point->point ); + point->point.x = 1; + point->point.y = 2; + Hit * damage = MixedHitsAdd( b.main, mixed->hits ); + damage->type = HitType::Damage; + damage->damage = 7; + MixedHitsAdd( b.main, mixed->hits ); // a None element in its place + *MixedBoundsAdd( b.main, mixed->bounds ) = 0; + *MixedBoundsAdd( b.main, mixed->bounds ) = 50; + *MixedBoundsAdd( b.main, mixed->bounds ) = 100; + static uint8_t wire[1u << 16]; + const int64_t measured = MixedMeasure( b ); + const int64_t n = MixedSave( b, wire, sizeof( wire ) ); + CHECK_EQ( measured, n ); + pin_golden( "list_mixed", wire, n ); + + const int64_t need = MixedLoadMeasure( wire, n ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport report; + const Mixed * loaded = MixedLoad( region, need, wire, n, &report ); + CHECK( loaded != NULL ); + report_silent( report, "list_mixed" ); + if ( loaded != NULL ) + { + CHECK_EQ( loaded->grades.size(), 3 ); + CHECK( loaded->grades[1] == Grade::C ); + CHECK_EQ( loaded->perms.size(), 2 ); + CHECK_EQ( loaded->perms[0], Perm_Read | Perm_Write ); + CHECK_EQ( loaded->hits.size(), 3 ); + CHECK( loaded->hits[0].type == HitType::Point && loaded->hits[0].point.y == 2 ); + CHECK( loaded->hits[1].type == HitType::Damage && loaded->hits[1].damage == 7 ); + CHECK( loaded->hits[2].type == HitType::None ); + CHECK_EQ( loaded->bounds.size(), 3 ); + CHECK_EQ( loaded->bounds[2], 100 ); + } + free( region ); + } +} + +// ---- the builder's three (docs/SPEC-TABLES.md §2.9) ---- + +static void test_builder() +{ + SaveBuilder b; + Save * save = b.GetRoot(); + + // ADD hands the element back at its declared defaults + Placement * first = SavePlacementsAdd( b.main, save->placements ); + CHECK( first != NULL ); + CHECK( first->x == 0.0f && first->model == 0 ); + CHECK_EQ( save->placements.count, 1 ); + first->model = 11; + + // MORE THAN ONE SEGMENT: an element's address is stable for the arena's + // life, so a pointer handed back by an early Add survives every later one + for ( int i = 1; i < 200; i++ ) + { + Placement * p = SavePlacementsAdd( b.main, save->placements ); + CHECK( p != NULL ); + p->model = (uint32_t) ( 11 + i ); + } + CHECK_EQ( save->placements.count, 200 ); + CHECK_EQ( first->model, 11 ); // NOTHING EVER MOVES (§6.4) + + // ERASE by the element's own pointer, from the MIDDLE. EACH on the builder + // is INDEX order, live elements only + int32_t seen = 0; + Placement * third = NULL; + for ( Placement * p : SavePlacementsEach( b.arena, save->placements ) ) + { + if ( seen == 2 ) { third = p; } + seen++; + } + CHECK_EQ( seen, 200 ); + CHECK( third != NULL && third->model == 13 ); + CHECK( SavePlacementsErase( b.arena, save->placements, third ) ); + CHECK_EQ( save->placements.count, 199 ); + seen = 0; + for ( Placement * p : SavePlacementsEach( b.arena, save->placements ) ) + { + CHECK( p->model != 13 ); // the dead element is skipped + if ( seen == 2 ) { CHECK_EQ( p->model, 14 ); } // INDICES ARE NOT STABLE ACROSS AN ERASE + seen++; + } + CHECK_EQ( seen, 199 ); + + // and the const form agrees once locked: what was index 3 is index 2 + CHECK( b.Lock() ); + const Save * locked = b.AsConst(); + CHECK( locked != NULL ); + if ( locked != NULL ) + { + CHECK_EQ( locked->placements.size(), 199 ); + CHECK_EQ( locked->placements[2].model, 14 ); + CHECK_EQ( locked->placements[198].model, 210 ); + } + + // a []*T: Add hands back the SLOT at null + SaveBuilder p; + TableRef * slot = SaveLogAdd( p.main, p.GetRoot()->log ); + CHECK( slot != NULL && slot->value == 0 ); + CHECK_EQ( p.GetRoot()->log.count, 1 ); + LogEntry * entry = LogEntryEmplace( p.main, *slot ); + CHECK( entry != NULL ); + entry->tick = 3; + int32_t slots = 0; + for ( TableRef * s : SaveLogEach( p.arena, p.GetRoot()->log ) ) { CHECK( LogEntryAt( p.arena, *s ) == entry ); slots++; } + CHECK_EQ( slots, 1 ); +} + +// ---- the const form: a locked region, a loaded one, an opened cook ---- + +static void check_const_form( const Save * s, const char * where ) +{ + if ( s == NULL ) { printf( "FAIL %s: no root\n", where ); failures++; return; } + CHECK_EQ( s->placements.size(), 3 ); + if ( s->placements.size() == 3 ) + { + CHECK( s->placements[0].x == 1.0f ); + CHECK_EQ( s->placements[2].model, 5 ); + int i = 0; + for ( const Placement & p : s->placements ) { CHECK_EQ( p.model, 3 + i ); i++; } + CHECK_EQ( i, 3 ); + } + // a []*T's const operator[] answers the RESOLVED pointer: TWO SLOTS, ONE + // NODE, and a null slot answers NULL + CHECK_EQ( s->log.size(), 3 ); + if ( s->log.size() == 3 ) + { + const LogEntry * a = s->log[0]; + const LogEntry * again = s->log[1]; + CHECK( a != NULL && a == again ); + if ( a != NULL ) { CHECK_EQ( a->tick, 7 ); } + CHECK( s->log[2] == NULL ); + int i = 0; + for ( const LogEntry * e : s->log ) { if ( i < 2 ) { CHECK( e == a ); } else { CHECK( e == NULL ); } i++; } + CHECK_EQ( i, 3 ); + } + CHECK_EQ( s->scores.size(), 3 ); + if ( s->scores.size() == 3 ) { CHECK_EQ( s->scores[1], 20 ); } +} + +static void test_const_forms() +{ + SaveBuilder b; + build_save( b ); + CHECK( b.Lock() ); + check_const_form( b.AsConst(), "the locked region" ); + + // a locked region re-saves the same bytes as the builder did + static uint8_t again[1u << 16]; + const int64_t n = SaveSave( b.AsConst(), again, sizeof( again ) ); + CHECK_EQ( n, bytes_tables ); + CHECK( memcmp( again, wire_tables, (size_t) bytes_tables ) == 0 ); + + // LOAD into the caller's exact-sized region + const int64_t need = SaveLoadMeasure( wire_tables, bytes_tables ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport report; + const Save * loaded = SaveLoad( region, need, wire_tables, bytes_tables, &report ); + check_const_form( loaded, "a loaded region" ); + report_silent( report, "a loaded region" ); + const int64_t back = SaveSave( loaded, again, sizeof( again ) ); + CHECK_EQ( back, bytes_tables ); + CHECK( memcmp( again, wire_tables, (size_t) bytes_tables ) == 0 ); + + // LoadBuilder is the TOOL's path, and it produces EXACTLY the same report + SaveBuilder into; + TableReport tool; + CHECK( SaveLoadBuilder( into, wire_tables, bytes_tables, &tool ) ); + reports_agree( tool, report ); + CHECK_EQ( into.GetRoot()->placements.count, 3 ); + CHECK_EQ( into.GetRoot()->log.count, 3 ); + const int64_t relocked = SaveSave( into, again, sizeof( again ) ); + CHECK_EQ( relocked, bytes_tables ); + CHECK( memcmp( again, wire_tables, (size_t) bytes_tables ) == 0 ); + + // the COOK: a region written verbatim, opened O(1) and indexed in place + const int64_t cook_bytes = SaveCookMeasure( loaded ); + CHECK( cook_bytes > 0 ); + void * cooked = MEASURED_CALLOC( cook_bytes, 0 ); + if ( cooked == NULL ) { free( region ); return; } + CHECK( SaveCook( loaded, cooked, (uint64_t) cook_bytes, host_byte_order() ) ); + check_const_form( SaveOpen( cooked, (uint64_t) cook_bytes ), "an opened cook" ); + save_cook( "save", cooked, cook_bytes ); + // and two cooks of one instance are ONE artifact, from the region and from the builder alike + void * twice = MEASURED_CALLOC( cook_bytes, 0 ); + if ( twice == NULL ) { free( cooked ); free( region ); return; } + CHECK( SaveCook( loaded, twice, (uint64_t) cook_bytes, host_byte_order() ) ); + CHECK( memcmp( cooked, twice, (size_t) cook_bytes ) == 0 ); + CHECK_EQ( SaveCookMeasure( into ), cook_bytes ); + CHECK( SaveCook( into, twice, (uint64_t) cook_bytes, host_byte_order() ) ); + CHECK( memcmp( cooked, twice, (size_t) cook_bytes ) == 0 ); + free( twice ); + free( cooked ); + // the region is EXACT: one byte short is refused. Load zeroes the region + // it is handed before it refuses, so this probe is the block's last act. + TableReport short_report; + CHECK( SaveLoad( region, need - 1, wire_tables, bytes_tables, &short_report ) == NULL ); + free( region ); +} + +// ---- the reader's rules, each on a hand-made body (§2.9) ---- + +// an `Ints` body written FROM THE GRAMMAR: a kind 14 array of kind 4 (int32) +// elements, `after` behind it, and one knob each for the controls +struct IntsSpec +{ + int32_t n; // elements written + int64_t declared_n; // -1: n itself + uint8_t element_kind; // 0: int32's own kind + int64_t body_len; // -1: the body's own length; else a FORGED array body length + int32_t after; +}; + +static IntsSpec ints_spec() +{ + IntsSpec spec = { 3, -1, 0, -1, 777 }; + return spec; +} + +struct Wire +{ + uint8_t bytes[4096]; + int64_t size; +}; + +static Wire build_ints( const IntsSpec & spec ) +{ + WireBuilder b; + b.field( "values", 14 ); + if ( spec.body_len < 0 ) + { + const int64_t body = b.open_len(); + b.u8( spec.element_kind != 0 ? spec.element_kind : 4 ); + b.leb( spec.declared_n >= 0 ? (uint64_t) spec.declared_n : (uint64_t) spec.n ); + for ( int32_t i = 0; i < spec.n; i++ ) { b.u32( (uint32_t) ( 10 * ( i + 1 ) ) ); } + b.close_len( body ); + } + else + { + // a body TOO SHORT for its own header, or short of its count + b.leb( (uint64_t) spec.body_len ); + for ( int64_t i = 0; i < spec.body_len; i++ ) { b.u8( i == 0 ? 4 : 0 ); } + } + b.field( "after", 4 ); + b.u32( (uint32_t) spec.after ); + b.end(); + Wire w; + w.size = b.finish( w.bytes ); + return w; +} + +struct Verdict +{ + int32_t count; + TableReport report; + int32_t after; + int32_t decoded[3]; +}; + +static Verdict read_ints( const Wire & w ) +{ + Verdict v = { -1, TableReport(), 0, { 0, 0, 0 } }; + const int64_t need = IntsLoadMeasure( w.bytes, w.size ); + if ( need < 0 ) { v.count = -2; return v; } // the measure REFUSED the framing + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return v; } + const Ints * ints = IntsLoad( region, need, w.bytes, w.size, &v.report ); + if ( ints != NULL ) + { + v.count = ints->values.size(); + v.after = ints->after; + for ( int32_t i = 0; i < v.count && i < 3; i++ ) { v.decoded[i] = ints->values[i]; } + } + free( region ); + + // EVERY LOAD PATH PRODUCES ONE REPORT (§2.9): the two paths agree on every + // wire either of them decodes + IntsBuilder into; + TableReport t; + IntsLoadBuilder( into, w.bytes, w.size, &t ); + reports_agree( t, v.report ); + CHECK_EQ( into.GetRoot()->values.count, v.count < 0 ? 0 : v.count ); + return v; +} + +static void test_reader() +{ + { + const Verdict v = read_ints( build_ints( ints_spec() ) ); + CHECK_EQ( v.count, 3 ); + CHECK_EQ( v.decoded[2], 30 ); + CHECK_EQ( v.after, 777 ); + report_silent( v.report, "a good Ints wire" ); + } + { + // CONTROL: the element-kind rule decodes anyway. An `Ints` wire read by + // `Floats` is §3's element-kind mismatch: the field reads EMPTY, one + // kind_mismatch counts, and the parent reads on. And the reverse. + Wire w = build_ints( ints_spec() ); + const int64_t need = FloatsLoadMeasure( w.bytes, w.size ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Floats * floats = FloatsLoad( region, need, w.bytes, w.size, &r ); + CHECK( floats != NULL ); + if ( floats != NULL ) + { + CHECK_EQ( floats->values.size(), 0 ); + CHECK_EQ( floats->after, 777 ); + } + CHECK_EQ( r.kind_mismatch, 1 ); + CHECK_EQ( r.unknown + r.clamped + r.duplicate + (int) r.malformed, 0 ); + free( region ); + + IntsSpec spec = ints_spec(); + spec.element_kind = 10; // float32's kind, under Ints' declaration + const Verdict v = read_ints( build_ints( spec ) ); + CHECK_EQ( v.count, 0 ); + CHECK_EQ( v.report.kind_mismatch, 1 ); + CHECK( !v.report.malformed ); + CHECK_EQ( v.after, 777 ); + } + { + // A COUNT THE BODY CANNOT COVER (§2.9): into a REGION, LoadMeasure + // answers -1 with the reason count_over_length and no Load runs. Into a + // BUILDER, the prefix the body covers lands, malformed counts, and the + // parent reads on past the field's L + IntsSpec spec = ints_spec(); + spec.declared_n = 1000; + Wire w = build_ints( spec ); + TableRefuseReason reason = count_over_extent_cap; + CHECK_EQ( IntsLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_length ); + IntsBuilder into; + TableReport t; + CHECK( IntsLoadBuilder( into, w.bytes, w.size, &t ) ); + CHECK( t.malformed ); + CHECK_EQ( t.kind_mismatch + t.clamped + t.unknown + t.duplicate, 0 ); + CHECK_EQ( into.GetRoot()->values.count, 3 ); // the prefix the body covers + CHECK_EQ( into.GetRoot()->after, 777 ); + } + { + // A BODY TOO SHORT TO CARRY ITS OWN HEADER IS INERT (§4): no element, + // no counter, the field keeps the value it has + IntsSpec spec = ints_spec(); + spec.body_len = 1; + const Verdict v = read_ints( build_ints( spec ) ); + CHECK_EQ( v.count, 0 ); + report_silent( v.report, "an inert list body" ); + CHECK_EQ( v.after, 777 ); + } + { + // an EMPTY list on the wire: a header and a zero count + IntsSpec spec = ints_spec(); + spec.n = 0; + const Verdict v = read_ints( build_ints( spec ) ); + CHECK_EQ( v.count, 0 ); + report_silent( v.report, "an empty list body" ); + CHECK_EQ( v.after, 777 ); + } + { + // A DAMAGED ELEMENT inside a good count: a list of TABLES whose third + // element's L runs past the body keeps the two it decoded, counts + // malformed, and the parent reads on past the field's L + WireBuilder b; + b.field( "items", 14 ); + const int64_t body = b.open_len(); + b.u8( 13 ); + b.leb( 3 ); + for ( int i = 0; i < 2; i++ ) + { + const int64_t elem = b.open_len(); + b.field( "v", 4 ); + b.u32( (uint32_t) ( 5 + i ) ); + b.end(); + b.close_len( elem ); + } + b.leb( 100 ); // the third element's L, past the body + b.close_len( body ); + b.field( "label", 4 ); + b.u32( 9 ); + b.end(); + Wire w; + w.size = b.finish( w.bytes ); + const int64_t need = RowLoadMeasure( w.bytes, w.size ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Row * row = RowLoad( region, need, w.bytes, w.size, &r ); + CHECK( row != NULL ); + CHECK( r.malformed ); + if ( row != NULL ) + { + CHECK_EQ( row->items.size(), 2 ); // the element that never landed is not counted + if ( row->items.size() == 2 ) { CHECK_EQ( row->items[1].v, 6 ); } + CHECK_EQ( row->label, 9 ); + } + RowBuilder into; + TableReport t; + RowLoadBuilder( into, w.bytes, w.size, &t ); + reports_agree( t, r ); + CHECK_EQ( into.GetRoot()->items.count, 2 ); + free( region ); + } +} + +// ---- THE CLAMP CONTROL AT 100,000 (docs/SPEC-TABLES.md §2.9) ---- +// +// CONTROL: the reader CLAMPS the count against something. A []uint8 carrying +// 100,000 elements is past 2^16 and so past any bound a schema on the page +// declares, so a clamp any control author happened to pick would show, and the +// decoded count goes red, and `clamped` stays at zero. +static void test_clamp_control() +{ + static const int32_t kElements = 100000; + BytesBuilder b; + Bytes * bytes = b.GetRoot(); + for ( int32_t i = 0; i < kElements; i++ ) + { + uint8_t * e = BytesDataAdd( b.main, bytes->data ); + if ( e == NULL ) { printf( "FAIL: Add answered NULL at %d\n", i ); failures++; return; } + *e = (uint8_t) ( i & 0xff ); + } + bytes->after = 4242; + CHECK_EQ( bytes->data.count, kElements ); + const int64_t measured = BytesMeasure( b ); + uint8_t * wire = (uint8_t *) MEASURED_CALLOC( measured, 0 ); + if ( wire == NULL ) { return; } + const int64_t n = BytesSave( b, wire, measured ); + CHECK_EQ( n, measured ); + + const int64_t need = BytesLoadMeasure( wire, n ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { free( wire ); return; } + TableReport r; + const Bytes * loaded = BytesLoad( region, need, wire, n, &r ); + CHECK( loaded != NULL ); + report_silent( r, "the clamp control" ); + if ( loaded != NULL ) + { + CHECK_EQ( loaded->data.size(), kElements ); + CHECK_EQ( r.clamped, 0 ); + bool intact = loaded->data.size() == kElements; + for ( int32_t i = 0; intact && i < kElements; i++ ) { if ( loaded->data[i] != (uint8_t) ( i & 0xff ) ) { intact = false; } } + CHECK( intact ); + CHECK_EQ( loaded->after, 4242 ); + } + BytesBuilder into; + TableReport t; + CHECK( BytesLoadBuilder( into, wire, n, &t ) ); + reports_agree( t, r ); + CHECK_EQ( into.GetRoot()->data.count, kElements ); + free( region ); + free( wire ); +} + +// ---- THE SIX LoadMeasure REFUSALS (docs/SPEC-TABLES.md §2.8, §2.9, §6.5) ---- +// +// A unit test and not a `report` row, because a refusal produces no counters. +// Each wire is built in memory with a SYNTHETIC count rather than a golden: +// a count above the int32 cap, which no golden could carry because the file +// would be two gigabytes, a count whose elements cannot fit the field's L, the +// same two at DEPTH, inside an element's own list, the same two inside an +// element's MAP, which answers by the one rule a list does, and a clean wire beside +// them, which must measure. Red if any of the six answers something other +// than -1 with its own reason, if the clean one refuses, or if any of them +// moves one of the report's counters. + +static Wire build_sheet( uint64_t rows, uint64_t items, int32_t real_items ) +{ + WireBuilder b; + b.field( "rows", 14 ); + const int64_t body = b.open_len(); + b.u8( 13 ); + b.leb( rows ); + { + const int64_t row = b.open_len(); + b.field( "items", 14 ); + const int64_t inner = b.open_len(); + b.u8( 13 ); + b.leb( items ); + for ( int32_t i = 0; i < real_items; i++ ) + { + const int64_t sample = b.open_len(); + b.field( "v", 4 ); + b.u32( (uint32_t) i ); + b.end(); + b.close_len( sample ); + } + b.close_len( inner ); + b.field( "label", 4 ); + b.u32( 1 ); + b.end(); + b.close_len( row ); + } + b.close_len( body ); + b.end(); + Wire w; + w.size = b.finish( w.bytes ); + return w; +} + +// an `Army` body written FROM THE GRAMMAR: a kind 14 array of kind 13 `Squad` +// elements, each holding its `roster` MAP as a kind 14 array of kind 13 +// entries (§2.8), with the map's declared count a knob of its own +static Wire build_army( uint64_t squads, uint64_t entries, int32_t real_entries ) +{ + WireBuilder b; + b.field( "squads", 14 ); + const int64_t body = b.open_len(); + b.u8( 13 ); + b.leb( squads ); + { + const int64_t squad = b.open_len(); + b.field( "roster", 14 ); + const int64_t inner = b.open_len(); + b.u8( 13 ); + b.leb( entries ); + for ( int32_t i = 0; i < real_entries; i++ ) + { + const int64_t entry = b.open_len(); + b.field( "key", 6 ); + b.u8( (uint8_t) ( 2 + i ) ); + b.field( "value", 13 ); + const int64_t item = b.open_len(); + b.field( "count", 4 ); + b.u32( (uint32_t) ( 20 + i ) ); + b.end(); + b.close_len( item ); + b.end(); + b.close_len( entry ); + } + b.close_len( inner ); + b.field( "name", 4 ); + b.u32( 50 ); + b.end(); + b.close_len( squad ); + } + b.close_len( body ); + b.field( "after", 4 ); + b.u32( 8 ); + b.end(); + Wire w; + w.size = b.finish( w.bytes ); + return w; +} + +static void test_measure_refusals() +{ + // a count above the int32 cap, at the ROOT + { + IntsSpec spec = ints_spec(); + spec.declared_n = 0x80000000ll; + Wire w = build_ints( spec ); + TableRefuseReason reason = count_over_length; + CHECK_EQ( IntsLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_extent_cap ); + // and into a BUILDER it is the refusal LoadBuilder answers NULL for, + // the report holding what it held when the count was met: nothing + IntsBuilder into; + TableReport t; + CHECK( !IntsLoadBuilder( into, w.bytes, w.size, &t ) ); + report_silent( t, "the over-cap refusal into a builder" ); + } + // a count the field's L cannot carry, at the ROOT + { + IntsSpec spec = ints_spec(); + spec.declared_n = 100000; + Wire w = build_ints( spec ); + TableRefuseReason reason = count_over_extent_cap; + CHECK_EQ( IntsLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_length ); + } + // the same two at DEPTH, inside an element's own list + { + Wire w = build_sheet( 1, 0x80000000ull, 1 ); + TableRefuseReason reason = count_over_length; + CHECK_EQ( SheetLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_extent_cap ); + SheetBuilder into; + TableReport t; + CHECK( !SheetLoadBuilder( into, w.bytes, w.size, &t ) ); + report_silent( t, "the over-cap refusal at depth into a builder" ); + } + { + Wire w = build_sheet( 1, 100000, 1 ); + TableRefuseReason reason = count_over_extent_cap; + CHECK_EQ( SheetLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_length ); + } + // the same two at DEPTH, inside an element's MAP (§2.8): a map's term + // answers the reasons a list's does, the int32 cap first, one rule for + // both constructs (§6.5) + { + Wire w = build_army( 1, 0x80000000ull, 1 ); + TableRefuseReason reason = count_over_length; + CHECK_EQ( ArmyLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_extent_cap ); + } + { + Wire w = build_army( 1, 100000, 1 ); + TableRefuseReason reason = count_over_extent_cap; + CHECK_EQ( ArmyLoadMeasure( w.bytes, w.size, NULL, &reason ), -1 ); + CHECK( reason == count_over_length ); + } + // and a clean map-holding wire beside them, which must measure and load silently + { + Wire w = build_army( 1, 2, 2 ); + TableRefuseReason reason = count_over_length; + const int64_t need = ArmyLoadMeasure( w.bytes, w.size, NULL, &reason ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Army * army = ArmyLoad( region, need, w.bytes, w.size, &r ); + CHECK( army != NULL ); + report_silent( r, "the clean map-holding wire beside the refusals" ); + if ( army != NULL && army->squads.size() == 1 ) + { + CHECK_EQ( army->squads[0].roster.size(), 2 ); + const Item * item = army->squads[0].roster.Find( (uint8_t) 3 ); + CHECK( item != NULL && item->count == 21 ); + } + free( region ); + } + // and a clean wire beside them, which must measure and load silently + { + Wire w = build_sheet( 1, 2, 2 ); + TableRefuseReason reason = count_over_length; + const int64_t need = SheetLoadMeasure( w.bytes, w.size, NULL, &reason ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Sheet * sheet = SheetLoad( region, need, w.bytes, w.size, &r ); + CHECK( sheet != NULL ); + report_silent( r, "the clean wire beside the refusals" ); + if ( sheet != NULL ) + { + CHECK_EQ( sheet->rows.size(), 1 ); + if ( sheet->rows.size() == 1 ) { CHECK_EQ( sheet->rows[0].items.size(), 2 ); } + } + free( region ); + } +} + +// ---- sharing and the walk order (docs/SPEC-TABLES.md §2.9, §3.1) ---- + +static void test_shared() +{ + { + // `list_shared`: two slots naming one node beside a null slot. CONTROL: + // a shared node is written TWICE: the region's byte count and the + // text round trip's &node resolution go red. + AlbumBuilder b; + Album * album = b.GetRoot(); + TableRef * slot = AlbumPhotosAdd( b.main, album->photos ); + Photo * photo = PhotoEmplace( b.main, *slot ); + photo->width = 640; + photo->height = 480; + *AlbumPhotosAdd( b.main, album->photos ) = *slot; + AlbumPhotosAdd( b.main, album->photos ); // null + static uint8_t wire[1u << 16]; + const int64_t n = AlbumSave( b, wire, sizeof( wire ) ); + CHECK_EQ( AlbumMeasure( b ), n ); + pin_golden( "list_shared", wire, n ); + + const int64_t need = AlbumLoadMeasure( wire, n ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Album * loaded = AlbumLoad( region, need, wire, n, &r ); + CHECK( loaded != NULL ); + report_silent( r, "list_shared" ); + if ( loaded != NULL ) + { + CHECK_EQ( loaded->photos.size(), 3 ); + if ( loaded->photos.size() == 3 ) + { + CHECK( loaded->photos[0] != NULL && loaded->photos[0] == loaded->photos[1] ); + CHECK( loaded->photos[2] == NULL ); + if ( loaded->photos[0] != NULL ) { CHECK_EQ( loaded->photos[0]->width, 640 ); } + } + CHECK( PhotoAt( loaded->cover ) == NULL ); + // ONE node in the region: the attribution names the root and one photo + int64_t attribution = 0; + CHECK( AlbumLoadMeasure( wire, n, &attribution ) == need ); + CHECK_EQ( attribution, 2 * (int64_t) sizeof( TableNodeDirEntry ) ); + } + // the text: one definition and one &node reference, a null in its place + CHECK( b.Lock() ); + const int64_t text_bytes = AlbumToJsonMeasure( b.AsConst() ); + char * text = (char *) MEASURED_CALLOC( text_bytes, 1 ); + if ( text == NULL ) { free( region ); return; } + CHECK_EQ( AlbumToJson( b.AsConst(), text, text_bytes ), text_bytes ); + CHECK( strstr( text, "&node" ) != NULL ); + CHECK( strstr( text, "null" ) != NULL ); + AlbumBuilder into; + TableReport t; + CHECK( AlbumFromJson( into, text, text_bytes, &t ) ); + report_silent( t, "list_shared from text" ); + static uint8_t from_text[1u << 16]; + CHECK_EQ( AlbumSave( into, from_text, sizeof( from_text ) ), n ); + CHECK( memcmp( from_text, wire, (size_t) n ) == 0 ); + free( text ); + free( region ); + } + { + // `list_before_pointer`: the []*T is DECLARED BEFORE `cover` and reaches + // the shared node first, so it numbers it first. CONTROL: the walk + // visits lists out of declaration order, grouped after the pointer + // fields, and the pinned wire goes red on the node numbering. + AlbumBuilder b; + Album * album = b.GetRoot(); + TableRef * first = AlbumPhotosAdd( b.main, album->photos ); + Photo * b_node = PhotoEmplace( b.main, *first ); + b_node->width = 2; + TableRef * second = AlbumPhotosAdd( b.main, album->photos ); + Photo * a_node = PhotoEmplace( b.main, *second ); + a_node->width = 1; + album->cover = *second; // the SAME node, through the pointer field declared after + static uint8_t wire[1u << 16]; + const int64_t n = AlbumSave( b, wire, sizeof( wire ) ); + CHECK_EQ( AlbumMeasure( b ), n ); + pin_golden( "list_before_pointer", wire, n ); + const int64_t need = AlbumLoadMeasure( wire, n ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Album * loaded = AlbumLoad( region, need, wire, n, &r ); + CHECK( loaded != NULL ); + report_silent( r, "list_before_pointer" ); + if ( loaded != NULL ) + { + CHECK_EQ( loaded->photos.size(), 2 ); + CHECK( PhotoAt( loaded->cover ) != NULL && PhotoAt( loaded->cover ) == loaded->photos[1] ); + CHECK( loaded->photos[0] != loaded->photos[1] ); + } + free( region ); + } +} + +// ---- where else a list rides in a holder's extent (§2.9) ---- + +static void test_nested() +{ + { + // `list_nested`: a list of tables that hold lists, and a pointed-at + // holder with a list of its own. CONTROL: LoadMeasure's term summed at + // ONE DEPTH only, and the measure goes red against the region Load fills. + SheetBuilder b; + Sheet * sheet = b.GetRoot(); + for ( int i = 0; i < 3; i++ ) + { + Row * row = SheetRowsAdd( b.main, sheet->rows ); + row->label = 10 + i; + for ( int k = 0; k <= i; k++ ) { RowItemsAdd( b.main, row->items )->v = 100 * i + k; } + } + Row * pinned = RowEmplace( b.main, sheet->pinned ); + pinned->label = 99; + RowItemsAdd( b.main, pinned->items )->v = 1; + RowItemsAdd( b.main, pinned->items )->v = 2; + static uint8_t wire[1u << 16]; + const int64_t n = SheetSave( b, wire, sizeof( wire ) ); + CHECK_EQ( SheetMeasure( b ), n ); + pin_golden( "list_nested", wire, n ); + + const int64_t need = SheetLoadMeasure( wire, n ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Sheet * loaded = SheetLoad( region, need, wire, n, &r ); + CHECK( loaded != NULL ); + report_silent( r, "list_nested" ); + if ( loaded != NULL ) + { + CHECK_EQ( loaded->rows.size(), 3 ); + for ( int32_t i = 0; i < loaded->rows.size(); i++ ) + { + const Row & row = loaded->rows[i]; + CHECK_EQ( row.label, 10 + i ); + CHECK_EQ( row.items.size(), i + 1 ); + for ( int32_t k = 0; k < row.items.size(); k++ ) { CHECK_EQ( row.items[k].v, 100 * i + k ); } + } + const Row * p = RowAt( loaded->pinned ); + CHECK( p != NULL ); + if ( p != NULL ) + { + CHECK_EQ( p->label, 99 ); + CHECK_EQ( p->items.size(), 2 ); + if ( p->items.size() == 2 ) { CHECK_EQ( p->items[1].v, 2 ); } + } + static uint8_t again[1u << 16]; + CHECK_EQ( SheetSave( loaded, again, sizeof( again ) ), n ); + CHECK( memcmp( again, wire, (size_t) n ) == 0 ); + + // the COOK, from the region and from the builder, one artifact + const int64_t cook_bytes = SheetCookMeasure( loaded ); + CHECK( cook_bytes > 0 ); + void * cooked = MEASURED_CALLOC( cook_bytes, 0 ); + if ( cooked != NULL ) + { + CHECK( SheetCook( loaded, cooked, (uint64_t) cook_bytes, host_byte_order() ) ); + // THE COOK IS PINNED on a little-endian host, so a LAYOUT + // sabotage (the element array laid after a nested container's, + // a wrong alignment, a dropped pre-order) goes red on a CHECK + // before any Open trusts the bytes. The big-endian leg writes + // the other order and skips the pin. + const bool trusted = host_byte_order() != TableByteOrder::Little || pin_golden( "list_nested_cook", (const uint8_t *) cooked, cook_bytes ); + const Sheet * opened = trusted ? SheetOpen( cooked, (uint64_t) cook_bytes ) : NULL; + CHECK( opened != NULL ); + if ( opened != NULL ) + { + CHECK_EQ( opened->rows.size(), 3 ); + if ( opened->rows.size() == 3 ) { CHECK_EQ( opened->rows[2].items[2].v, 202 ); } + const Row * op = RowAt( opened->pinned ); + CHECK( op != NULL && op->items.size() == 2 && op->items[0].v == 1 ); + } + CHECK_EQ( SheetCookMeasure( b ), cook_bytes ); + void * twice = MEASURED_CALLOC( cook_bytes, 0 ); + if ( twice != NULL ) + { + CHECK( SheetCook( b, twice, (uint64_t) cook_bytes, host_byte_order() ) ); + CHECK( memcmp( cooked, twice, (size_t) cook_bytes ) == 0 ); + free( twice ); + } + save_cook( "sheet", cooked, cook_bytes ); + // THE FORGERY for `schema cook-check`: the root's `rows` slot is the + // first sixteen bytes of the data part, and its delta is pointed + // past the region, which §7.4's containment clause must refuse + if ( getenv( "SCHEMA_LIST_COOK_DIR" ) != NULL ) + { + uint8_t * forged = (uint8_t *) cooked; + const int64_t delta = cook_bytes; // past every byte the file has + memcpy( forged + 64, &delta, sizeof( delta ) ); + save_cook( "sheet-forged", forged, cook_bytes ); + } + free( cooked ); + } + } + // and the TOOL's path reads every depth into a builder + SheetBuilder into; + TableReport t; + CHECK( SheetLoadBuilder( into, wire, n, &t ) ); + reports_agree( t, r ); + CHECK_EQ( into.GetRoot()->rows.count, 3 ); + static uint8_t relocked[1u << 16]; + CHECK_EQ( SheetSave( into, relocked, sizeof( relocked ) ), n ); + CHECK( memcmp( relocked, wire, (size_t) n ) == 0 ); + // the region is EXACT: one byte short is refused, so the extent scan + // counted every depth and every node and none twice. Load zeroes the + // region before it refuses, so this is the block's last act. + TableReport short_report; + CHECK( SheetLoad( region, need - 1, wire, n, &short_report ) == NULL ); + free( region ); + } + { + // `list_of_maps`: an element that holds a MAP. CONTROL: the element + // array is laid out AFTER a nested container's, breaking the pre-order + // rule, and the region's byte compare goes red. + ArmyBuilder b; + Army * army = b.GetRoot(); + for ( int i = 0; i < 2; i++ ) + { + Squad * squad = ArmySquadsAdd( b.main, army->squads ); + squad->name = 50 + i; + SquadRosterInsert( b.main, squad->roster, (uint8_t) ( 9 - i ) )->count = 90 + i; + SquadRosterInsert( b.main, squad->roster, (uint8_t) ( 2 + i ) )->count = 20 + i; + } + army->after = 8; + static uint8_t wire[1u << 16]; + const int64_t n = ArmySave( b, wire, sizeof( wire ) ); + CHECK_EQ( ArmyMeasure( b ), n ); + pin_golden( "list_of_maps", wire, n ); + const int64_t need = ArmyLoadMeasure( wire, n ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Army * loaded = ArmyLoad( region, need, wire, n, &r ); + CHECK( loaded != NULL ); + report_silent( r, "list_of_maps" ); + if ( loaded != NULL ) + { + CHECK_EQ( loaded->squads.size(), 2 ); + for ( int32_t i = 0; i < loaded->squads.size(); i++ ) + { + const Squad & squad = loaded->squads[i]; + CHECK_EQ( squad.name, 50 + i ); + CHECK_EQ( squad.roster.size(), 2 ); + const Item * low = squad.roster.Find( (uint8_t) ( 2 + i ) ); + CHECK( low != NULL && low->count == 20 + i ); + if ( squad.roster.size() == 2 ) { CHECK_EQ( squad.roster.begin().at->key, 2 + i ); } // sorted + } + CHECK_EQ( loaded->after, 8 ); + static uint8_t again[1u << 16]; + CHECK_EQ( ArmySave( loaded, again, sizeof( again ) ), n ); + CHECK( memcmp( again, wire, (size_t) n ) == 0 ); + // the cook lays the element array FIRST, then each element's map + const int64_t cook_bytes = ArmyCookMeasure( loaded ); + void * cooked = MEASURED_CALLOC( cook_bytes, 0 ); + if ( cooked != NULL ) + { + CHECK( ArmyCook( loaded, cooked, (uint64_t) cook_bytes, host_byte_order() ) ); + const bool trusted = host_byte_order() != TableByteOrder::Little || pin_golden( "list_of_maps_cook", (const uint8_t *) cooked, cook_bytes ); + const Army * opened = trusted ? ArmyOpen( cooked, (uint64_t) cook_bytes ) : NULL; + + CHECK( opened != NULL ); + + if ( opened != NULL && opened->squads.size() == 2 ) + { + const Item * item = opened->squads[1].roster.Find( (uint8_t) 8 ); + CHECK( item != NULL && item->count == 91 ); + } + save_cook( "army", cooked, cook_bytes ); + free( cooked ); + } + } + ArmyBuilder into; + TableReport t; + CHECK( ArmyLoadBuilder( into, wire, n, &t ) ); + reports_agree( t, r ); + static uint8_t relocked[1u << 16]; + CHECK_EQ( ArmySave( into, relocked, sizeof( relocked ) ), n ); + CHECK( memcmp( relocked, wire, (size_t) n ) == 0 ); + TableReport short_report; // one byte short is refused; last, because Load zeroes the region first + CHECK( ArmyLoad( region, need - 1, wire, n, &short_report ) == NULL ); + free( region ); + } + + { + // AN UNREACHED NON-EMPTY LIST SLOT IS REFUSED by Cook and by Lock, the + // same refusal §7.6 gives a pointer in that position. The WIRE is not + // refused, a counted array rides its live slots. + DeckBuilder past; + Deck * d = past.GetRoot(); + d->hands_count = 1; + RowItemsAdd( past.main, d->hands[0].items )->v = 1; + RowItemsAdd( past.main, d->hands[2].items )->v = 9; // past the count + static uint8_t rides[1u << 16]; + const int64_t measured = DeckMeasure( past ); + CHECK_EQ( DeckSave( past, rides, sizeof( rides ) ), measured ); + CHECK( !past.Lock() ); + CHECK( past.AsConst() == NULL ); // nothing partial + CHECK_EQ( DeckCookMeasure( past ), -1 ); + } +} + +// ---- the migration itself (docs/SPEC-TABLES.md §2.9) ---- + +static void test_migrates() +{ + // ONE content, TWO declarations of the holder, [..8]Unit and []Unit, ONE + // pinned wire both write byte for byte and both read into equal values, + // the report silent in both directions. The bound is above the count, so + // the row proves the framing and not the clamp. + static uint8_t bounded_wire[1u << 16]; + static uint8_t unbounded_wire[1u << 16]; + int64_t bounded_bytes = 0, unbounded_bytes = 0; + { + Bounded value; + BoundedReset( value ); + value.items_count = 3; + for ( int i = 0; i < 3; i++ ) { value.items[i].v = 7 * ( i + 1 ); } + value.tag = 42; + bounded_bytes = BoundedSave( value, bounded_wire, sizeof( bounded_wire ) ); + CHECK_EQ( BoundedMeasure( value ), bounded_bytes ); + } + { + UnboundedBuilder b; + Unbounded * value = b.GetRoot(); + for ( int i = 0; i < 3; i++ ) { UnboundedItemsAdd( b.main, value->items )->v = 7 * ( i + 1 ); } + value->tag = 42; + unbounded_bytes = UnboundedSave( b, unbounded_wire, sizeof( unbounded_wire ) ); + CHECK_EQ( UnboundedMeasure( b ), unbounded_bytes ); + } + CHECK_EQ( unbounded_bytes, bounded_bytes ); + CHECK( memcmp( bounded_wire, unbounded_wire, (size_t) bounded_bytes ) == 0 ); + pin_golden( "list_migrates", unbounded_wire, unbounded_bytes ); + + // each reads the other's wire silently, into equal values + { + Bounded back; + TableReport r; + CHECK( BoundedLoad( back, unbounded_wire, unbounded_bytes, &r ) ); + report_silent( r, "the bounded reader over the unbounded wire" ); + CHECK_EQ( back.items_count, 3 ); + CHECK_EQ( back.items[2].v, 21 ); + CHECK_EQ( back.tag, 42 ); + } + { + const int64_t need = UnboundedLoadMeasure( bounded_wire, bounded_bytes ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Unbounded * back = UnboundedLoad( region, need, bounded_wire, bounded_bytes, &r ); + CHECK( back != NULL ); + report_silent( r, "the unbounded reader over the bounded wire" ); + if ( back != NULL ) + { + CHECK_EQ( back->items.size(), 3 ); + if ( back->items.size() == 3 ) { CHECK_EQ( back->items[2].v, 21 ); } + CHECK_EQ( back->tag, 42 ); + } + free( region ); + } +} + +// ---- the TEXT form (docs/SPEC-TABLES.md §2.9, §16) ---- + +static void test_text() +{ + SaveBuilder b; + build_save( b ); + CHECK( b.Lock() ); + const Save * locked = b.AsConst(); + + const int64_t need = SaveToJsonMeasure( locked ); + CHECK( need > 0 ); + char * text = (char *) MEASURED_CALLOC( need, 1 ); + if ( text == NULL ) { return; } + const int64_t written = SaveToJson( locked, text, need ); + CHECK_EQ( written, need ); + + // A JSON ARRAY, in INDEX order, the pointer row per element of a []*T + CHECK( strstr( text, "\"placements\": [" ) != NULL ); + CHECK( strstr( text, "\"scores\": [" ) != NULL ); + CHECK( strstr( text, "&node" ) != NULL ); + CHECK( strstr( text, "null" ) != NULL ); + const char * ten = strstr( text, "10" ); + const char * thirty = strstr( text, "30" ); + CHECK( ten != NULL && thirty != NULL && ten < thirty ); + + // and the text reads back: one instance, one text, both ways + SaveBuilder into; + TableReport report; + CHECK( SaveFromJson( into, text, written, &report ) ); + report_silent( report, "list_tables from text" ); + CHECK_EQ( into.GetRoot()->placements.count, 3 ); + CHECK_EQ( into.GetRoot()->log.count, 3 ); + CHECK_EQ( into.GetRoot()->scores.count, 3 ); + CHECK( into.Lock() ); + const int64_t again_bytes = SaveToJsonMeasure( into.AsConst() ); + char * again = (char *) MEASURED_CALLOC( again_bytes, 1 ); + if ( again == NULL ) { free( text ); return; } + CHECK_EQ( SaveToJson( into.AsConst(), again, again_bytes ), again_bytes ); + CHECK_EQ( again_bytes, written ); + CHECK( memcmp( again, text, (size_t) written ) == 0 ); // byte-stable + static uint8_t from_text[1u << 16]; + CHECK_EQ( SaveSave( into.AsConst(), from_text, sizeof( from_text ) ), bytes_tables ); + CHECK( memcmp( from_text, wire_tables, (size_t) bytes_tables ) == 0 ); + free( again ); + free( text ); + + // `[]` is an empty list, null is kind_mismatch, a wrong-shaped element + // counts and keeps its slot at defaults, and EVERY element the text + // carries is read, because there is no bound to drop a tail against + struct TextRow { const char * text; int32_t count; int32_t mismatch; int32_t after; }; + const TextRow rows[] = { + { "{\"values\":[],\"after\":5}", 0, 0, 5 }, + { "{\"values\":null,\"after\":5}", 0, 1, 5 }, + { "{\"values\":[1,\"x\",3],\"after\":5}", 3, 1, 5 }, + { "{\"values\":[1,2],\"values\":[9],\"after\":5}", 1, 0, 5 }, // LAST WINS, whole + }; + for ( int i = 0; i < 4; i++ ) + { + IntsBuilder rb; + TableReport r; + CHECK( IntsFromJson( rb, rows[i].text, (int64_t) strlen( rows[i].text ), &r ) ); + CHECK_EQ( rb.GetRoot()->values.count, rows[i].count ); + CHECK_EQ( r.kind_mismatch, rows[i].mismatch ); + CHECK_EQ( r.clamped, 0 ); + CHECK_EQ( rb.GetRoot()->after, rows[i].after ); + } + { + static char many[8192]; + int at = snprintf( many, sizeof( many ), "{\"values\":[" ); + for ( int i = 0; i < 300; i++ ) { at += snprintf( many + at, sizeof( many ) - (size_t) at, "%s%d", i > 0 ? "," : "", i ); } + snprintf( many + at, sizeof( many ) - (size_t) at, "]}" ); + IntsBuilder rb; + TableReport r; + CHECK( IntsFromJson( rb, many, (int64_t) strlen( many ), &r ) ); + report_silent( r, "300 elements from text" ); + CHECK_EQ( rb.GetRoot()->values.count, 300 ); + int32_t i = 0; + for ( const int32_t * v : IntsValuesEach( rb.arena, rb.GetRoot()->values ) ) { if ( *v != i ) { failures++; printf( "FAIL element %d read as %d\n", i, *v ); break; } i++; } + } +} + +// ---- the reading path allocates nothing (§2.9, §6.5) ---- + +static void test_allocation_audit() +{ + const long long before = allocations; + const int64_t need = SaveLoadMeasure( wire_tables, bytes_tables ); + CHECK( need > 0 ); + uint8_t * region = (uint8_t *) MEASURED_CALLOC( need, 0 ); + if ( region == NULL ) { return; } + TableReport r; + const Save * loaded = SaveLoad( region, need, wire_tables, bytes_tables, &r ); + CHECK( loaded != NULL ); + long long sum = 0; + if ( loaded != NULL ) + { + for ( int32_t i = 0; i < loaded->placements.size(); i++ ) { sum += loaded->placements[i].model; } + for ( const Placement & p : loaded->placements ) { sum += p.model; } + for ( const LogEntry * e : loaded->log ) { if ( e != NULL ) { sum += e->tick; } } + for ( int32_t i = 0; i < loaded->scores.size(); i++ ) { sum += loaded->scores[i]; } + } + CHECK( sum > 0 ); + const int64_t cook_bytes = SaveCookMeasure( loaded ); + void * cooked = MEASURED_CALLOC( cook_bytes, 0 ); + if ( cooked != NULL ) + { + CHECK( SaveCook( loaded, cooked, (uint64_t) cook_bytes, host_byte_order() ) ); + const Save * opened = SaveOpen( cooked, (uint64_t) cook_bytes ); + CHECK( opened != NULL ); + if ( opened != NULL ) { for ( const Placement & p : opened->placements ) { sum += p.model; } } + free( cooked ); + } + free( region ); + CHECK_EQ( allocations - before, 0 ); // not one operator new on the reading path +} + +int main( int argc, char ** argv ) +{ + + if ( argc > 1 && strcmp( argv[1], "measure-refusals" ) == 0 ) + { + test_measure_refusals(); + if ( failures != 0 ) + { + printf( "\n%d measure refusal check(s) failed\n", failures ); + return 1; + } + printf( "list measure refusals: six -1s with their reasons, two clean measures, no counter moved (docs/SPEC-TABLES.md §2.8, §2.9, §6.5)\n" ); + return 0; + } + test_writer(); + test_builder(); + test_const_forms(); + test_reader(); + test_clamp_control(); + test_measure_refusals(); + test_shared(); + test_nested(); + test_migrates(); + test_text(); + test_allocation_audit(); + + if ( failures != 0 ) + { + printf( "\n%d list check(s) failed\n", failures ); + return 1; + } + printf( "lists: all checks passed (docs/SPEC-TABLES.md §2.9)\n" ); + return 0; +} diff --git a/testdata/golden/tables/blobs/AssetsTable.cpp b/testdata/golden/tables/blobs/AssetsTable.cpp index e1620d36e..86b2bb020 100644 --- a/testdata/golden/tables/blobs/AssetsTable.cpp +++ b/testdata/golden/tables/blobs/AssetsTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2750,6 +2767,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace blobdemo #endif // BLOBDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/block/PaddedTable.cpp b/testdata/golden/tables/block/PaddedTable.cpp index 37fc3c912..de8b50825 100644 --- a/testdata/golden/tables/block/PaddedTable.cpp +++ b/testdata/golden/tables/block/PaddedTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace blockdemo #endif // BLOCKDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/block/RenderTable.cpp b/testdata/golden/tables/block/RenderTable.cpp index 72a6a2f46..63c5f85e2 100644 --- a/testdata/golden/tables/block/RenderTable.cpp +++ b/testdata/golden/tables/block/RenderTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace blockdemo #endif // BLOCKDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/blockhome/DataTable.cpp b/testdata/golden/tables/blockhome/DataTable.cpp index 889b5b8c8..fc9089196 100644 --- a/testdata/golden/tables/blockhome/DataTable.cpp +++ b/testdata/golden/tables/blockhome/DataTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace blockhome #endif // BLOCKHOME_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/blockhome/FrameTable.cpp b/testdata/golden/tables/blockhome/FrameTable.cpp index 42e9edab9..66de719df 100644 --- a/testdata/golden/tables/blockhome/FrameTable.cpp +++ b/testdata/golden/tables/blockhome/FrameTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace blockhome #endif // BLOCKHOME_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/GuardedTable.cpp b/testdata/golden/tables/examples/GuardedTable.cpp index 5e0e1dd5d..ad26d6936 100644 --- a/testdata/golden/tables/examples/GuardedTable.cpp +++ b/testdata/golden/tables/examples/GuardedTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/KeyedTable.cpp b/testdata/golden/tables/examples/KeyedTable.cpp index 119f639c0..767dababf 100644 --- a/testdata/golden/tables/examples/KeyedTable.cpp +++ b/testdata/golden/tables/examples/KeyedTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/NestedTable.cpp b/testdata/golden/tables/examples/NestedTable.cpp index 837d9ddf5..e5f0acf7d 100644 --- a/testdata/golden/tables/examples/NestedTable.cpp +++ b/testdata/golden/tables/examples/NestedTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/PackTable.cpp b/testdata/golden/tables/examples/PackTable.cpp index 51aefb058..d33ec128a 100644 --- a/testdata/golden/tables/examples/PackTable.cpp +++ b/testdata/golden/tables/examples/PackTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/RangesTable.cpp b/testdata/golden/tables/examples/RangesTable.cpp index ef14d69c6..1e4129090 100644 --- a/testdata/golden/tables/examples/RangesTable.cpp +++ b/testdata/golden/tables/examples/RangesTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/TablesTable.cpp b/testdata/golden/tables/examples/TablesTable.cpp index eee81f0e6..970077ae9 100644 --- a/testdata/golden/tables/examples/TablesTable.cpp +++ b/testdata/golden/tables/examples/TablesTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/examples/WideTable.cpp b/testdata/golden/tables/examples/WideTable.cpp index 6c69959c8..ca0e06912 100644 --- a/testdata/golden/tables/examples/WideTable.cpp +++ b/testdata/golden/tables/examples/WideTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace tabledemo #endif // TABLEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/lists/HoldersTable.cpp b/testdata/golden/tables/lists/HoldersTable.cpp new file mode 100644 index 000000000..127d117a7 --- /dev/null +++ b/testdata/golden/tables/lists/HoldersTable.cpp @@ -0,0 +1,3217 @@ +// Code generated by the schema compiler from Holders.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — the TABLE wire's text form (docs/SPEC-TABLES.md §16). +// Compile this file to use FromJson / ToJson; a project that +// never reads or writes a text does not compile it and pays nothing. + +#include "HoldersTable.h" + +#include // the text form: number formatting +#include // the text form: exact number conversion +#include // the text form: the runtime's decimal point + +// The guard is not vestigial. Several listdemo Table.cpp files may be +// concatenated into ONE translation unit — a unity build — and without it +// each would redefine the walk. It is also why the walk's functions may be +// weak (vague linkage) across separate objects: ODR requires their +// definitions to be token-identical, and the generic-walk gate is what +// proves that, byte for byte, across every generated .cpp. +#ifndef LISTDEMO_SCHEMA_TABLE_JSON +#define LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +// ---- the pointer adapters (docs/SPEC-TABLES.md §16.7) ---- +// +// The walk below is ONE walk, byte-identical in every generated .cpp, and a +// pointer is the one kind it cannot walk alone: reading one needs the +// builder's arena and writing one needs a region's deref, and neither exists +// in a unit that declares no pointer. So the walk calls these three and does +// not define them. A unit with no pointer defines them as stubs no field ever +// reaches; a pointered unit defines them in the graph half that follows the +// walk. + +struct TableJsonIn; +struct TableJsonOut; + +// a pointer field's object, or the `&node` reference standing in for it, into +// the slot; the cursor is on the opening brace +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// the node a pointer slot names, in place — or as `&node` when it is shared +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// the FIRST key of an object the walk is skipping begins with `&`: the cursor is +// on its value. A dropped definition still takes its label (§16.7); a fixed reader +// skips the value whole, as it skips everything else it does not place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); + +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- +// +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map +inline bool TableJsonIsMap( const TableFieldInfo * f ); +// the map as a plain JSON object keyed by the KEY, in ASCENDING key order +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that object back into the slot, in whatever order the text gives it +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); + +// ---- json walk: begin ---- +// +// The TEXT form (docs/SPEC-TABLES.md §16): one table, one text, one walk over the +// reflection descriptors (§8). Reading fills ONE caller-owned instance and +// allocates nothing beyond it; writing targets a caller buffer with the +// wire's measure/write symmetry. Everything AROUND this — which file goes +// with which instance, what key an instance is filed under, how instances +// link into a root table's collections — is a packer's opinion and stays +// with the tool that holds it. +// +// The dialect: trailing commas are accepted on read (the authoring files +// this exists for carry them) and never written; comments are not JSON and +// are refused; unknown keys are skipped and counted; a duplicate key is +// last-wins and counted; a key present with the wrong JSON type is skipped +// and counted, never coerced. + +static const int32_t kTableJsonMaxDepth = 128; + +// A key longer than this cannot name a field, so it is skipped as unknown. +static const int32_t kTableJsonMaxKey = 256; + +// The longest numeric token the walk will convert. Anything longer is a +// value no field can hold and counts as a kind mismatch. +static const int32_t kTableJsonMaxNumber = 512; + +// The decimal point the C runtime is CURRENTLY using. Number conversion is +// the one locale-sensitive corner of the grammar — JSON's point is always +// '.', the runtime's is whatever the program set — so every number crosses +// this one character on the way out and on the way back in. Nothing else in +// the walk consults the locale. +inline char TableJsonDecimalPoint() +{ + const struct lconv * conv = localeconv(); + if ( conv != NULL && conv->decimal_point != NULL && conv->decimal_point[0] != 0 ) + { + return conv->decimal_point[0]; + } + return '.'; +} + +// ---- storage access: the descriptors give an offset and a width, and the +// ---- storage is the HOST's, so every load and store goes through a width +// ---- switch rather than a memcpy into the low bytes of a wider word + +// finite: not a NaN, not an infinity. Written without — the walk's +// runtime surface stays the handful of functions it already names. +// A vocabulary entry the descriptor could not spell. The generated name +// functions answer "???" for a value outside the declared set, and that is +// not a name — writing it would put a spelling in the text that the reader +// then counts as unknown, turning a refusal into a silent loss. +inline bool TableJsonNamed( const char * name ) +{ + return name != NULL && strcmp( name, "???" ) != 0; +} + +inline bool TableJsonFinite( double v ) +{ + return v == v && v <= 1.7976931348623157e308 && v >= -1.7976931348623157e308; +} + +inline uint64_t TableJsonGetRaw( const void * storage, uint32_t width ) +{ + switch ( width ) + { + case 1: { uint8_t v = 0; memcpy( &v, storage, 1 ); return v; } + case 2: { uint16_t v = 0; memcpy( &v, storage, 2 ); return v; } + case 4: { uint32_t v = 0; memcpy( &v, storage, 4 ); return v; } + case 8: { uint64_t v = 0; memcpy( &v, storage, 8 ); return v; } + } + return 0; +} + +inline void TableJsonSetRaw( void * storage, uint32_t width, uint64_t value ) +{ + switch ( width ) + { + case 1: { uint8_t v = (uint8_t) value; memcpy( storage, &v, 1 ); break; } + case 2: { uint16_t v = (uint16_t) value; memcpy( storage, &v, 2 ); break; } + case 4: { uint32_t v = (uint32_t) value; memcpy( storage, &v, 4 ); break; } + case 8: { uint64_t v = value; memcpy( storage, &v, 8 ); break; } + } +} + +inline int64_t TableJsonGetSigned( const void * storage, uint32_t width ) +{ + uint64_t raw = TableJsonGetRaw( storage, width ); + if ( width < 8 ) + { + uint64_t sign = uint64_t( 1 ) << ( width * 8 - 1 ); + if ( ( raw & sign ) != 0 ) + { + raw |= ~( ( sign << 1 ) - 1 ); + } + } + return (int64_t) raw; +} + +// ---- the WIDE kinds (docs/SPEC-TABLES.md §3, §16.2) ---- +// +// The 128-bit integers and the fixed-point family convert EXACTLY, over two +// 64-bit lanes: a 128-bit integer is a decimal integer, a fixed value a +// decimal in WHOLE UNITS (1.0, -0.25, 3.0000152587890625) and nothing +// on either path passes through a double. Nothing here needs a 128-bit type +// either, which is what keeps this walk one text for every unit. +struct TableJsonWide +{ + uint64_t lo; + uint64_t hi; +}; + +inline bool TableJsonKindWide( uint8_t kind ) { return kind >= 18 && kind <= 29; } +inline bool TableJsonKindWideSigned( uint8_t kind ) { return kind == 18 || ( kind >= 20 && kind <= 24 ); } +inline bool TableJsonKindFixed( uint8_t kind ) { return kind >= 20 && kind <= 29; } + +inline bool TableJsonWideZero( TableJsonWide v ) { return v.lo == 0 && v.hi == 0; } +inline bool TableJsonWideNegative( TableJsonWide v ) { return ( v.hi >> 63 ) != 0; } + +inline int TableJsonWideCompare( TableJsonWide a, TableJsonWide b, bool is_signed ) +{ + if ( is_signed && TableJsonWideNegative( a ) != TableJsonWideNegative( b ) ) { return TableJsonWideNegative( a ) ? -1 : 1; } + if ( a.hi != b.hi ) { return a.hi < b.hi ? -1 : 1; } + if ( a.lo != b.lo ) { return a.lo < b.lo ? -1 : 1; } + return 0; +} + +inline TableJsonWide TableJsonWideShl( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.hi = v.lo << ( n - 64 ); return r; } + r.hi = ( v.hi << n ) | ( v.lo >> ( 64 - n ) ); + r.lo = v.lo << n; + return r; +} + +inline TableJsonWide TableJsonWideShr( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.lo = v.hi >> ( n - 64 ); return r; } + r.lo = ( v.lo >> n ) | ( v.hi << ( 64 - n ) ); + r.hi = v.hi >> n; + return r; +} + +inline TableJsonWide TableJsonWideNeg( TableJsonWide v ) +{ + TableJsonWide r; + r.lo = ~v.lo + 1; + r.hi = ~v.hi + ( r.lo == 0 ? 1 : 0 ); + return r; +} + +// v = v * m + a; the return is the carry out of 128 bits +inline uint32_t TableJsonWideMulAdd( TableJsonWide * v, uint32_t m, uint32_t a ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t carry = a; + for ( int i = 0; i < 4; i++ ) + { + uint64_t p = limb[i] * m + carry; + limb[i] = p & 0xffffffffull; + carry = p >> 32; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) carry; +} + +// v = v / d; the return is the remainder +inline uint32_t TableJsonWideDiv( TableJsonWide * v, uint32_t d ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t rem = 0; + for ( int i = 3; i >= 0; i-- ) + { + uint64_t cur = ( rem << 32 ) | limb[i]; + limb[i] = cur / d; + rem = cur % d; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) rem; +} + +// The storage of a wide kind, as lanes. A sixteen-byte storage is serialize's +// pair — native __int128 in the host's byte order, or the emulated struct with +// its low lane first — so the lanes are read in the host's order; a narrower +// storage is one lane, sign-extended for a signed kind. +inline TableJsonWide TableJsonWideLoad( const void * storage, uint32_t width, bool is_signed ) +{ + TableJsonWide v = { 0, 0 }; + if ( width == 16 ) + { + uint64_t half[2]; + memcpy( half, storage, 16 ); + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + v.lo = little ? half[0] : half[1]; + v.hi = little ? half[1] : half[0]; + return v; + } + v.lo = is_signed ? (uint64_t) TableJsonGetSigned( storage, width ) : TableJsonGetRaw( storage, width ); + v.hi = ( is_signed && ( v.lo >> 63 ) != 0 ) ? ~uint64_t( 0 ) : 0; + return v; +} + +inline void TableJsonWideStore( void * storage, uint32_t width, TableJsonWide v ) +{ + if ( width == 16 ) + { + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + uint64_t half[2]; + half[0] = little ? v.lo : v.hi; + half[1] = little ? v.hi : v.lo; + memcpy( storage, half, 16 ); + return; + } + TableJsonSetRaw( storage, width, v.lo ); +} + +// a counted field's companion: a string's length, a bytes' length, a counted +// array's count. Bounded by the declared extent on the way out, so a storage +// invariant a caller broke cannot walk off the end of the array. +inline int32_t TableJsonCount( const void * base, const TableFieldInfo * f ) +{ + if ( !f->counted ) + { + return f->array_bound; + } + int32_t count = 0; + memcpy( &count, (const uint8_t *) base + f->count_offset, sizeof( count ) ); + if ( count < 0 ) { count = 0; } + if ( count > f->array_bound ) { count = f->array_bound; } + return count; +} + +inline void TableJsonSetCount( void * base, const TableFieldInfo * f, int32_t count ) +{ + if ( f->counted ) + { + memcpy( (uint8_t *) base + f->count_offset, &count, sizeof( count ) ); + } +} + +// ---- what a field's kind expects to see in the text ---- +// +// One classifier, consulted by both directions, so a reader and a writer can +// never disagree about a kind's JSON form. 'o' object, 'a' array, 's' +// string, 'n' number, 'b' boolean. +// +// A vocabulary field is spelled by NAME: an enum is one name, a flags mask +// is the array of the names of its set bits. The two are told apart by the +// id column — an enum variant rides under a wire id, a flags BIT never does +// (docs/SPEC-TABLES.md §4), so a name function with no id function is flags. +// +// bytes(N) is the one kind whose element kind does not decide its form: it +// shares u8 with a plain array of u8, and rides as base64. The schema type +// name settles it, and "bytes" is a keyword no declaration can claim. +inline bool TableJsonIsBytes( const TableFieldInfo * f ) +{ + return f->is_array && f->kind == 6 && strcmp( f->type_name, "bytes" ) == 0; +} + +// An ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): its JSON form is an OBJECT +// keyed by variant name, not a positional array, because that is what the +// storage is — one slot per variant, addressed by the variant. +inline bool TableJsonIsKeyed( const TableFieldInfo * f ) +{ + return f->key_name != NULL; +} + +// THE KEY A STORAGE SLOT HOLDS (§2.4, §8): the storage shifts left, so slot i +// holds the key i + 1 and nothing is stored for None. This is the ONE place +// the walker spells the shift. +inline uint64_t TableJsonKeyedSlotKey( int64_t slot ) +{ + return (uint64_t) ( slot + 1 ); +} + +// A slot whose key names a variant of the keying enum. Every slot in +// [0, array_bound) does, unless the enum carries max-headroom variants outside +// a table closure, where a reserved value names nothing and its key id is 0 — +// the reserved id no declared name can fold to (§5). +inline bool TableJsonKeyedSlotValid( const TableFieldInfo * f, int64_t slot ) +{ + return f->key_id( TableJsonKeyedSlotKey( slot ) ) != 0; +} + +inline bool TableJsonIsFlags( const TableFieldInfo * f ) +{ + return f->enum_name != NULL && f->variant_id == NULL; +} + +inline bool TableJsonIsEnum( const TableFieldInfo * f ) +{ + return f->variant_id != NULL && f->arms == NULL; +} + +inline char TableJsonShape( const TableFieldInfo * f ) +{ + if ( TableJsonIsMap( f ) ) return 'o'; // a MAP: an object keyed by the KEY (§2.8) + if ( f->kind == 12 ) return 's'; // string + if ( TableJsonIsBytes( f ) ) return 's'; // bytes: base64 + if ( TableJsonIsKeyed( f ) ) return 'o'; // an object keyed by variant NAME + if ( f->is_array ) return 'a'; + if ( f->arms != NULL ) return 'o'; // union: an object with ONE key + if ( f->kind == 13 ) return 'o'; // nested table or type + if ( f->kind == 17 ) return f->table != NULL ? 'o' : 's'; // a pointer: the pointee's object in place, or null (§16.7); a byte buffer's string (§2.5) + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// the ELEMENT shape of an array field — the same classifier one level down +inline char TableJsonElementShape( const TableFieldInfo * f ) +{ + if ( f->arms != NULL ) return 'o'; // an element of an array of unions: one key, the arm (§2.6) + if ( f->kind == 13 ) return 'o'; + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// A guarded group rides only when its guard reads true — the wire's own +// elision (§4), carried into the text so a text and a wire written from one +// instance say the same thing. The guard is spelled as its branch condition +// over bool fields of the SAME type ("at_rest", "!at_rest", +// "active && has_target"), so evaluating it is a walk of the same +// descriptor. Nothing is inferred in the other direction: reading places +// every key it can name, and the guard is a plain bool key (§16.2). +inline bool TableJsonGuardHolds( const void * base, const TableTypeInfo * info, const char * guard ) +{ + const char * p = guard; + for ( ;; ) + { + while ( *p == ' ' || *p == '&' ) { p++; } + if ( *p == 0 ) { return true; } + bool want = true; + if ( *p == '!' ) { want = false; p++; } + const char * start = p; + while ( *p != 0 && *p != ' ' && *p != '&' ) { p++; } + size_t length = (size_t) ( p - start ); + bool value = false; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( strlen( f->name ) == length && strncmp( f->name, start, length ) == 0 ) + { + value = TableJsonGetRaw( (const uint8_t *) base + f->offset, f->elem_size ) != 0; + break; + } + } + if ( value != want ) { return false; } + } +} + +// ---- writing ---- + +// The writer sink MEASURES when the buffer is NULL and WRITES when it is +// not, over one code path — so measure and write agree byte for byte, the +// wire's invariant (§9) carried across. +struct TableJsonOut +{ + char * buffer; + int64_t capacity; + int64_t offset; + bool overflow; + void * graph; // the pointered write's identity map (§16.7); NULL for a fixed table + + void raw( const char * data, int64_t count ) + { + if ( buffer != NULL ) + { + if ( offset + count > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) count ); + } + offset += count; + } + void put( char c ) { raw( &c, 1 ); } + void text( const char * s ) { raw( s, (int64_t) strlen( s ) ); } + void line( int32_t depth ) + { + put( '\n' ); + for ( int32_t i = 0; i < depth; i++ ) { raw( " ", 2 ); } + } +}; + +inline const char * TableJsonBase64Alphabet() +{ + return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; +} + +inline void TableJsonWriteBase64( TableJsonOut & out, const uint8_t * data, int32_t length ) +{ + const char * alphabet = TableJsonBase64Alphabet(); + out.put( '"' ); + int32_t i = 0; + for ( ; i + 3 <= length; i += 3 ) + { + uint32_t triple = ( uint32_t( data[i] ) << 16 ) | ( uint32_t( data[i+1] ) << 8 ) | uint32_t( data[i+2] ); + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], + alphabet[ ( triple >> 6 ) & 0x3f ], alphabet[ triple & 0x3f ] }; + out.raw( quad, 4 ); + } + if ( i < length ) + { + int32_t left = length - i; + uint32_t triple = uint32_t( data[i] ) << 16; + if ( left == 2 ) { triple |= uint32_t( data[i+1] ) << 8; } + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], '=', '=' }; + if ( left == 2 ) { quad[2] = alphabet[ ( triple >> 6 ) & 0x3f ]; } + out.raw( quad, 4 ); + } + out.put( '"' ); +} + +// One UTF-8 sequence at s, or -1 when the bytes there are not one. Rejects +// the lot: a stray continuation, an overlong form, a surrogate half, and +// anything past U+10FFFF. +inline int32_t TableJsonUtf8( const char * s, int32_t remaining, int32_t * width ) +{ + unsigned char lead = (unsigned char) s[0]; + int32_t want = 0; + int32_t code = 0; + if ( lead < 0x80 ) { *width = 1; return lead; } + else if ( lead >= 0xc2 && lead <= 0xdf ) { want = 2; code = lead & 0x1f; } + else if ( lead >= 0xe0 && lead <= 0xef ) { want = 3; code = lead & 0x0f; } + else if ( lead >= 0xf0 && lead <= 0xf4 ) { want = 4; code = lead & 0x07; } + else { return -1; } + if ( remaining < want ) { return -1; } + for ( int32_t i = 1; i < want; i++ ) + { + unsigned char next = (unsigned char) s[i]; + if ( ( next & 0xc0 ) != 0x80 ) { return -1; } + code = ( code << 6 ) | ( next & 0x3f ); + } + if ( want == 3 && code < 0x800 ) { return -1; } // overlong + if ( want == 4 && code < 0x10000 ) { return -1; } // overlong + if ( code >= 0xd800 && code <= 0xdfff ) { return -1; } // a surrogate half + if ( code > 0x10ffff ) { return -1; } + *width = want; + return code; +} + +// A JSON text MUST be valid UTF-8 (RFC 8259 §8.1). The read path is +// byte-transparent — the wire imposes no encoding (§3) and a string may hold +// anything — so the WRITER is where that obligation is met: a byte that is +// not part of a well-formed sequence is written as U+FFFD, one per bad byte, +// and never raw. A text this walk writes is therefore readable by any +// conforming parser, which a raw byte would not be. The cost is stated +// plainly: for a string holding invalid UTF-8, the round trip is NOT +// byte-identical, because the alternative is emitting a text that is not +// JSON. +inline void TableJsonWriteString( TableJsonOut & out, const char * s, int32_t length ) +{ + static const char hex[] = "0123456789abcdef"; + out.put( '"' ); + for ( int32_t i = 0; i < length; i++ ) + { + unsigned char c = (unsigned char) s[i]; + switch ( c ) + { + case '"': out.raw( "\\\"", 2 ); break; + case '\\': out.raw( "\\\\", 2 ); break; + case '\b': out.raw( "\\b", 2 ); break; + case '\f': out.raw( "\\f", 2 ); break; + case '\n': out.raw( "\\n", 2 ); break; + case '\r': out.raw( "\\r", 2 ); break; + case '\t': out.raw( "\\t", 2 ); break; + default: + if ( c < 0x20 ) + { + char escape[6] = { '\\', 'u', '0', '0', hex[ c >> 4 ], hex[ c & 0xf ] }; + out.raw( escape, 6 ); + } + else if ( c < 0x80 ) + { + out.put( (char) c ); + } + else + { + int32_t width = 0; + if ( TableJsonUtf8( s + i, length - i, &width ) < 0 ) + { + out.raw( "\xef\xbf\xbd", 3 ); // U+FFFD, one per bad byte + } + else + { + out.raw( s + i, width ); + i += width - 1; + } + } + break; + } + } + out.put( '"' ); +} + +inline void TableJsonWriteUnsigned( TableJsonOut & out, uint64_t value ) +{ + char digits[24]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) ( value % 10 ) ); + value /= 10; + } while ( value != 0 ); + char text[24]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); +} + +inline void TableJsonWriteSigned( TableJsonOut & out, int64_t value ) +{ + if ( value < 0 ) + { + out.put( '-' ); + TableJsonWriteUnsigned( out, uint64_t( 0 ) - (uint64_t) value ); + return; + } + TableJsonWriteUnsigned( out, (uint64_t) value ); +} + +// A wide kind writes its raw storage as §16.2's text: a 128-bit integer as a +// decimal integer; a fixed value in WHOLE UNITS as the shortest exact decimal +// with at least one fractional digit (1.0, -0.25), the spelling the schema text +// gives a fixed default. The fraction terminates because a dyadic fraction has +// a finite decimal expansion — at most F digits. +inline void TableJsonWriteWide( TableJsonOut & out, const void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + TableJsonWide v = TableJsonWideLoad( storage, f->elem_size, is_signed ); + if ( is_signed && TableJsonWideNegative( v ) ) + { + out.put( '-' ); + v = TableJsonWideNeg( v ); + } + int frac = f->frac_bits; + TableJsonWide whole = TableJsonWideShr( v, frac ); + char digits[40]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) TableJsonWideDiv( &whole, 10 ) ); + } while ( !TableJsonWideZero( whole ) ); + char text[40]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); + if ( !TableJsonKindFixed( f->kind ) ) { return; } + out.put( '.' ); + // the fraction bits alone: v with everything at and above bit F cleared + TableJsonWide fraction = v; + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + if ( frac == 0 ) { fraction.lo = 0; } + if ( TableJsonWideZero( fraction ) ) + { + out.put( '0' ); + return; + } + while ( !TableJsonWideZero( fraction ) ) + { + // ×10: the digit is what lands at and above bit F, including the + // carry out of 128 bits when F leaves no room for it below + uint32_t carry = TableJsonWideMulAdd( &fraction, 10, 0 ); + uint64_t digit = TableJsonWideShr( fraction, frac ).lo; + if ( frac > 64 ) { digit |= uint64_t( carry ) << ( 128 - frac ); } + out.put( (char) ( '0' + (int) digit ) ); + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + } +} + +// A float writes at the SHORTEST precision that reads back as the same value +// at the field's own width, so a round trip is exact and a text stays +// readable. Non-finite values have no JSON spelling at all, and the writer +// REFUSES rather than losing one silently — the same rule measure and save +// already apply to an enum value no variant names (§5). +inline bool TableJsonWriteFloat( TableJsonOut & out, double value, bool single ) +{ + if ( !TableJsonFinite( value ) ) { return false; } + char text[64]; + int low = single ? 6 : 15; + int high = single ? 9 : 17; + int length = 0; + for ( int digits = low; ; digits++ ) + { + length = snprintf( text, sizeof( text ), "%.*g", digits, value ); + if ( length <= 0 || length >= (int) sizeof( text ) ) { return false; } + if ( digits >= high ) { break; } + // the round-trip check runs BEFORE the decimal point is normalised: + // the token still carries whatever point snprintf just produced + if ( single ) + { + if ( (double) strtof( text, NULL ) == value ) { break; } + } + else + { + if ( strtod( text, NULL ) == value ) { break; } + } + } + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int i = 0; i < length; i++ ) + { + if ( text[i] == point ) { text[i] = '.'; } + } + } + out.raw( text, length ); + return true; +} + +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration writes through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ); + +// one scalar, at one storage address: a nested object, a union, a +// vocabulary, or a number +inline bool TableJsonWriteScalar( TableJsonOut & out, const void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; None is {} + const TableUnionInfo * arms = f->arms(); + uint64_t tag = TableJsonGetRaw( (const uint8_t *) storage + arms->tag_offset, arms->tag_size ); + if ( tag == 0 ) + { + out.raw( "{}", 2 ); + return true; + } + if ( (int64_t) tag > f->enum_max ) + { + return false; // a tag no arm names, exactly as measure refuses it + } + const char * arm = f->enum_name( tag ); + // and refuse on the NAME, not merely on the bound: §16.2 says a value + // no variant NAMES is refused, so the check is the name. Writing + // whatever came back would emit "???", a spelling the reader counts + // as unknown — a silent round-trip loss in place of a refusal. + if ( !TableJsonNamed( arm ) ) { return false; } + out.put( '{' ); + out.line( depth + 1 ); + TableJsonWriteString( out, arm, (int32_t) strlen( arm ) ); + out.raw( ": ", 2 ); + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2): an arm that names + // no declaration carries the FIELD descriptor a field of its type + // would carry, offsets taken inside the union storage (§2.6), so the + // value walks through the field writer one key down. + if ( arms->arms[tag].field != NULL ) + { + if ( !TableJsonWriteField( out, storage, arms->arms[tag].field, depth + 1 ) ) + { + return false; + } + } + else if ( arms->arms[tag].table == NULL ) + { + out.raw( "null", 4 ); // a payload-free arm: the name selects it (§2.6) + } + else if ( !TableJsonWriteValue( out, (const uint8_t *) storage + arms->arms[tag].offset, arms->arms[tag].table, depth + 1 ) ) + { + return false; + } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->kind == 13 ) + { + return TableJsonWriteValue( out, storage, f->table, depth ); + } + if ( TableJsonIsEnum( f ) ) + { + uint64_t value = TableJsonGetRaw( storage, f->elem_size ); + // a value no variant names has no text spelling, exactly as it has no + // wire identity: the writer REFUSES rather than writing None over it, + // the rule measure and save already apply (docs/SPEC-TABLES.md §5) + if ( (int64_t) value > f->enum_max ) { return false; } + if ( value != 0 && f->variant_id( value ) == 0 ) { return false; } + const char * name = f->enum_name( value ); + if ( !TableJsonNamed( name ) ) { return false; } + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + return true; + } + if ( TableJsonIsFlags( f ) ) + { + uint64_t bits = TableJsonGetRaw( storage, f->elem_size ); + if ( bits == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + bool first = true; + for ( int64_t bit = 0; bit < 64; bit++ ) + { + if ( ( bits & ( uint64_t( 1 ) << bit ) ) == 0 ) { continue; } + if ( bit > f->enum_max ) + { + return false; // a bit no variant names has no text spelling + } + const char * name = f->enum_name( (uint64_t) bit ); + if ( !TableJsonNamed( name ) ) { return false; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + } + out.line( depth ); + out.put( ']' ); + return true; + } + switch ( f->kind ) + { + case 1: + out.text( TableJsonGetRaw( storage, f->elem_size ) != 0 ? "true" : "false" ); + return true; + case 10: + { + float v = 0.0f; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, (double) v, true ); + } + case 11: + { + double v = 0.0; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, v, false ); + } + case 2: case 3: case 4: case 5: + TableJsonWriteSigned( out, TableJsonGetSigned( storage, f->elem_size ) ); + return true; + default: + if ( TableJsonKindWide( f->kind ) ) + { + TableJsonWriteWide( out, storage, f ); + return true; + } + TableJsonWriteUnsigned( out, TableJsonGetRaw( storage, f->elem_size ) ); + return true; + } +} + +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ) +{ + const uint8_t * storage = (const uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonWriteMap( out, (const void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } + if ( f->kind == 17 && !f->is_array ) + { + return TableJsonWritePointer( out, storage, f, depth ); + } + if ( f->kind == 17 ) + { + // an ARRAY OF POINTERS (§2.1): the pointer row per element — the + // pointee's object in place, null, or `&node` for a shared one (§16.7) + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWritePointer( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; + } + if ( f->kind == 12 ) + { + TableJsonWriteString( out, (const char *) storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + TableJsonWriteBase64( out, storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + // one entry per SLOT, keyed by the variant that owns it, so inserting + // a variant next season moves nothing in the text either. Slot i holds + // the key i + 1: nothing is stored for None, so nothing is written for it. + out.put( '{' ); + bool first = true; + for ( int64_t slot = 0; slot < f->array_bound; slot++ ) + { + if ( !TableJsonKeyedSlotValid( f, slot ) ) { continue; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + const char * key = f->key_name( TableJsonKeyedSlotKey( slot ) ); + TableJsonWriteString( out, key, (int32_t) strlen( key ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteScalar( out, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + if ( first ) { out.raw( "}", 1 ); return true; } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->is_array ) + { + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWriteScalar( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + out.line( depth ); + out.put( ']' ); + return true; + } + return TableJsonWriteScalar( out, storage, f, depth ); +} + +// One instance's fields, in DECLARATION ORDER, defaults included — a text is +// for people and tools, and a text that elides is a text a reader has to know +// the schema to complete. `any` says whether the object is already open on +// entry — a shared node's `&node` opens it before the fields (§16.7) — and +// whether it is open on return. +inline bool TableJsonWriteFields( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth, bool & any ) +{ + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( f->guard[0] != 0 && !TableJsonGuardHolds( base, info, f->guard ) ) { continue; } + // an ABSENT optional writes no key: presence of the key IS the + // presence (§16.2), so an absent field is an absent key and nothing + // else would read back as absent + if ( f->optional && + TableJsonGetRaw( (const uint8_t *) base + f->present_offset, 1 ) == 0 ) + { + continue; + } + if ( !any ) { out.put( '{' ); } + else { out.put( ',' ); } + any = true; + out.line( depth + 1 ); + TableJsonWriteString( out, f->json, (int32_t) strlen( f->json ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, base, f, depth + 1 ) ) { return false; } + } + return true; +} + +// One instance as one object. The writer carries the reader's depth cap +// (§16.2): a pointer chain nests as deep as it is long (§16.7), and a text the +// writer produced past the cap would be a text the reader refuses. +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { return false; } + bool any = false; + if ( !TableJsonWriteFields( out, base, info, depth, any ) ) { return false; } + if ( !any ) + { + out.raw( "{}", 2 ); + return true; + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- reading ---- + +struct TableJsonIn +{ + const char * text; + int64_t size; + int64_t pos; + TableReport * report; + bool bad; // the text is not JSON: the walk stops and keeps what it placed + void * graph; // the pointered read's builder and label map (§16.7); NULL for a fixed table +}; + +inline void TableJsonSpace( TableJsonIn & in ) +{ + while ( in.pos < in.size ) + { + char c = in.text[in.pos]; + if ( c == ' ' || c == '\t' || c == '\n' || c == '\r' ) { in.pos++; continue; } + // comments are not JSON, and a walk that guessed at one would be + // reading a dialect nobody wrote down + if ( c == '/' ) { in.bad = true; } + return; + } +} + +inline char TableJsonPeek( TableJsonIn & in ) +{ + TableJsonSpace( in ); + return in.pos < in.size ? in.text[in.pos] : 0; +} + +// the shape of the value sitting at the cursor, without consuming it +inline char TableJsonValueShape( TableJsonIn & in ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return 'o'; + case '[': return 'a'; + case '"': return 's'; + case 't': case 'f': return 'b'; + case 'n': return 'z'; + case 0: return 0; + default: return 'n'; + } +} + +inline bool TableJsonLiteral( TableJsonIn & in, const char * word ) +{ + int64_t length = (int64_t) strlen( word ); + if ( in.pos + length > in.size || memcmp( in.text + in.pos, word, (size_t) length ) != 0 ) + { + in.bad = true; + return false; + } + in.pos += length; + return true; +} + +// one \uXXXX escape body; -1 when the four hex digits are not there +inline int TableJsonHex4( TableJsonIn & in ) +{ + if ( in.pos + 4 > in.size ) { return -1; } + int value = 0; + for ( int i = 0; i < 4; i++ ) + { + char c = in.text[in.pos + i]; + int digit; + if ( c >= '0' && c <= '9' ) { digit = c - '0'; } + else if ( c >= 'a' && c <= 'f' ) { digit = c - 'a' + 10; } + else if ( c >= 'A' && c <= 'F' ) { digit = c - 'A' + 10; } + else { return -1; } + value = ( value << 4 ) | digit; + } + in.pos += 4; + return value; +} + +inline int32_t TableJsonEncodeUtf8( uint32_t code, char * unit ) +{ + if ( code < 0x80 ) { unit[0] = (char) code; return 1; } + if ( code < 0x800 ) + { + unit[0] = (char) ( 0xc0 | ( code >> 6 ) ); + unit[1] = (char) ( 0x80 | ( code & 0x3f ) ); + return 2; + } + if ( code < 0x10000 ) + { + unit[0] = (char) ( 0xe0 | ( code >> 12 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( code & 0x3f ) ); + return 3; + } + unit[0] = (char) ( 0xf0 | ( code >> 18 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 12 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[3] = (char) ( 0x80 | ( code & 0x3f ) ); + return 4; +} + +// Scan one JSON string into a caller buffer. Bytes are appended ONE CODE +// POINT AT A TIME — an escape's encoding, or a UTF-8 sequence read whole — +// so a string longer than the field is clamped AT A CODE POINT BOUNDARY and +// never cut through a multi-byte character. Clamping is counted, never +// fatal, exactly as it is on the wire (§4). A NULL destination scans past a +// string without keeping it. +inline bool TableJsonScanString( TableJsonIn & in, char * out, int32_t capacity, int32_t * length ) +{ + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + int32_t placed = 0; + bool clamped = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos]; + if ( c == '"' ) { in.pos++; break; } + char unit[4]; + int32_t unit_length = 0; + if ( c == '\\' ) + { + in.pos++; + if ( in.pos >= in.size ) { in.bad = true; return false; } + char escape = in.text[in.pos++]; + switch ( escape ) + { + case '"': unit[0] = '"'; unit_length = 1; break; + case '\\': unit[0] = '\\'; unit_length = 1; break; + case '/': unit[0] = '/'; unit_length = 1; break; + case 'b': unit[0] = '\b'; unit_length = 1; break; + case 'f': unit[0] = '\f'; unit_length = 1; break; + case 'n': unit[0] = '\n'; unit_length = 1; break; + case 'r': unit[0] = '\r'; unit_length = 1; break; + case 't': unit[0] = '\t'; unit_length = 1; break; + case 'u': + { + int high = TableJsonHex4( in ); + if ( high < 0 ) { in.bad = true; return false; } + uint32_t code = (uint32_t) high; + if ( high >= 0xd800 && high <= 0xdbff && in.pos + 2 <= in.size && + in.text[in.pos] == '\\' && in.text[in.pos + 1] == 'u' ) + { + int64_t mark = in.pos; + in.pos += 2; + int low = TableJsonHex4( in ); + if ( low >= 0xdc00 && low <= 0xdfff ) + { + code = 0x10000 + ( ( (uint32_t) high - 0xd800 ) << 10 ) + ( (uint32_t) low - 0xdc00 ); + } + else + { + in.pos = mark; // a lone lead surrogate rides as itself + } + } + // a surrogate half that never found its partner has no + // UTF-8 encoding: encoding it anyway would manufacture + // CESU-8 — invalid UTF-8 — out of input that was valid + // JSON, so it reads as the replacement character + if ( code >= 0xd800 && code <= 0xdfff ) { code = 0xfffd; } + unit_length = TableJsonEncodeUtf8( code, unit ); + break; + } + default: in.bad = true; return false; + } + } + else if ( (unsigned char) c < 0x20 ) + { + in.bad = true; // a raw control character is not a JSON string body + return false; + } + else + { + // a UTF-8 sequence read WHOLE, so the clamp below can only land + // between code points. Only bytes that ACTUALLY look like + // continuations are taken: the wire imposes no encoding (§3), so + // a string may legitimately hold a stray lead byte, and one at + // the end of a text must not swallow the closing quote. + unsigned char lead = (unsigned char) c; + int32_t want = 1; + if ( ( lead & 0xe0 ) == 0xc0 ) { want = 2; } + else if ( ( lead & 0xf0 ) == 0xe0 ) { want = 3; } + else if ( ( lead & 0xf8 ) == 0xf0 ) { want = 4; } + unit[0] = c; + in.pos++; + unit_length = 1; + while ( unit_length < want && in.pos < in.size && + ( (unsigned char) in.text[in.pos] & 0xc0 ) == 0x80 ) + { + unit[unit_length++] = in.text[in.pos++]; + } + } + if ( out == NULL ) + { + placed += unit_length; // measured and not kept: a byte buffer's read sizes its node this way (§2.5) + } + else if ( placed + unit_length <= capacity ) + { + memcpy( out + placed, unit, (size_t) unit_length ); + placed += unit_length; + } + else + { + clamped = true; + } + } + if ( clamped ) { in.report->clamped++; } + if ( length != NULL ) { *length = placed; } + return true; +} + +// the numeric token at the cursor, copied out whole; false = not a number +// Scan one number, to JSON's OWN grammar (RFC 8259 §6) and not to a run of +// number-ish characters: +// +// number = [ "-" ] int [ frac ] [ exp ] +// int = "0" / ( digit1-9 *digit ) +// frac = "." 1*digit +// exp = ( "e" / "E" ) [ "-" / "+" ] 1*digit +// +// Scanning the production is what makes a typo in an authoring file a +// DIAGNOSTIC rather than a value: "1-2" scans as 1 and leaves "-2" where the +// object expects a comma, so the text is malformed — which is what §16.2 +// already promises. A permissive scan would hand "1-2" to a digit loop and +// report a clamp, and a config pipeline would never hear about it. Leading +// "+", leading zeros, ".5" and "3." are not JSON either. +inline bool TableJsonWalkNumber( TableJsonIn & in, bool * integral ) +{ + TableJsonSpace( in ); + bool whole = true; + if ( in.pos < in.size && in.text[in.pos] == '-' ) { in.pos++; } + // int: a lone zero, or a non-zero digit and any digits after it + if ( in.pos >= in.size ) { return false; } + if ( in.text[in.pos] == '0' ) + { + in.pos++; + } + else if ( in.text[in.pos] >= '1' && in.text[in.pos] <= '9' ) + { + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + } + else + { + return false; + } + // frac + if ( in.pos < in.size && in.text[in.pos] == '.' ) + { + in.pos++; + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + // exp + if ( in.pos < in.size && ( in.text[in.pos] == 'e' || in.text[in.pos] == 'E' ) ) + { + in.pos++; + if ( in.pos < in.size && ( in.text[in.pos] == '-' || in.text[in.pos] == '+' ) ) { in.pos++; } + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + *integral = whole; + return true; +} + +// the same production, with the token kept for conversion +inline bool TableJsonScanNumber( TableJsonIn & in, char * token, int32_t capacity, int32_t * length, bool * integral ) +{ + TableJsonSpace( in ); + int64_t start = in.pos; + if ( !TableJsonWalkNumber( in, integral ) ) { return false; } + int64_t count = in.pos - start; + if ( count <= 0 || count >= capacity ) { return false; } + memcpy( token, in.text + start, (size_t) count ); + token[count] = 0; + *length = (int32_t) count; + return true; +} + +// the token's exact double, through the runtime's own converter — which +// speaks the LOCALE's decimal point, so the token crosses back over that +// character on its way in +inline double TableJsonTokenDouble( const char * token, int32_t length, bool single ) +{ + char work[kTableJsonMaxNumber]; + memcpy( work, token, (size_t) length ); + work[length] = 0; + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int32_t i = 0; i < length; i++ ) + { + if ( work[i] == '.' ) { work[i] = point; } + } + } + if ( single ) { return (double) strtof( work, NULL ); } + return strtod( work, NULL ); +} + +// the token's exact integer, parsed digit by digit so no width and no +// locale can move it. Saturation is reported as a clamp, the wire's rule for +// a value outside what the reader can hold (§4). +inline int64_t TableJsonTokenInteger( const char * token, int32_t length, bool is_signed, bool * saturated ) +{ + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) + { + negative = token[i] == '-'; + i++; + } + uint64_t magnitude = 0; + bool over = false; + for ( ; i < length; i++ ) + { + uint64_t digit = (uint64_t) ( token[i] - '0' ); + if ( magnitude > ( UINT64_MAX - digit ) / 10 ) { over = true; break; } + magnitude = magnitude * 10 + digit; + } + if ( !is_signed ) + { + // -0 IS zero, and clamping it would report an event that did not + // happen; only a real negative magnitude is out of range here + if ( negative ) { *saturated = magnitude != 0; return 0; } + if ( over ) { *saturated = true; return (int64_t) UINT64_MAX; } + *saturated = false; + return (int64_t) magnitude; + } + if ( negative ) + { + if ( over || magnitude > ( uint64_t( 1 ) << 63 ) ) { *saturated = true; return INT64_MIN; } + *saturated = false; + if ( magnitude == ( uint64_t( 1 ) << 63 ) ) { return INT64_MIN; } + return -(int64_t) magnitude; + } + if ( over || magnitude > (uint64_t) INT64_MAX ) { *saturated = true; return INT64_MAX; } + *saturated = false; + return (int64_t) magnitude; +} + +// A number token into a wide kind's raw storage (docs/SPEC-TABLES.md §16.2). A +// 128-bit integer takes any token whose VALUE is integral; a fixed field any +// token whose value is EXACTLY representable in its Q I.F — a finer fraction +// is the wrong shape for the field, counted as a kind mismatch and never +// rounded, the rule SPEC.md §4.6 gives a fixed default. A magnitude past 128 +// bits saturates and counts as a clamp, as an int64 field saturates at +// INT64_MAX; the declared range clamps after it, on the RAW scale, as it does +// for every bounded scalar. +// +// The token is normalized to its digits with the decimal point after "point" +// of them. An integer part past 40 digits is above 2^128 whatever the digits +// are, and a value below 10^-40 is finer than 2^-127, the finest fraction any +// F can spell — so outside that band the answer is known without the +// arithmetic, and a token spelling 1e999999999 costs nothing to refuse. +inline bool TableJsonReadWide( TableJsonIn & in, const char * token, int32_t length, void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + int frac = f->frac_bits; + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { negative = token[i] == '-'; i++; } + const char * int_digits = token + i; + int32_t int_len = 0; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { int_len++; i++; } + const char * frac_digits = token + i; + int32_t frac_len = 0; + if ( i < length && token[i] == '.' ) + { + i++; + frac_digits = token + i; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { frac_len++; i++; } + } + int64_t exp = 0; + if ( i < length && ( token[i] == 'e' || token[i] == 'E' ) ) + { + i++; + bool exp_negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { exp_negative = token[i] == '-'; i++; } + while ( i < length && token[i] >= '0' && token[i] <= '9' ) + { + if ( exp < 100000 ) { exp = exp * 10 + ( token[i] - '0' ); } + i++; + } + if ( exp_negative ) { exp = -exp; } + } + // the digits, with the point after "point" of them; leading and trailing + // zeros stripped. digit( k ) reads the k-th of the int and frac runs. + int32_t start = 0, end = int_len + frac_len; + int64_t point = int_len + exp; + while ( start < end && ( start < int_len ? int_digits[start] : frac_digits[start - int_len] ) == '0' ) { start++; point--; } + while ( end > start && ( end - 1 < int_len ? int_digits[end - 1] : frac_digits[end - 1 - int_len] ) == '0' ) { end--; } + + TableJsonWide raw = { 0, 0 }; + bool saturated = false; + TableJsonWide signed_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) >> 1 }; + TableJsonWide signed_min = { 0, uint64_t( 1 ) << 63 }; + TableJsonWide unsigned_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) }; + if ( start == end ) + { + // zero, and -0 IS zero + } + else if ( point > 40 ) + { + saturated = true; + if ( !negative ) { raw = is_signed ? signed_max : unsigned_max; } + else if ( is_signed ) { raw = signed_min; } + } + else if ( point < -40 ) + { + in.report->kind_mismatch++; // finer than any F can spell + return true; + } + else + { + // the fraction FIRST, so an inexact value is the wrong shape whatever + // its magnitude: its digits, with the zeros a negative point puts in + // front, doubled F times; each doubling's carry is the next bit, and + // the value is exact iff nothing is left after the last one + char fd[kTableJsonMaxNumber + 48]; + int32_t fn = 0; + for ( int64_t z = point; z < 0; z++ ) { fd[fn++] = 0; } + for ( int32_t k = (int32_t) ( point > 0 ? point : 0 ) + start; k < end; k++ ) + { + fd[fn++] = (char) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ); + } + TableJsonWide fraction = { 0, 0 }; + for ( int b = 0; b < frac; b++ ) + { + int carry = 0; + for ( int32_t k = fn - 1; k >= 0; k-- ) + { + int d = fd[k] * 2 + carry; + fd[k] = (char) ( d % 10 ); + carry = d / 10; + } + fraction = TableJsonWideShl( fraction, 1 ); + fraction.lo |= (uint64_t) carry; + } + for ( int32_t k = 0; k < fn; k++ ) + { + if ( fd[k] != 0 ) + { + in.report->kind_mismatch++; + return true; + } + } + // then the whole part, saturating past 128 bits + TableJsonWide whole = { 0, 0 }; + for ( int64_t k = start; k < start + point && !saturated; k++ ) + { + uint32_t digit = k < end ? (uint32_t) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ) : 0; + if ( TableJsonWideMulAdd( &whole, 10, digit ) != 0 ) { saturated = true; } + } + if ( !saturated && frac > 0 && !TableJsonWideZero( TableJsonWideShr( whole, 128 - frac ) ) ) { saturated = true; } + if ( !saturated ) + { + raw = TableJsonWideShl( whole, frac ); + raw.lo |= fraction.lo; + raw.hi |= fraction.hi; + } + if ( is_signed ) + { + if ( !saturated && !negative && TableJsonWideNegative( raw ) ) { saturated = true; } + if ( !saturated && negative && TableJsonWideCompare( raw, signed_min, false ) > 0 ) { saturated = true; } + if ( saturated ) { raw = negative ? signed_min : signed_max; } + else if ( negative ) { raw = TableJsonWideNeg( raw ); } + } + else + { + if ( saturated ) { raw = unsigned_max; } + if ( negative && !TableJsonWideZero( raw ) ) { raw.lo = 0; raw.hi = 0; saturated = true; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->wide != NULL ) + { + TableJsonWide lo = { f->wide->lo[0], f->wide->lo[1] }; + TableJsonWide hi = { f->wide->hi[0], f->wide->hi[1] }; + if ( TableJsonWideCompare( raw, lo, is_signed ) < 0 ) { raw = lo; in.report->clamped++; } + else if ( TableJsonWideCompare( raw, hi, is_signed ) > 0 ) { raw = hi; in.report->clamped++; } + } + TableJsonWideStore( storage, f->elem_size, raw ); + return true; +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ); + +inline bool TableJsonSkipContainer( TableJsonIn & in, char close, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; // the opening bracket + bool first = true; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == close ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + if ( close == '}' ) + { + // the key is kept, because a skipped OBJECT may still be a + // pointer's: an `&node` opening it names a node the storage could + // not hold, and the numbering has to survive the drop (§16.7). + // Anywhere but first, the prefix is the reserved key out of place + // — in a pointered unit; a fixed unit skips the value whole. + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( key[0] == '&' && in.graph != NULL ) + { + if ( !first ) { in.report->malformed = true; in.bad = true; return false; } + if ( !TableJsonSkippedAmpersand( in, key, depth ) ) { return false; } + first = false; + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } + } + first = false; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return TableJsonSkipContainer( in, '}', depth ); + case '[': return TableJsonSkipContainer( in, ']', depth ); + case '"': return TableJsonScanString( in, NULL, 0, NULL ); + case 't': return TableJsonLiteral( in, "true" ); + case 'f': return TableJsonLiteral( in, "false" ); + case 'n': return TableJsonLiteral( in, "null" ); + case 0: in.bad = true; return false; + default: + { + // consumed, never converted: skipping needs no buffer, and this + // is the one walk a hostile text drives to the depth cap. It is + // the SAME production the value path scans, so an unknown key + // cannot smuggle past a number a named key would refuse. + bool integral = false; + if ( !TableJsonWalkNumber( in, &integral ) ) { in.bad = true; return false; } + return true; + } + } +} + +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration reads through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ); + +// place one scalar at one storage address +inline bool TableJsonReadScalar( TableJsonIn & in, void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; {} is None, and + // two keys is a text this walk will not guess at + const TableUnionInfo * arms = f->arms(); + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, 0 ); + if ( TableJsonPeek( in ) == '}' ) { in.pos++; return true; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t tag = 0; + for ( int64_t t = 1; t <= f->enum_max; t++ ) + { + if ( strcmp( f->enum_name( (uint64_t) t ), key ) == 0 ) { tag = t; break; } + } + if ( tag == 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + void * payload = (uint8_t *) storage + arms->arms[tag].offset; + const TableFieldInfo * arm = arms->arms[tag].field; + bool placed = true; + if ( arm != NULL ) + { + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2). A value of + // the wrong shape for that row is a KIND MISMATCH: the union + // reads None, the event is counted, and the enclosing object + // continues — the rule a FIELD's value lives under, one key + // down. A pointer arm's null is a null pointer, not a shape + // error, exactly as a pointer field's is (§16.7). + char got = TableJsonValueShape( in ); + if ( arm->kind == 17 && !arm->is_array && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + memset( payload, 0, (size_t) arms->arms[tag].size ); + } + else if ( got != TableJsonShape( arm ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( arm->kind == 17 && !arm->is_array ) + { + // A POINTER ARM'S VALUE IS THE POINTEE IN PLACE, or a + // node reference to one (§16.7) — the read a pointer + // FIELD takes, which is not the scalar walk + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadPointer( in, payload, arm, depth + 1 ) ) { return false; } + } + else + { + // SELECTION ZERO-ESTABLISHES THE ARM (SPEC §5): an arm + // takes no specified default, so zero is the establish + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadField( in, storage, arm, depth + 1 ) ) { return false; } + } + } + else if ( arms->arms[tag].table != NULL ) + { + if ( TableJsonValueShape( in ) != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else + { + arms->arms[tag].table->reset( payload ); + if ( !TableJsonReadTable( in, payload, arms->arms[tag].table, depth + 1 ) ) { return false; } + } + } + else + { + // A PAYLOAD-FREE ARM'S VALUE IS null (§2.6): the arm name + // selects it and there is nothing to place + if ( TableJsonValueShape( in ) != 'z' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( !TableJsonLiteral( in, "null" ) ) + { + return false; + } + } + if ( placed ) + { + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, (uint64_t) tag ); + } + } + char c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; // a second key: a one-of with two arms is not a value + return false; + } + if ( f->kind == 13 ) + { + f->table->reset( storage ); + return TableJsonReadTable( in, storage, f->table, depth + 1 ); + } + if ( TableJsonIsEnum( f ) ) + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + for ( int64_t v = 0; v <= f->enum_max; v++ ) + { + if ( strcmp( f->enum_name( (uint64_t) v ), name ) == 0 ) + { + TableJsonSetRaw( storage, f->elem_size, (uint64_t) v ); + return true; + } + } + // a name this build cannot name reads as None and counts as unknown, + // exactly as an unknown variant id does on the wire (§4) + TableJsonSetRaw( storage, f->elem_size, 0 ); + in.report->unknown++; + return true; + } + if ( TableJsonIsFlags( f ) ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + uint64_t bits = 0; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( c != '"' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + bool found = false; + for ( int64_t bit = 0; bit <= f->enum_max; bit++ ) + { + if ( strcmp( f->enum_name( (uint64_t) bit ), name ) == 0 ) + { + bits |= uint64_t( 1 ) << bit; + found = true; + break; + } + } + if ( !found ) { in.report->unknown++; } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + TableJsonSetRaw( storage, f->elem_size, bits ); + return true; + } + if ( f->kind == 1 ) + { + char c = TableJsonPeek( in ); + if ( c == 't' ) { if ( !TableJsonLiteral( in, "true" ) ) { return false; } TableJsonSetRaw( storage, f->elem_size, 1 ); return true; } + if ( !TableJsonLiteral( in, "false" ) ) { return false; } + TableJsonSetRaw( storage, f->elem_size, 0 ); + return true; + } + char token[kTableJsonMaxNumber]; + int32_t length = 0; + bool integral = false; + if ( !TableJsonScanNumber( in, token, kTableJsonMaxNumber, &length, &integral ) ) + { + in.bad = true; + return false; + } + if ( TableJsonKindWide( f->kind ) ) + { + return TableJsonReadWide( in, token, length, storage, f ); + } + if ( f->kind == 10 || f->kind == 11 ) + { + bool single = f->kind == 10; + double value = TableJsonTokenDouble( token, length, single ); + // A magnitude the field's format cannot hold is the WRONG SHAPE for + // the kind, and it never reaches storage: 1e400 is not a float64 and + // 1e300 is not a float32. Storing the infinity the conversion + // produced would leave an instance this walk called CLEAN that + // ToJsonMeasure then refuses forever (a non-finite float has no JSON + // spelling), and §16.1's one invariant is that a text which reads + // clean writes back. + if ( !TableJsonFinite( value ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( f->has_range ) + { + if ( value < f->range_min ) { value = f->range_min; in.report->clamped++; } + else if ( value > f->range_max ) { value = f->range_max; in.report->clamped++; } + } + if ( single ) + { + float narrow = (float) value; + if ( !TableJsonFinite( (double) narrow ) ) + { + in.report->kind_mismatch++; + return true; + } + memcpy( storage, &narrow, sizeof( narrow ) ); + } + else + { + memcpy( storage, &value, sizeof( value ) ); + } + return true; + } + // JSON HAS ONE NUMBER TYPE. 2.0 IS the integer 2 and 1e3 IS 1000, and a + // library that round-trips numbers through a double emits them that way — + // this walker's own float writer emits 1e+21. So an integer field takes + // any number whose VALUE is integral, however it was spelled; only a + // genuinely fractional value is the wrong shape for it. + bool is_signed = f->kind >= 2 && f->kind <= 5; + bool saturated = false; + int64_t value = 0; + if ( integral ) + { + value = TableJsonTokenInteger( token, length, is_signed, &saturated ); + } + else + { + double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( is_signed ) + { + if ( d >= 9223372036854775808.0 ) { value = INT64_MAX; saturated = true; } + else if ( d < -9223372036854775808.0 ) { value = INT64_MIN; saturated = true; } + else if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) d; } + } + else + { + if ( d < 0.0 ) + { + // a negative for an unsigned field clamps to zero, as the + // exact digit path already does + if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + value = 0; + saturated = true; + } + else if ( d >= 18446744073709551616.0 ) { value = (int64_t) UINT64_MAX; saturated = true; } + else if ( d != (double) (uint64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) (uint64_t) d; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->has_range ) + { + if ( (double) value < f->range_min ) { value = (int64_t) f->range_min; in.report->clamped++; } + else if ( (double) value > f->range_max ) { value = (int64_t) f->range_max; in.report->clamped++; } + } + // the field's own storage width is the last bound: a value past it + // clamps rather than wrapping, which is what the wire does too + if ( f->elem_size < 8 ) + { + if ( is_signed ) + { + int64_t high = ( int64_t( 1 ) << ( f->elem_size * 8 - 1 ) ) - 1; + int64_t low = -high - 1; + if ( value > high ) { value = high; in.report->clamped++; } + else if ( value < low ) { value = low; in.report->clamped++; } + } + else + { + uint64_t high = ( uint64_t( 1 ) << ( f->elem_size * 8 ) ) - 1; + if ( value < 0 ) { value = 0; in.report->clamped++; } + else if ( (uint64_t) value > high ) { value = (int64_t) high; in.report->clamped++; } + } + } + // at eight bytes the storage IS the parser's width, and an unsigned value + // past INT64_MAX rides here as a negative int64 by design — the token + // parser already turned a NEGATIVE token for an unsigned field into a + // clamped zero, so there is nothing left to bound. + TableJsonSetRaw( storage, f->elem_size, (uint64_t) value ); + return true; +} + +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ) +{ + uint8_t * storage = (uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonReadMap( in, (void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + + if ( f->kind == 12 ) + { + int32_t length = 0; + if ( !TableJsonScanString( in, (char *) storage, f->array_bound, &length ) ) { return false; } + storage[length] = 0; + TableJsonSetCount( base, f, length ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + // base64 decodes STRAIGHT INTO the field's storage, six bits at a + // time — no window, no temporary, so a bytes(N) of any declared + // extent reads the same way. A base64 body carries no escapes, so a + // backslash in one is simply not an alphabet character. + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + memset( storage, 0, (size_t) f->array_bound ); + TableJsonSetCount( base, f, 0 ); + const char * alphabet = TableJsonBase64Alphabet(); + int32_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + bool clamped = false; + bool malformed = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + const char * at = c != 0 ? strchr( alphabet, c ) : NULL; + if ( at == NULL ) { malformed = true; continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( at - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < f->array_bound ) + { + storage[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); + } + else + { + clamped = true; + } + } + } + if ( malformed ) + { + // a body that is not base64 is the wrong shape for the kind: the + // field keeps its default and the event is counted + in.report->kind_mismatch++; + return true; + } + if ( clamped ) { in.report->clamped++; } + TableJsonSetCount( base, f, placed ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + // every slot back to its declared defaults first, so a key the text + // omits keeps them and a repeated field key cannot leave an earlier + // occurrence's slots standing + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + void * slot = storage + (int64_t) i * f->elem_size; + if ( f->kind == 13 ) { f->table->reset( slot ); } + else { memset( slot, 0, (size_t) f->elem_size ); } + } + char shape = TableJsonElementShape( f ); + // A KEYED OBJECT'S KEYS ARE KEYS: a variant named twice is a duplicate + // key like any other, last-wins and counted (§16.2). Tracked the way + // a table's own field keys are — a bounded, allocation-free bitmask; + // a vocabulary wider than this still reads, its repeats simply stop + // being counted. + uint64_t seen[8] = {}; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t slot = -1; + for ( int64_t v = 0; v < f->array_bound; v++ ) + { + // nothing is stored for None, so "None" finds no slot and is + // an unknown key like any other name this reader cannot place + if ( !TableJsonKeyedSlotValid( f, v ) ) { continue; } + if ( strcmp( f->key_name( TableJsonKeyedSlotKey( v ) ), key ) == 0 ) { slot = v; break; } + } + if ( slot >= 0 && slot < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( slot & 63 ); + if ( ( seen[slot >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[slot >> 6] |= bit; + } + if ( slot < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; + } + if ( f->is_array ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + // LAST WINS has to be true of a repeated ARRAY key too, and it is + // wire-visible: a fixed array writes every slot, so a second, shorter + // occurrence overlaying a prefix would leave the first occurrence's + // tail standing. The field goes back to its declared defaults before + // this occurrence's elements are placed — the re-establishment a nested + // table and a union arm already get. A table element's defaults are + // its own (the reset hook); every other element kind's storage + // default is zero, which is what the generated array declares. + if ( f->kind == 13 ) + { + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + f->table->reset( storage + (int64_t) i * f->elem_size ); + } + } + else + { + memset( storage, 0, (size_t) f->array_bound * (size_t) f->elem_size ); + } + TableJsonSetCount( base, f, 0 ); + int32_t placed = 0; + char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( placed >= f->array_bound ) + { + // more elements than the reader's bound: the bounded prefix + // is kept and the excess counts, the wire's rule (§4) + in.report->clamped++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( f->kind == 17 ) + { + // an element of an ARRAY OF POINTERS (§2.1): null is a null slot, an + // object is the pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( storage + (int64_t) placed * f->elem_size, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed++; + } + else + { + if ( !TableJsonReadScalar( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + // a fixed array's tail keeps the defaults the prefill left there, + // exactly as a short wire count does + TableJsonSetCount( base, f, placed ); + return true; + } + return TableJsonReadScalar( in, storage, f, depth ); +} + +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ); + +// ONE table object: keys are field keys, unknown ones are skipped and +// counted, a repeated key is last-wins and counted. The instance is already +// at its declared defaults when this is entered, so a key the text never +// mentions keeps the default an absent field takes on the wire (§4). +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + return TableJsonReadTableKeys( in, base, info, depth, NULL ); +} + +// The keys of an object whose brace is already consumed. A pointer's object +// opens the same way a table's does, but its FIRST key may be `&node` (§16.7) +// and the adapter that reads it has to scan the key to know — so it hands the +// key it scanned in as `first_key`, with the colon consumed, and this places +// it before scanning the rest. +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ) +{ + // duplicate tracking, bounded and allocation-free: a table with more + // fields than this still reads, its repeats simply stop being counted + uint64_t seen[8] = {}; + for ( ;; ) + { + char key[kTableJsonMaxKey]; + char c = 0; + if ( first_key != NULL ) + { + memcpy( key, first_key, strlen( first_key ) + 1 ); // scanned into a buffer this size by the caller + first_key = NULL; + } + else + { + c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + } + int32_t index = -1; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + if ( strcmp( info->fields[i].json, key ) == 0 ) { index = i; break; } + } + if ( key[0] == '&' ) + { + // THE AMPERSAND PREFIX IS RESERVED TO THE FORM (docs/SPEC-TABLES.md + // §16.7). No declaration may take a key beginning with it, so this + // is never a field this build lacks — it is the sharing construct + // somewhere it cannot stand: `&node` is the FIRST key of a pointer's + // object and nothing else, and the adapter that reads a pointer + // has consumed it before these keys are read. MALFORMED, refused + // and counted; never counted as unknown, never skipped. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( index < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + const TableFieldInfo * f = &info->fields[index]; + if ( index < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( index & 63 ); + if ( ( seen[index >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[index >> 6] |= bit; + } + // PRESENCE OF THE KEY IS THE PRESENCE (§16.2): reaching this line + // is the key being present, so an optional is set present + // whatever its value — with one exception the page names: a JSON + // null, which reads as ABSENT rather than as a value. + char got = TableJsonValueShape( in ); + if ( f->kind == 17 && !f->is_array ) + { + // a pointer: null is a null pointer, an object is the pointee + // in place or an `&node` reference to one (§16.7), a string is + // a BYTE BUFFER's bytes (§2.5), and anything else is the wrong + // shape for the kind + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) base + f->offset, f->elem_size, 0 ); + } + else if ( got != TableJsonShape( f ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) base + f->offset, f, depth ) ) + { + return false; + } + } + else if ( f->optional && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + // absent, and back at its defaults: a repeated key whose last + // occurrence is null must not leave an earlier value standing + if ( f->table != NULL ) { f->table->reset( (uint8_t *) base + f->offset ); } + else { memset( (uint8_t *) base + f->offset, 0, (size_t) f->elem_size ); } + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 0 ); + } + else + { + if ( got != TableJsonShape( f ) ) + { + // the wrong JSON type for the kind: skipped, never coerced + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, base, f, depth ) ) + { + return false; + } + if ( f->optional ) + { + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 1 ); + } + } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +// ---- the two entry points the per-table wrappers name ---- + +inline bool TableJsonRead( void * value, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = NULL; + info->reset( value ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, value, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +inline int64_t TableJsonWrite( const void * value, const TableTypeInfo * info, char * buffer, int64_t capacity ) +{ + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = NULL; + if ( !TableJsonWriteValue( out, value, info, 0 ) ) { return -1; } + // THE CANONICAL TEXT ENDS WITH EXACTLY ONE NEWLINE (docs/SPEC-TABLES.md + // §16.1). Every writer emits it — this walk, the C# walk and + // "schema unpack" — and every reader accepts a text with or without one, + // because the trailing whitespace a read already skips is what makes the + // two the same text. It is a byte of the FORM rather than a file + // convention: a text that is written to a file, pasted into a diff and + // handed back through a pipe has to be one text in all three places, and a + // buffer whose last byte is a closing brace is the one shape that is not. + out.put( '\n' ); + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json walk: end ---- + +// ---- json graph walk: begin ---- +// +// THE VARIABLE CLASS's half of the text form (docs/SPEC-TABLES.md §16.7). The +// walk above places every kind but one; this defines the three adapters it +// calls for that one, and the two entry points a pointered table's wrappers +// name. The text is the fixed class's — a pointee is an object in place — and a +// node named more than once carries `&node`: defined once, with its fields, +// and referenced after by `{ "&node": N }` alone. + +// ---- the identity map ---- +// +// ONE map shape serves both directions. Writing keys it by a node's ADDRESS and +// counts the slots that name the node, so the second pass knows at a node's +// first occurrence whether it will be named again; reading keys it by the +// text's own label and answers the node it defined. Open addressing, a +// multiply-shift hash and quadrupling growth — TablePackMap's shape (§6.2), on +// the same terms: proportional to nodes, never to bytes, on the authoring +// side, and released before the call returns. + +struct TableJsonGraphEntry +{ + uint64_t key; // a node's address (write) or a label (read); 0 is an empty slot + int64_t count; // write: how many slots name this node + int64_t label; // write: the `&node` label assigned at its first write, 0 until then + uint8_t open; // the descent is still open: a reference here is a cycle (write), a self-reference (read) + uint32_t node; // read: the node's arena offset; 0 for a definition the reader dropped + const TableTypeInfo * type; // read: the node's table; NULL for a dropped one +}; + +struct TableJsonGraphMap +{ + TableJsonGraphEntry * entries; + int64_t capacity; // a power of two, or zero while empty + int64_t count; + TableAllocator allocator; // the caller's pair (§6.5): the builder's on read, the one handed to ToJson on write +}; + +inline void TableJsonGraphMapInit( TableJsonGraphMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TableJsonGraphMapShutdown( TableJsonGraphMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TableJsonGraphMapInit( map, map.allocator ); +} + +inline int64_t TableJsonGraphMapSlot( const TableJsonGraphMap & map, uint64_t key ) +{ + uint64_t hash = key * 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != 0 && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TableJsonGraphEntry * TableJsonGraphMapFind( TableJsonGraphMap & map, uint64_t key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +inline bool TableJsonGraphMapGrow( TableJsonGraphMap & map ) +{ + TableJsonGraphMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 64; + grown.count = 0; + grown.entries = (TableJsonGraphEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TableJsonGraphEntry ) ); // zeroed, by the pair's contract + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == 0 ) { continue; } + grown.entries[ TableJsonGraphMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// the entry for a key, made if it was not there; `taken` says which. NULL is the +// allocator refusing, and the walk refuses with it. +inline TableJsonGraphEntry * TableJsonGraphMapReach( TableJsonGraphMap & map, uint64_t key, bool & taken ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TableJsonGraphMapGrow( map ) ) { return NULL; } + } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + taken = entry->key != key; + if ( taken ) + { + entry->key = key; + map.count++; + } + return entry; +} + +// ---- reading: into a builder ---- + +struct TableJsonGraphIn +{ + TableWorker * worker; // where every node comes from + TableJsonGraphMap labels; // a label -> the node it defined +}; + +// `&node`'s value, the LABEL: a positive integer spelled as one — digits, no sign, no +// fraction, no exponent, no leading zero (§16.7). Anything else is malformed. +inline bool TableJsonScanLabel( TableJsonIn & in, uint64_t & label ) +{ + TableJsonSpace( in ); + if ( in.pos >= in.size || in.text[in.pos] < '1' || in.text[in.pos] > '9' ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + uint64_t value = 0; + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) + { + uint64_t digit = (uint64_t) ( in.text[in.pos] - '0' ); + if ( value > ( UINT64_MAX - digit ) / 10 ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + value = value * 10 + digit; + in.pos++; + } + label = value; + return true; +} + +// A BYTE BUFFER's text (docs/SPEC-TABLES.md §2.5, §16.2): a string. For a +// *string the string's bytes become the blob; for a *bytes the string is base64 +// and its decoded bytes do. The blob is allocated at EXACTLY the decoded +// length — the string is scanned once without keeping it to learn the length, +// and once into the node — so a blob of any size reads with no window and no +// bound to clamp against. A *bytes body that is not base64 is the wrong shape +// for the kind: the reference stays null and the event is counted. +inline bool TableJsonReadBlob( TableJsonIn & in, void * slot, const TableFieldInfo * f ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + TableRef * ref = (TableRef *) slot; + ref->value = 0; + if ( strcmp( f->type_name, "string" ) == 0 ) + { + const int64_t mark = in.pos; + int32_t length = 0; + if ( !TableJsonScanString( in, NULL, 0, &length ) ) { return false; } + in.pos = mark; + char * data = TableStringEmplace( *graph->worker, *ref, NULL, (int64_t) length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int32_t placed = 0; + return TableJsonScanString( in, data, length, &placed ); + } + // base64: the alphabet characters decide the length, six bits apiece + const char * alphabet = TableJsonBase64Alphabet(); + const int64_t mark = in.pos + 1; + int64_t symbols = 0; + bool malformed = false; + in.pos++; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + if ( c == 0 || strchr( alphabet, c ) == NULL ) { malformed = true; continue; } + symbols++; + } + if ( malformed ) + { + in.report->kind_mismatch++; + return true; + } + const int64_t length = ( symbols * 6 ) / 8; + uint8_t * data = TableBytesEmplace( *graph->worker, *ref, length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int64_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + for ( int64_t at = mark; ; at++ ) + { + char c = in.text[at]; + if ( c == '"' ) { break; } + const char * symbol = c != '=' ? strchr( alphabet, c ) : NULL; + if ( symbol == NULL ) { continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( symbol - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < length ) { data[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); } + } + } + return true; +} + +// A pointer's object. Its FIRST key decides what it is: `&node` naming a label not +// yet defined, with fields after it, is a DEFINITION; `&node` naming one already +// defined, alone, is a REFERENCE; any other key is a node named once, its +// object in place. The node comes from the +// builder's arena, and the slot holds its arena offset (§6.3). A pointer whose +// target is a BYTE BUFFER — no table — takes a string instead (§2.5). +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( f->table == NULL ) { return TableJsonReadBlob( in, slot, f ); } + // the pointee nests one level down, exactly as a by-value table does, and + // takes the same cap: a chain nests as deep as it is long (§16.7) + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + char c = TableJsonPeek( in ); + if ( c == '}' ) + { + // an empty object: a node at its defaults, named once + in.pos++; + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + return true; + } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( strcmp( key, "&node" ) != 0 ) + { + // a node named once: the pointee's object in place, and this key is + // its first field — unless it is the reserved prefix under a spelling + // this form does not have, which ReadTableKeys refuses + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return TableJsonReadTableKeys( in, node, f->table, depth + 1, key ); + } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->labels, label, taken ); + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + // ONE SPELLING, and what follows the label says which half it is: fields + // after a label the text has not defined DEFINE it, and a label alone that + // the text has defined REFERS to it. The other two are malformed — a label + // alone that the text never defined, which would otherwise read as a default + // node under a silent report, and a field after a label already defined, + // which would be a second definition. That is what keeps a typo loud. + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + bool bare = c == '}'; + if ( bare == taken ) { in.report->malformed = true; in.bad = true; return false; } + if ( bare ) + { + // A REFERENCE. A label is defined when its object CLOSES, so a + // reference met inside its own definition — at any depth of by-value + // nesting — names a node whose descent is still open: the cycle the + // wire refuses (§3.1), refused here where it is written. A definition + // the reader dropped names no node, so the slot stays null with + // nothing more counted — the drop was counted where it happened. A + // node of another table than the slot declares is a kind mismatch, as + // on the wire. + in.pos++; + if ( entry->open != 0 ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + if ( entry->type == NULL ) + { + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + if ( entry->type != f->table ) + { + memcpy( slot, &ref, sizeof( ref ) ); + in.report->kind_mismatch++; + return true; + } + ref.value = (int64_t) entry->node; + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + // A DEFINITION: the node is allocated, the label is its, and the keys after + // `&node` are its fields. The entry is OPEN until the object closes, so a + // reference to the label from inside the node's own fields is refused as + // the cycle it is; the node and its table are filled in at the close. + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + entry->open = 1; + if ( !TableJsonReadTableKeys( in, node, f->table, depth + 1, NULL ) ) { return false; } + entry = TableJsonGraphMapFind( graph->labels, label ); // the map may have grown under the descent + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + memcpy( &ref, slot, sizeof( ref ) ); + entry->node = (uint32_t) ref.value; + entry->type = f->table; + entry->open = 0; + return true; +} + +// An `&`-prefixed key opening an object the walk is SKIPPING — a value past an +// array's bound, an unknown key's value, a value of the wrong shape. A +// definition in there still takes its label, so the numbering survives whatever +// the storage could not hold (§16.7): the label is registered with no node, and a +// reference to it reads null. Any other prefixed key is the reserved prefix +// out of place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL || strcmp( key, "&node" ) != 0 ) { in.report->malformed = true; in.bad = true; return false; } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + if ( TableJsonGraphMapReach( graph->labels, label, taken ) == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return true; // a fresh entry is node 0, type NULL: a definition with no node +} + +// ---- writing: from a region's const root ---- + +struct TableJsonGraphOut +{ + TableJsonGraphMap nodes; // a node's address -> how many slots name it, and its `&node` once assigned + bool counting; // PASS ONE: count the references, refuse a cycle, emit nothing + int64_t next_label; +}; + +// The node a slot names: null as `null`, a node named once as its object in +// place, and a node named more than once under the construct. Which of the +// last two it is was learned in pass one; pass two spells it. +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphOut * graph = (TableJsonGraphOut *) out.graph; + if ( graph == NULL ) { return false; } + const void * node = f->resolve( slot ); + if ( node == NULL ) + { + out.raw( "null", 4 ); + return true; + } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->nodes, (uint64_t) (uintptr_t) node, taken ); + if ( entry == NULL ) { return false; } + if ( f->table == NULL ) + { + // A BYTE BUFFER (§2.5, §16.7): its text is a string, which has no + // first key to carry `&node`, so a blob named from more than one + // slot has no spelling this form can carry and the graph is refused — + // as a shared node with nothing to write is. A blob named once is its + // bytes in place: base64 for a *bytes, the string itself for a *string. + if ( graph->counting ) { entry->count++; return true; } + if ( entry->count > 1 ) { return false; } + const TableBlob * blob = (const TableBlob *) node; + if ( blob->length > (uint32_t) 0x7fffffff ) { return false; } + if ( strcmp( f->type_name, "string" ) == 0 ) { TableJsonWriteString( out, (const char *) ( blob + 1 ), (int32_t) blob->length ); } + else { TableJsonWriteBase64( out, (const uint8_t *) ( blob + 1 ), (int32_t) blob->length ); } + return true; + } + if ( graph->counting ) + { + // PASS ONE: one visit per node, every slot that names it counted, and + // a reference to a node whose descent is still open is a cycle — + // refused here as the wire refuses it (§3.1) + entry->count++; + if ( !taken ) { return entry->open == 0; } + entry->open = 1; + if ( !TableJsonWriteValue( out, node, f->table, depth ) ) { return false; } + entry = TableJsonGraphMapFind( graph->nodes, (uint64_t) (uintptr_t) node ); // the map may have grown under the descent + if ( entry == NULL ) { return false; } + entry->open = 0; + return true; + } + // PASS TWO: a node named once is its object in place; a node named more + // than once is DEFINED at its first occurrence — `&node` first, then its + // fields — and REFERENCED by `&node` alone after that, spelled the same way at + // every site. Labels run from 1 in first-write order and are the text's own, + // so a stray number in a hand-edited text is most often one never defined. + if ( entry->count <= 1 ) + { + return TableJsonWriteValue( out, node, f->table, depth ); + } + if ( depth > kTableJsonMaxDepth ) { return false; } + if ( entry->label != 0 ) + { + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + out.line( depth ); + out.put( '}' ); + return true; + } + entry->label = ++graph->next_label; + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + bool any = true; + int64_t before = out.offset; + if ( !TableJsonWriteFields( out, node, f->table, depth, any ) ) { return false; } + // a definition carries at least one field, because a label alone is a + // reference: a shared node with nothing to write has no definition this + // form can spell, and the writer refuses it as it refuses any value it + // cannot spell (§16.3) + if ( out.offset == before ) { return false; } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- the two entry points a pointered table's wrappers name ---- + +// The text into the builder's root. Every node the text names is allocated in +// the builder's arena through the field's own Emplace; the label map is the +// walk's, released before this returns. The root itself takes no label — nothing +// may name it (§16.7) — so an `&node` at the root is refused like any other key +// of the prefix. +inline bool TableJsonReadGraph( TableWorker & worker, void * root, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + if ( worker.arena == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } + TableJsonGraphIn graph; + graph.worker = &worker; + TableJsonGraphMapInit( graph.labels, worker.arena->allocator ); + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = &graph; + info->reset( root ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, root, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + TableJsonGraphMapShutdown( graph.labels ); + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +// The text of a region's const root: measured when the buffer is NULL, written +// when it is not, over one code path. Two passes over one walk — the first +// counts how many slots name each node and refuses a cycle, the second writes +// — so a node's first occurrence knows whether it will be named again. The +// ROOT's entry is open for the whole first pass, so a reference back at it is +// the cycle it is (§3.1), and it takes no label. +inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * info, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + if ( root == NULL ) { return -1; } + TableJsonGraphOut graph; + TableJsonGraphMapInit( graph.nodes, allocator ); + graph.counting = true; + graph.next_label = 0; + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph.nodes, (uint64_t) (uintptr_t) root, taken ); + if ( entry == NULL ) { TableJsonGraphMapShutdown( graph.nodes ); return -1; } + entry->open = 1; + TableJsonOut count; + count.buffer = NULL; + count.capacity = 0; + count.offset = 0; + count.overflow = false; + count.graph = &graph; + bool ok = TableJsonWriteValue( count, root, info, 0 ); + graph.counting = false; + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = &graph; + if ( ok ) { ok = TableJsonWriteValue( out, root, info, 0 ); } + TableJsonGraphMapShutdown( graph.nodes ); + if ( !ok ) { return -1; } + out.put( '\n' ); // the canonical text ends with exactly one newline (§16.1) + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json graph walk: end ---- + +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + +// ---- json map walk: begin ---- + +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} + +// the entry's two rows: fields[0] IS the key and fields[1] IS the value, which +// is what makes a user's own table of pairs the same bytes (§2.8) +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } + +inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } +inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } + +// AN INTEGER KEY IS THE INTEGER'S DECIMAL SPELLING, QUOTED, because a JSON +// object's keys are strings. Written digit by digit so no locale can move it. +inline void TableJsonWriteMapIntegerKey( TableJsonOut & out, const void * storage, const TableFieldInfo * key ) +{ + uint64_t magnitude = 0; + bool negative = false; + if ( TableJsonMapKeySigned( key ) ) + { + int64_t value = 0; + switch ( key->kind ) + { + case 2: value = (int64_t) *(const int8_t *) storage; break; + case 3: value = (int64_t) *(const int16_t *) storage; break; + case 4: value = (int64_t) *(const int32_t *) storage; break; + default: value = *(const int64_t *) storage; break; + } + negative = value < 0; + magnitude = negative ? ( ~(uint64_t) value ) + 1 : (uint64_t) value; + } + else + { + switch ( key->kind ) + { + case 6: magnitude = (uint64_t) *(const uint8_t *) storage; break; + case 7: magnitude = (uint64_t) *(const uint16_t *) storage; break; + case 8: magnitude = (uint64_t) *(const uint32_t *) storage; break; + default: magnitude = *(const uint64_t *) storage; break; + } + } + char digits[24]; + int32_t at = (int32_t) sizeof( digits ); + do { digits[--at] = (char) ( '0' + ( magnitude % 10 ) ); magnitude /= 10; } while ( magnitude != 0 ); + if ( negative ) { digits[--at] = '-'; } + TableJsonWriteString( out, digits + at, (int32_t) sizeof( digits ) - at ); +} + +inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const TableFieldInfo * key ) +{ + const uint8_t * storage = (const uint8_t *) entry + key->offset; + if ( TableJsonMapKeyIsString( key ) ) + { + // A STRING KEY IS THE STRING (§2.8): every JSON key of a map object is + // a KEY OF THE MAP and none is a field key, so the `&` prefix §16.7 + // reserves for field keys is ordinary data here. + TableJsonWriteString( out, (const char *) storage, *(const int32_t *) ( (const uint8_t *) entry + key->count_offset ) ); + return; + } + TableJsonWriteMapIntegerKey( out, (const void *) storage, key ); +} + +// ToJson WRITES ENTRIES IN ASCENDING KEY ORDER, so unpack then pack is +// byte-stable and a diff of two texts is a diff of two maps (§2.8, §17.2). +// A region holds them in that order already, so this is the array in place. +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "{}", 2 ); return true; } + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); + out.put( '{' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); + TableJsonWriteMapKey( out, entry, key ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// AN INTEGER KEY IS READ BY §16.2's INTEGER RULE AND BY NOTHING ELSE, so +// "2.0" and "1e3" are the integers 2 and 1000 and "-0" is zero. The token is +// walked as a JSON number over its own bytes; a token that rule calls +// malformed makes the KEY malformed, and a genuinely fractional value, or one +// outside the key kind's range, is kind_mismatch for that entry. +inline bool TableJsonMapKeyValue( const char * token, int32_t length, const TableFieldInfo * key, + int64_t & value, bool & fits ) +{ + fits = false; + TableReport scratch; + TableJsonIn probe = { token, (int64_t) length, 0, &scratch, false, NULL }; + bool integral = false; + if ( !TableJsonWalkNumber( probe, &integral ) ) { return false; } + if ( probe.pos != (int64_t) length ) { return false; } // trailing bytes: not a number + if ( !integral ) + { + const double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) { return true; } // a value no key kind holds + const double whole = d < 0 ? -d : d; + if ( whole != (double) (int64_t) whole ) { return true; } // genuinely fractional + } + bool saturated = false; + const bool is_signed = TableJsonMapKeySigned( key ); + value = integral ? TableJsonTokenInteger( token, length, is_signed, &saturated ) + : (int64_t) TableJsonTokenDouble( token, length, false ); + if ( saturated ) { return true; } // outside every width: kind_mismatch, never clamped + switch ( key->kind ) + { + case 2: fits = value >= -128 && value <= 127; break; + case 3: fits = value >= -32768 && value <= 32767; break; + case 4: fits = value >= -2147483647 - 1 && value <= 2147483647; break; + case 5: fits = true; break; + case 6: fits = value >= 0 && value <= 255; break; + case 7: fits = value >= 0 && value <= 65535; break; + case 8: fits = value >= 0 && (uint64_t) value <= 4294967295ull; break; + default: fits = integral; break; // uint64: the token's own magnitude + } + return true; +} + +// FromJson READS KEYS IN WHATEVER ORDER THE TEXT GIVES THEM. A repeated key is +// last-wins and counted duplicate, the object rule (§16.2) applied inside the +// map. An empty object is an empty map, and null is kind_mismatch. +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const char shape = TableJsonShape( value ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char token[kTableJsonMaxKey]; + int32_t token_length = 0; + if ( !TableJsonScanString( in, token, kTableJsonMaxKey - 1, &token_length ) ) { return false; } + token[token_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t key_value = 0; + bool place = true; + if ( !TableJsonMapKeyIsString( key ) ) + { + bool fits = false; + if ( !TableJsonMapKeyValue( token, token_length, key, key_value, fits ) ) + { + // A MALFORMED KEY STOPS THE READ where §16.1's rule stops it, + // with the instance holding what was placed before the stop. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( !fits ) { in.report->kind_mismatch++; place = false; } + } + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; + if ( place && entry == NULL ) + { + // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the + // wire's rule, because a clamped key is a merged entry (§2.8). + in.report->clamped++; + } + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) + { + in.report->duplicate++; // last-wins, the object rule inside the map + } + const char got = TableJsonValueShape( in ); + if ( entry == NULL ) + { + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( value->kind == 17 && !value->is_array ) + { + // A POINTER VALUE IS SHARED EXACTLY AS A POINTER FIELD IS (§2.8): + // null is a null slot, an object is the pointee in place or an + // &node reference to one (§16.7), anything else is the wrong shape — + // the same three the field-key loop gives a pointer field, because + // an entry's value IS a field line. + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) entry + value->offset, value->elem_size, 0 ); + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) entry + value->offset, value, depth + 1 ) ) + { + return false; + } + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, entry, value, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json map walk: end ---- + +// ---- json list walk: begin ---- + +// an unbounded array is the out-of-line array that is not a map (§8.1) +inline bool TableJsonIsList( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && !TableJsonIsMap( f ); +} + +// ToJson WRITES THE ELEMENTS IN INDEX ORDER, which is the only order there is, +// so unpack then pack is byte-stable without a rule of its own (§2.9, §17.2). +// A region holds the array in place, so this steps it at the descriptor's pitch. +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + const uint8_t * elements = TableJsonExtentElements( slot ); + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const uint8_t * element = elements + (int64_t) i * f->elem_size; + if ( f->kind == 17 ) + { + // a []*T's elements take the pointer row (§16.7): the pointee's + // object in place, null, or `&node` for a shared one + if ( !TableJsonWritePointer( out, element, f, depth + 1 ) ) { return false; } + } + else if ( !TableJsonWriteScalar( out, element, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; +} + +// FromJson READS EVERY ELEMENT THE TEXT CARRIES, appending each through the +// descriptor's place resolver: `[]` is an empty list, and null is +// kind_mismatch, the array row's own rule (§16.2). LAST WINS holds for a +// repeated key: the list goes back to EMPTY before this occurrence's elements +// land, the builder's storage being reclaimed at reset (§2.9). +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) slot, 8, 0 ); + TableJsonSetRaw( (uint8_t *) slot + 8, 4, 0 ); + const char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + void * element = f->place( *graph->worker, slot, NULL, 0, 0 ); + if ( element == NULL ) + { + // NOT ADDED: the arena could not carve another segment, or the + // count met the int32 cap. The text cannot be placed whole, and + // the read stops where §16.1's rule stops it. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( f->kind == 17 ) + { + // an element of a []*T (§2.9): null is a null slot, an object is the + // pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) element, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, element, f, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + // the wrong shape for the element kind: the slot keeps its + // defaults and the event counts, the array row's rule (§16.2) + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, element, f, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json list walk: end ---- + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +bool SampleFromJson( Sample & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, SampleTableType(), text, bytes, report ); +} + +int64_t SampleToJsonMeasure( const Sample & value ) +{ + return TableJsonWrite( &value, SampleTableType(), NULL, 0 ); +} + +int64_t SampleToJson( const Sample & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, SampleTableType(), buffer, capacity ); +} + +bool RowFromJson( RowBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Row * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, RowTableType(), text, bytes, report ); +} + +int64_t RowToJsonMeasure( const Row * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, RowTableType(), NULL, 0, allocator ); +} + +int64_t RowToJson( const Row * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, RowTableType(), buffer, capacity, allocator ); +} + +bool SheetFromJson( SheetBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Sheet * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, SheetTableType(), text, bytes, report ); +} + +int64_t SheetToJsonMeasure( const Sheet * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, SheetTableType(), NULL, 0, allocator ); +} + +int64_t SheetToJson( const Sheet * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, SheetTableType(), buffer, capacity, allocator ); +} + +bool ItemFromJson( Item & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, ItemTableType(), text, bytes, report ); +} + +int64_t ItemToJsonMeasure( const Item & value ) +{ + return TableJsonWrite( &value, ItemTableType(), NULL, 0 ); +} + +int64_t ItemToJson( const Item & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, ItemTableType(), buffer, capacity ); +} + +bool SquadFromJson( SquadBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Squad * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, SquadTableType(), text, bytes, report ); +} + +int64_t SquadToJsonMeasure( const Squad * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, SquadTableType(), NULL, 0, allocator ); +} + +int64_t SquadToJson( const Squad * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, SquadTableType(), buffer, capacity, allocator ); +} + +bool ArmyFromJson( ArmyBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Army * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, ArmyTableType(), text, bytes, report ); +} + +int64_t ArmyToJsonMeasure( const Army * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, ArmyTableType(), NULL, 0, allocator ); +} + +int64_t ArmyToJson( const Army * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, ArmyTableType(), buffer, capacity, allocator ); +} + +bool DeckFromJson( DeckBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Deck * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, DeckTableType(), text, bytes, report ); +} + +int64_t DeckToJsonMeasure( const Deck * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, DeckTableType(), NULL, 0, allocator ); +} + +int64_t DeckToJson( const Deck * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, DeckTableType(), buffer, capacity, allocator ); +} + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/HoldersTable.h b/testdata/golden/tables/lists/HoldersTable.h new file mode 100644 index 000000000..e3bf2cfa0 --- /dev/null +++ b/testdata/golden/tables/lists/HoldersTable.h @@ -0,0 +1,11521 @@ +// Code generated by the schema compiler from Holders.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — protocol id 0xa5fbe602c119cdd9 (packets only: tables version by field id, not by protocol id) +// The TABLE wire (evolution-tolerant, docs/SPEC-TABLES.md): no serialize +// dependency — includable from any TU. + +#pragma once + +#include +#include // the prefill's scalar-array fills +#include // offsetof, for the reflection descriptors + +// ---- the hooks (docs/USAGE.md, "the C++ table runtime's hooks") ---- +// +// schema_assert — the runtime's own assert, and the refusal a debugger reads. +// NDEBUG removes it, exactly as it removes assert. A caller who already routes +// serialize's asserts writes `#define schema_assert serialize_assert` before +// including this header and both halves land in one handler. +#ifndef schema_assert +#include +#define schema_assert assert +#endif // #ifndef schema_assert + +// schema_fatal — what stands after the assert on a path that cannot continue. +// NDEBUG does not remove it. Supply it and is never included. +#ifndef schema_fatal +#include // abort +#define schema_fatal abort +#endif // #ifndef schema_fatal + +// schema_allocate / schema_release — what "no allocator handed in" means for +// this program. schema_allocate hands back ZEROED bytes and NULL on failure: +// an arena segment is copied whole, padding included, so anything left +// uninitialized here would reach a packed region. Supply both and +// is never included; hand a TableAllocator to a builder to route one +// structure's allocations somewhere else again. +#ifndef schema_allocate +#include // calloc, free +#define schema_allocate( bytes ) calloc( (size_t) 1, (size_t) ( bytes ) ) +#define schema_release( pointer ) free( pointer ) +#endif // #ifndef schema_allocate +#include // a node's lifetime starts in arena storage (placement new) +#include // one atomic per slab: the arena is lock-free by ownership + +#include "Holders.h" + +#ifndef LISTDEMO_SCHEMA_TABLE_PRIMITIVES +#define LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +// THE CODEC DOES NOT DEPEND ON THE COMPILER'S INLINING BUDGET. A table of a +// realistic field count emits one large body per type, and the cursor a body +// writes through lives in the caller's `TableWriter`: across a call boundary +// that cursor round-trips through memory, and a `uint8_t *` store may alias the +// writer itself, so every put reloads it. When a budget runs out mid-body the +// codec silently degrades to that shape. Forcing the primitives and the +// fixed-class bodies inline is what keeps the cursor in registers and lets +// adjacent constant framing bytes merge into one store. +#if defined( _MSC_VER ) +#define LISTDEMO_TABLE_INLINE __forceinline +#elif defined( __GNUC__ ) || defined( __clang__ ) +#define LISTDEMO_TABLE_INLINE inline __attribute__(( always_inline )) +#else +#define LISTDEMO_TABLE_INLINE inline +#endif + +namespace listdemo { + +// WHY A READ WAS REFUSED, by name (docs/SPEC-TABLES.md §3.3, §11). A REFUSAL +// is not one of §4's events: nothing is decoded, no counter moves and no +// damage is reported, so five zero counters and a false flag are what a clean +// read prints too and only the verdict tells them apart. The reason says which +// refusal it was. +// +// This is the MESSAGE PATH's vocabulary and not the cooked form's (§7.4): a +// caller meeting one of these has been refused a MESSAGE on a connection, +// which is a different recovery with a different owner than a file a header +// match turned down. +enum TableMessageReason +{ + newer_form, // a FORM BYTE this reader does not carry (§3) + no_vocabulary, // no table for this connection: the message arrived before the announcement, or after a refused one + second_announcement, // a second announcement on a connection: it sets nothing, amends nothing, and the connection closes + vocabulary_too_large, // an announcement above the receiver's declared bound, refused before an entry is touched + message_form_as_file // a form 2 wire where a FILE was expected: its table is somewhere else +}; + +// The table-wire read report — the permissive contract's ledger. Silence +// (all zero) means the data matched this reader's schema exactly. +struct TableReport +{ + int32_t unknown = 0; // unknown field ids skipped (newer data) + int32_t kind_mismatch = 0; // known id, changed type — skipped, never misdecoded + int32_t clamped = 0; // out-of-range values clamped to declared bounds + // a key the TEXT form saw twice: last wins, and the repeat is counted + // (docs/SPEC-TABLES.md §16.2). The wire never raises it — a body carrying an + // id twice is legal input whose last occurrence wins, silently (§3). + int32_t duplicate = 0; + bool malformed = false; // framing damage; decode stopped, partial result kept + // THE REFUSAL VERDICT, which is not one of §4's events and moves no counter + // (docs/SPEC-TABLES.md §3): a FORM BYTE this reader does not carry. Five + // zero counters and a false flag are what a clean read prints too, so the + // verdict is what tells the two apart. + bool refused = false; + // WHICH refusal, and it is read only when refused is set: a read that + // was not refused has no reason, and this member is the one the caller + // must not look at then (docs/SPEC-TABLES.md §3.3). + TableMessageReason reason = newer_form; +}; + + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; +// ---- reflection (tables only, docs/SPEC-TABLES.md) ---- +// +// Static field descriptors for every type in the table closure: name, wire +// id/kind, storage offset, bounds, ranges, enum names and branch guards — +// enough to walk, print, diff, edit or bind any table value at runtime with +// no RTTI and no schema files. TableType() returns X's descriptor. + +struct TableTypeInfo; + +// One arm of a union field: where its payload sits inside the union's storage +// and what its payload looks like. The arm's NAME and its table-wire id come +// from the field's enum_name/variant_id functions at the same tag, so nothing +// is spelled twice (docs/SPEC-TABLES.md §8). +struct TableFieldInfo; + +struct TableUnionArmInfo +{ + uint32_t offset; // offsetof the arm's payload within the union storage + const TableTypeInfo * table; // the arm payload's descriptor, or NULL + // AN ARM IS A FIELD LINE (docs/SPEC-TABLES.md §2.6): an arm that names no + // declared type or table carries the FIELD descriptor a field of that + // type would carry instead — offsets taken within the union storage — so + // a generic walk meets an arm's kind, width, bounds and companions where + // it meets a field's. Exactly one of the two is non-NULL on a set arm. + const TableFieldInfo * field; + uint32_t size; // the arm's whole storage, which selection zero-establishes +}; + +// A union field's shape: the tag, and the arms indexed by it. Arms run +// [0, enum_max]; index 0 is the EMPTY arm and carries no payload. +struct TableUnionInfo +{ + uint32_t tag_offset; // offsetof the tag within the union storage + uint32_t tag_size; // sizeof the tag + const TableUnionArmInfo * arms; +}; + +// The exact raw range of a wide-kind field (docs/SPEC-TABLES.md §8.2): two 128-bit +// values as 64-bit lanes, low lane first, two's complement for the signed kinds. +struct TableWideRange +{ + uint64_t lo[2]; + uint64_t hi[2]; +}; + +// the arena's allocation front, defined with the variable-length runtime +// below; a descriptor names it only through a pointer parameter. +struct TableWorker; + +struct TableFieldInfo +{ + const char * name; // schema field name, e.g. "health" + const char * json; // the TEXT form's key: the json = "key" attribute, else name (§16.3) + const char * type_name; // schema type name, e.g. "float32", "Grade" + uint64_t id; // table-wire field id: fnv1a64 of the name, of the was alias after a rename (§5) + uint8_t kind; // table-wire kind; for arrays/strings/bytes, the ELEMENT kind + bool is_array; // fixed or counted array (bytes included) + bool is_pointer; // a *T pointer field: storage is an 8-byte TableRef; the target is a table + // THE TWO THE TEXT FORM NEEDS (docs/SPEC-TABLES.md §16.7), and they + // are here for the same reason is_pointer is: the walk is ONE walk + // over descriptors and cannot spell a target's own At or + // Emplace. `resolve` reads a slot in a REGION and answers the + // node it names, or NULL; `emplace` allocates one in a BUILDER's + // arena and points the slot at it. NULL on every field that is not + // a pointer, and emitted only in a unit that declares one. + const void * (*resolve)( const void * slot ); + void * (*emplace)( TableWorker & worker, void * slot ); + bool counted; // a _count/_length int32 companion exists (counted arrays, strings, bytes) + bool optional; // a ?T field: a _present bool companion decides whether it rides + int32_t array_bound; // array capacity / string max length; 0 for plain scalars + uint32_t offset; // offsetof the storage member + uint32_t elem_size; // sizeof the member (element size for arrays) + uint32_t count_offset; // offsetof the _count/_length companion, or 0xffffffff + uint32_t present_offset; // offsetof the _present companion, or 0xffffffff + const TableTypeInfo * table; // nested table's descriptor, or NULL + bool has_range; // a declared [min, max] (int or float) + double range_min; // NOTE: int64 ranges beyond 2^53 lose precision here + double range_max; + // the WIDE kinds (18-29, docs/SPEC-TABLES.md §3, §8.2): frac_bits is a fixed + // field's F — its storage holds units × 2^F — and wide is the declared + // range on that RAW scale, exact, as two 128-bit two's-complement values + // in 64-bit lanes (low lane first). NULL where the declaration bounds + // nothing (a bare uint128) and for every other kind; frac_bits is 0 for + // every kind that is not fixed-point. range_min/range_max still carry + // the declared bounds as doubles — whole units for a fixed field — for + // a walker that only shows them. + uint8_t frac_bits; + const TableWideRange * wide; + int64_t enum_max; // enums: highest valid value (None = 0 always valid); + // unions: the arm count (tag range [0, enum_max]); + // flags: the highest declared BIT INDEX; else -1 + // the vocabulary's names, indexed the same way enum_max bounds: an enum's + // value -> name, a union's tag -> arm name, a FLAGS field's bit index -> + // variant name. NULL for every other kind. + const char * (*enum_name)( uint64_t value ); + // the TABLE-WIRE id of one variant (docs/SPEC-TABLES.md §5): for an enum, the + // hash of the variant's name; for a union, the hash of the arm's name. + // 0 is the reserved id — an enum's None, a union's empty. NULL for every + // other kind — a FLAGS field's variants have no per-variant wire id (§4), + // so a NULL here beside a non-NULL enum_name is what says "flags". + // Walk [0, enum_max] to enumerate a vocabulary and its ids. + uint64_t (*variant_id)( uint64_t value ); + // an ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): the array has one slot per + // variant of key_type_name, indexed by the variant's value, and its slots + // ride under variant ids rather than positions. key_name and key_id are + // the key's vocabulary — walk [0, array_bound) to print slots by name. + // NULL on every other field. + const char * key_type_name; + const char * (*key_name)( uint64_t value ); + uint64_t (*key_id)( uint64_t value ); + // union fields: the tag and its arms, behind a function so the whole + // descriptor stays CONSTANT-INITIALISED (a captureless lambda converts to + // a function pointer at compile time; the arms themselves are a static + // inside it). NULL for every other kind. + const TableUnionInfo * (*arms)(); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded +}; + +struct TableTypeInfo +{ + const char * name; // schema type name + uint32_t size; // sizeof the storage struct + int32_t num_fields; + const TableFieldInfo * fields; + // put one instance back at its declared defaults, in place. A generic + // walker that fills a value has to be able to establish the defaults an + // absent field takes, and it holds no type to spell — this is the one + // thing the descriptors could not express without it. Placement-new + // value-init, exactly what the wire's read path does, and no temporary. + void (*reset)( void * storage ); + // the DERIVED mode (docs/SPEC-TABLES.md): false = fixed-size, a plain + // relocatable struct; true = variable-length, built through a Builder + // and read through a region root. Nobody declares it; the compiler + // works it out. + bool variable; +}; + +struct TableWriter +{ + uint8_t * buffer; + int64_t capacity; + int64_t offset = 0; + bool overflow = false; + + // the parameters do not repeat the member names: a parameter that hides a + // member is a warning the estate's compilers disagree about (gcc's + // -Wshadow and cl's C4458 refuse it, clang's -Wshadow does not), and this + // is a header a consumer compiles under its OWN flags + TableWriter( uint8_t * to_buffer, int64_t to_capacity ) : buffer( to_buffer ), capacity( to_capacity ) {} + + LISTDEMO_TABLE_INLINE void raw( const void * data, int64_t bytes ) + { + if ( offset + bytes > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) bytes ); + offset += bytes; + } + LISTDEMO_TABLE_INLINE void put8( uint8_t v ) { raw( &v, 1 ); } + LISTDEMO_TABLE_INLINE void put16( uint16_t v ) { uint8_t b[2] = { uint8_t( v ), uint8_t( v >> 8 ) }; raw( b, 2 ); } + LISTDEMO_TABLE_INLINE void put32( uint32_t v ) { uint8_t b[4] = { uint8_t( v ), uint8_t( v >> 8 ), uint8_t( v >> 16 ), uint8_t( v >> 24 ) }; raw( b, 4 ); } + LISTDEMO_TABLE_INLINE void put64( uint64_t v ) { put32( uint32_t( v ) ); put32( uint32_t( v >> 32 ) ); } + // a 128-bit value as two lanes, the low half first (docs/SPEC-TABLES.md §3) + LISTDEMO_TABLE_INLINE void put128( uint64_t lo, uint64_t hi ) { put64( lo ); put64( hi ); } + // EVERY LENGTH, COUNT, INDEX AND ID REFERENCE IS ONE CANONICAL UNSIGNED + // LEB128 (docs/SPEC-TABLES.md §3): seven value bits a byte, the lowest + // group first, the high bit set on every byte but the last. One value has + // one spelling, so two conforming writers agree byte for byte. + LISTDEMO_TABLE_INLINE void putleb( uint64_t v ) + { + while ( v >= 0x80 ) { put8( uint8_t( v ) | 0x80 ); v >>= 7; } + put8( uint8_t( v ) ); + } +}; + +// TableLebBytes is one value's spelling length, which a MEASURE needs before +// the bytes exist — the length of a body has to be known before it is written, +// because a length whose own width moves cannot be patched in place. +inline int64_t TableLebBytes( uint64_t v ) +{ + int64_t n = 1; + while ( v >= 0x80 ) { v >>= 7; n++; } + return n; +} + +// THE ID TABLE, WRITER SIDE (docs/SPEC-TABLES.md §3). It holds every id the +// body used, once each, in FIRST-USE order over the whole wire, and the body +// names them by position: reference k is the kth entry, counted from 1, and +// reference 0 names NO ID. +// +// Its capacity is a COMPILE-TIME fact of the unit — the distinct names its +// table closure can spell — so a save allocates nothing: the table is a local +// of Measure and of Save. The bucket chain makes ref constant time and makes +// truncate constant time too, which is what an ELIDED field needs: a field +// that turns out not to ride costs nothing in the id table either, so the walk +// interns its id, builds the payload that decides, and undoes the entry when +// nothing rides. +struct TableIds +{ + static const int32_t kCapacity = 60; + static const int32_t kBuckets = 128; + + uint64_t ids[ kCapacity ]; + int32_t chain[ kCapacity ]; + int32_t head[ kBuckets ]; + int32_t count; + bool overflow; + // THE MESSAGE FORM'S SLOTS (docs/SPEC-TABLES.md §3.3). A form 2 wire + // names ids through the CONNECTION's table, which is the unit's whole + // vocabulary in a compiler-settled order — so every reference is known at + // compile time and rides at the header as a literal beside the id. This + // flag is what selects it: false interns the id in first-use order and + // writes a trailer, true answers the slot and writes none, and the walk + // that decides is one walk. + bool vocabulary; + + TableIds() : count( 0 ), overflow( false ), vocabulary( false ) + { + for ( int32_t i = 0; i < kBuckets; i++ ) { head[i] = -1; } + } + + static LISTDEMO_TABLE_INLINE uint32_t bucket_of( uint64_t id ) + { + return uint32_t( ( id * 0x9E3779B97F4A7C15ull ) >> 57 ) & uint32_t( kBuckets - 1 ); + } + + // the reference an id takes: its message-form SLOT under the connection's + // table, or the file's own first-use entry + LISTDEMO_TABLE_INLINE uint64_t ref( uint64_t id, uint64_t slot ) + { + if ( vocabulary ) { return slot; } + return intern( id ); + } + + // the FILE form's half, appending the id on first use + uint64_t intern( uint64_t id ) + { + const uint32_t b = bucket_of( id ); + for ( int32_t i = head[b]; i >= 0; i = chain[i] ) + { + if ( ids[i] == id ) { return uint64_t( i ) + 1; } + } + if ( count >= kCapacity ) { overflow = true; return 1; } + ids[count] = id; chain[count] = head[b]; head[b] = count; count++; + return uint64_t( count ); + } + + // undo every entry appended since mark. An entry removed is the most + // recent one in its bucket, so it sits at that bucket's head. + void truncate( int32_t mark ) + { + // a SLOT costs no entry, so an elided field has nothing to undo + if ( vocabulary ) { return; } + while ( count > mark ) + { + count--; + head[ bucket_of( ids[count] ) ] = chain[count]; + } + } +}; + +// TableIdsBytes is the trailer's own size: the entries, each a fixed +// little-endian u64, and the ENTRY COUNT, the one fixed-width number on the +// wire (docs/SPEC-TABLES.md §3). +inline int64_t TableIdsBytes( const TableIds & ids ) { return int64_t( ids.count ) * 8 + 8; } + +// TableIdsWrite puts the trailer where the walk ended: a writer never patches, +// because first-use order is known only when the walk ends. +inline void TableIdsWrite( TableWriter & w, const TableIds & ids ) +{ + for ( int32_t i = 0; i < ids.count; i++ ) { w.put64( ids.ids[i] ); } + w.put64( uint64_t( ids.count ) ); +} + +// THE ID TABLE, READER SIDE (docs/SPEC-TABLES.md §3). A reader locates it from +// the END of the wire and resolves it ONCE, at open: the entries are eight +// bytes each and a body names them by position, so every field dispatches +// through an index rather than through a search over hashes. +struct TableIdTable +{ + const uint8_t * entries = NULL; + int64_t count = 0; + + // the id a reference names. ref is 1-based and bounds-checked by the + // caller: a reference ABOVE the entry count is framing damage on the body + // that carries it, and 0 names no id at all. + uint64_t at( uint64_t ref ) const + { + const uint8_t * e = entries + ( ref - 1 ) * 8; + uint64_t lo = uint64_t( e[0] ) | uint64_t( e[1] ) << 8 | uint64_t( e[2] ) << 16 | uint64_t( e[3] ) << 24; + uint64_t hi = uint64_t( e[4] ) | uint64_t( e[5] ) << 8 | uint64_t( e[6] ) << 16 | uint64_t( e[7] ) << 24; + return lo | ( hi << 32 ); + } +}; + +struct TableReader +{ + const uint8_t * buffer; + int64_t size; + int64_t offset = 0; + TableReport * report; + const TableIdTable * ids = NULL; + // ONLY THE ROOT BODY CARRIES THE NODE TABLE (docs/SPEC-TABLES.md §3.1), so + // a body has to know which it is: the reserved id inside a NESTED body is + // malformed, because a second numbering cannot exist. Every reader made + // for a payload is nested; the two the wire surfaces make for a root say so. + bool nested = true; + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report ) + : buffer( from_buffer ), size( from_size ), report( to_report ) {} + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report, const TableIdTable * to_ids ) + : buffer( from_buffer ), size( from_size ), report( to_report ), ids( to_ids ) {} + + LISTDEMO_TABLE_INLINE bool has( int64_t bytes ) const { return offset + bytes <= size; } + // A LENGTH IS A 64-BIT NUMBER AND A BUFFER IS NOT (docs/SPEC-TABLES.md + // §3): every length, count and index on this wire has sixty-four bits of + // capability, so one past what remains must be compared UNSIGNED. Casting + // it to int64 first turns 0xFFFFFFFFFFFFFFFF into -1, and a negative + // length looks like room. + LISTDEMO_TABLE_INLINE bool room( uint64_t bytes ) const { return bytes <= (uint64_t) ( size - offset ); } + LISTDEMO_TABLE_INLINE uint8_t get8() { return buffer[offset++]; } + LISTDEMO_TABLE_INLINE uint16_t get16() { uint16_t v = uint16_t( buffer[offset] ) | uint16_t( buffer[offset+1] ) << 8; offset += 2; return v; } + LISTDEMO_TABLE_INLINE uint32_t get32() { uint32_t v = uint32_t( buffer[offset] ) | uint32_t( buffer[offset+1] ) << 8 | uint32_t( buffer[offset+2] ) << 16 | uint32_t( buffer[offset+3] ) << 24; offset += 4; return v; } + LISTDEMO_TABLE_INLINE uint64_t get64() { uint64_t lo = get32(); uint64_t hi = get32(); return lo | ( hi << 32 ); } + LISTDEMO_TABLE_INLINE void get128( uint64_t & lo, uint64_t & hi ) { lo = get64(); hi = get64(); } + + // ONE CANONICAL UNSIGNED LEB128 (docs/SPEC-TABLES.md §3), and a + // non-minimal spelling is MALFORMED: 0x80 0x00 and 0x00 both spell zero, + // and only the second is legal input. An encoding past ten bytes, or a + // tenth byte with a bit above the 64th value bit, is malformed on the same + // rule. false = framing damage on the body carrying it. + bool getleb( uint64_t & value ) + { + // A NUMBER THIS READER REFUSES LEAVES THE CURSOR WHERE IT WAS. The + // caller's next question is often "did this body end exactly at its + // L", and a rejected number that had moved the cursor would answer + // that question with the damage already stepped over. + const int64_t at = offset; + value = 0; + uint32_t shift = 0; + for ( int32_t i = 0; i < 10; i++ ) + { + if ( !has( 1 ) ) { offset = at; return false; } + const uint8_t b = get8(); + if ( i == 9 && b > 1 ) { offset = at; return false; } + value |= uint64_t( b & 0x7F ) << shift; + if ( ( b & 0x80 ) == 0 ) + { + if ( i > 0 && b == 0 ) { offset = at; return false; } // a redundant continuation + return true; + } + shift += 7; + } + offset = at; + return false; + } + + // resolve one id reference against the file's table. false = a reference + // ABOVE the entry count, or a 0 where an id is required, both of which + // are framing damage on the body that carries it. + bool getid( uint64_t & id ) + { + uint64_t ref = 0; + if ( !getleb( ref ) ) { return false; } + if ( ref == 0 || ids == NULL || ref > (uint64_t) ids->count ) { return false; } + id = ids->at( ref ); + return true; + } + + // skip one payload by kind; false = framing damage. FOUR RULES COVER THE + // SET (docs/SPEC-TABLES.md §3), and a kind outside it is not skippable — + // which is why the set is closed and why kind 31 exists. + bool skip( uint8_t kind ) + { + switch ( kind ) + { + // the fixed-width kinds, each by its width: 18-29 are the 128-bit integers and + // the fixed-point family at every storage width (docs/SPEC-TABLES.md §3) + case 1: case 2: case 6: case 20: case 25: return has( 1 ) ? ( offset += 1, true ) : false; + case 3: case 7: case 21: case 26: return has( 2 ) ? ( offset += 2, true ) : false; + case 4: case 8: case 10: case 22: case 27: return has( 4 ) ? ( offset += 4, true ) : false; + case 5: case 9: case 11: case 23: case 28: return has( 8 ) ? ( offset += 8, true ) : false; + case 18: case 19: case 24: case 29: return has( 16 ) ? ( offset += 16, true ) : false; + case 17: case 30: // a NODE INDEX (§3.1) and an ENUM's variant reference: one LEB128 and stop + { + uint64_t ignored = 0; + return getleb( ignored ); + } + case 12: case 13: case 14: case 16: case 31: case 32: // 31 is the ESCAPE, 32 the payload-free kind + { + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + case 15: // union: the arm id reference, then its kind, its L and its payload (reference 0 = empty) + { + uint64_t arm = 0; + if ( !getleb( arm ) ) return false; + if ( arm == 0 ) return true; + if ( !has( 1 ) ) return false; + offset += 1; // the arm's kind byte + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + } + return false; + } +}; + +// The RESERVED node-table id, the one id the language holds back +// (docs/SPEC-TABLES.md §3.1, §5). It rides in every unit, pointered or not, +// because every body has to know that a NESTED body claiming one is damaged. +static const uint64_t kTableNodeTableFieldId = 0xFFFFFFFFFFFFFFFFull; + +// TableWireForm is the FORM BYTE, and it is the whole header +// (docs/SPEC-TABLES.md §3). A reader that meets a byte it does not know +// refuses the wire by name and never reports damage. +const uint8_t kTableWireForm = 1; + +// TableOpen reads the form byte and the trailer, in that order, and hands back +// the ROOT BODY. It answers one of three verdicts, because five zero counters +// and a false flag are what a clean read prints too: +// +// TableOpenOk the form is known and the table read whole +// TableOpenRefused a FORM BYTE this reader does not carry: nothing is +// decoded, nothing is counted, and no damage is reported +// TableOpenDamaged a table that cannot be read whole — fewer than eight +// bytes, a count whose entries run past the front of the +// file, a count that leaves no room for the form byte, or +// ONE ID IN TWO ENTRIES. The whole wire is malformed, +// nothing is decoded, and one event is counted. +// TableOpenBodyStopped the form and the table were good and the ROOT BODY +// could not be walked to its own terminator. What it +// decoded before that is kept, as everywhere on this wire. +enum TableOpenVerdict { TableOpenOk, TableOpenRefused, TableOpenDamaged, TableOpenBodyStopped }; + +inline TableOpenVerdict TableOpen( const uint8_t * buffer, int64_t bytes, TableIdTable & table, int64_t & body_bytes ) +{ + if ( bytes < 1 ) { return TableOpenDamaged; } + if ( buffer[0] != kTableWireForm ) { return TableOpenRefused; } + if ( bytes < 9 ) { return TableOpenDamaged; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + uint64_t count = lo | ( hi << 32 ); + if ( count > (uint64_t) ( bytes / 8 ) ) { return TableOpenDamaged; } + const int64_t span = (int64_t) count * 8 + 8; + if ( span + 1 > bytes ) { return TableOpenDamaged; } + table.entries = buffer + bytes - span; + table.count = (int64_t) count; + // THE ENTRIES ARE DISTINCT: a table that carries one id twice is malformed + // for the whole wire, because no wire this schema writes carries a repeat + // and it would leave one more shape of table for a hostile writer to aim + // at (docs/SPEC-TABLES.md §3). + for ( int64_t i = 1; i < table.count; i++ ) + { + const uint64_t id = table.at( uint64_t( i ) + 1 ); + for ( int64_t j = 0; j < i; j++ ) + { + if ( table.at( uint64_t( j ) + 1 ) == id ) { return TableOpenDamaged; } + } + } + body_bytes = bytes - span - 1; + return TableOpenOk; +} + +// TableBodyExtent walks a body's framing to the zero reference that ends it, +// so a reader can tell a body that ENDED EARLY — leaving bytes no field claims +// — from one that is merely damaged. ANY BYTE BETWEEN THE ROOT'S TERMINATOR +// AND THE TABLE'S FIRST ENTRY IS MALFORMED, because no field claims it and the +// two ends of the file have met (docs/SPEC-TABLES.md §3). +inline bool TableBodyEndsEarly( const uint8_t * body, int64_t bytes, const TableIdTable & table ) +{ + TableReport ignored; + TableReader r( body, bytes, &ignored, &table ); + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { return false; } + if ( ref == 0 ) { return r.offset != bytes; } + if ( ref > (uint64_t) table.count ) { return false; } + if ( !r.has( 1 ) ) { return false; } + if ( !r.skip( r.get8() ) ) { return false; } + } +} + +// THE MESSAGE FORM (docs/SPEC-TABLES.md §3.3): a FILE carries its own id +// table and a MESSAGE STREAM announces one and then carries none. +// +// A form 2 wire is TWO PARTS, the form byte and the root body: the body ends +// at its own zero reference as it does in a file, there is no trailer, and the +// message's last byte is the body's terminator. Its references resolve against +// the CONNECTION's table, which is the unit's whole vocabulary in the order +// the compiler settled. +const uint8_t kTableWireMessageForm = 2; + +// The RESERVED build-version id, the second id the language holds back (§5, +// §11), beside the node table's. It is the announcement's one required field, +// and a reserved id in any body but the one whose transport it is, is +// malformed (§3.1). +static const uint64_t kTableBuildVersionFieldId = 0xFFFFFFFFFFFFFFFEull; + +// The reserved NODE-TABLE id's own slot in this unit's vocabulary (§3.3). A +// pointered message names the node table through it, exactly as every other +// field header names its id through a slot. +static const uint64_t kTableNodeTableFieldSlot = 38; + +// THE UNIT'S ANNOUNCEMENT, byte for byte: 61 entries and 508 bytes. It is an +// ordinary form 1 FILE — the form byte, a body carrying the BUILD VERSION +// under the reserved id at kind 9, and the trailer that IS the connection's +// table, slot 1 the reserved id and slots 2 and up the vocabulary under one +// numbering. +// +// The vocabulary is the unit's whole closure in the COOK PROJECTION's order +// (§20.2) — each record in the order the projection renders it and each +// record's fields in the order the projection renders them, then each enum's +// variants and each union's arms — followed by the tail the projection does +// not name: the reserved node-table id, the three blob type ids as bytes, +// string and wstring, and every table's own name id in the projection's sorted +// record order. The tail is UNCONDITIONAL, so an ordinary edit only ever grows +// it at its end and never moves a slot a generated field header carries as a +// literal. +static const int64_t kTableAnnounceBytes = 508; +static const uint8_t kTableAnnounce[ kTableAnnounceBytes ] = { + 0x01, 0x01, 0x09, 0xc7, 0x71, 0x45, 0xca, 0xda, 0x0e, 0x7c, 0x8d, 0x00, + 0xfe, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x30, 0xb1, 0x3a, 0xff, + 0x4a, 0xd9, 0xb1, 0x40, 0x20, 0xea, 0x4d, 0x40, 0x8e, 0xa7, 0x19, 0xaa, + 0x26, 0xa9, 0x02, 0x0c, 0x9b, 0x01, 0x48, 0x78, 0xe9, 0xea, 0x71, 0x6f, + 0x0f, 0x01, 0x82, 0xbf, 0x6f, 0x2c, 0x41, 0x4f, 0xbf, 0x84, 0x78, 0x3e, + 0xf3, 0xa4, 0x48, 0x44, 0x19, 0xab, 0xd7, 0x56, 0x05, 0x4a, 0xa3, 0x30, + 0x67, 0x55, 0x5b, 0x85, 0xc9, 0xe2, 0x4e, 0x30, 0x69, 0x6a, 0xb4, 0x81, + 0xfb, 0x67, 0x4d, 0x1a, 0xcf, 0x7b, 0x27, 0x21, 0x74, 0xa2, 0x79, 0x44, + 0x8e, 0xe2, 0xe5, 0xb1, 0x84, 0x76, 0xbc, 0x2e, 0xef, 0x83, 0x76, 0x1e, + 0xc5, 0x99, 0xf7, 0x82, 0x76, 0x4e, 0x0a, 0xd9, 0xa8, 0x2e, 0x86, 0x70, + 0x84, 0xed, 0xf2, 0x4a, 0xbb, 0xf0, 0x0c, 0x9b, 0xcc, 0xfb, 0x2d, 0x73, + 0x68, 0xb7, 0xf0, 0xae, 0x4c, 0x0c, 0xf6, 0x52, 0xbf, 0xe9, 0xd1, 0x2f, + 0x93, 0xcd, 0xda, 0xdb, 0x22, 0x72, 0x34, 0x7d, 0xf6, 0x0b, 0x72, 0x17, + 0x07, 0x17, 0x02, 0x86, 0x4c, 0xf5, 0x63, 0xaf, 0x54, 0x15, 0x02, 0x86, + 0x4c, 0xf4, 0x63, 0xaf, 0x3a, 0x70, 0x6e, 0x3e, 0x93, 0x43, 0xe5, 0x9d, + 0x3d, 0x62, 0xcb, 0x8f, 0xec, 0xfc, 0xf7, 0x39, 0x09, 0x06, 0x02, 0x86, + 0x4c, 0xeb, 0x63, 0xaf, 0x09, 0x4b, 0x4d, 0x57, 0xaa, 0x33, 0x47, 0xd2, + 0x31, 0x54, 0xaf, 0x1d, 0x19, 0x73, 0x50, 0x12, 0xb2, 0x0f, 0x40, 0x27, + 0x0b, 0x6b, 0x98, 0x01, 0x38, 0x81, 0x0a, 0xf1, 0x1f, 0x06, 0xa7, 0xa3, + 0x0f, 0x62, 0xad, 0x07, 0x77, 0x47, 0x82, 0x5f, 0x42, 0x4f, 0x4f, 0x30, + 0x0d, 0x39, 0x84, 0x1c, 0x86, 0x1b, 0x63, 0x8e, 0xba, 0xad, 0xbc, 0xc4, + 0xec, 0x10, 0x5b, 0x36, 0x19, 0x4a, 0xc9, 0x3d, 0xea, 0x0c, 0xe8, 0x30, + 0x94, 0xfd, 0xe4, 0x7c, 0xec, 0x22, 0x02, 0x86, 0x4c, 0xfc, 0x63, 0xaf, + 0x05, 0x28, 0x02, 0x86, 0x4c, 0xff, 0x63, 0xaf, 0x52, 0x26, 0x02, 0x86, + 0x4c, 0xfe, 0x63, 0xaf, 0xb1, 0x45, 0xc3, 0x44, 0x35, 0xab, 0xfe, 0x73, + 0xc0, 0x7f, 0xb3, 0x8a, 0xbe, 0x08, 0x63, 0x7f, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xe4, 0x4f, 0x1c, 0x4f, 0x47, 0xc0, 0x2e, 0x2f, + 0x58, 0xfc, 0xaf, 0xfa, 0xd8, 0xe0, 0x4b, 0x70, 0xc7, 0xd4, 0x7b, 0x26, + 0xb0, 0x9d, 0x29, 0x5f, 0xcc, 0x14, 0x15, 0x7f, 0xcb, 0xc2, 0x58, 0xd8, + 0x84, 0x9e, 0x3a, 0x55, 0x8b, 0x37, 0xe2, 0x06, 0x2a, 0x89, 0xf5, 0x12, + 0x09, 0xc1, 0xca, 0x0a, 0x44, 0xa2, 0x31, 0xc1, 0xad, 0xa7, 0xee, 0xee, + 0xe8, 0xcf, 0xbf, 0x43, 0x73, 0x18, 0x43, 0xd0, 0x42, 0xad, 0xf6, 0xf8, + 0x59, 0x86, 0x63, 0x91, 0xb7, 0xce, 0x00, 0x7c, 0xd1, 0xc5, 0x34, 0x20, + 0x06, 0x68, 0x47, 0x98, 0xd1, 0xa1, 0xcf, 0x52, 0x5f, 0x82, 0x58, 0xac, + 0x36, 0x15, 0x78, 0x5e, 0xb8, 0x8b, 0x59, 0x6f, 0xc9, 0xc6, 0x86, 0xbb, + 0xc3, 0x64, 0x89, 0x50, 0xd2, 0x8d, 0xa7, 0xf1, 0x80, 0xea, 0x3a, 0xb9, + 0xf1, 0x21, 0xf7, 0x41, 0x11, 0xed, 0xd9, 0xce, 0x96, 0x92, 0x43, 0x8a, + 0xfb, 0x06, 0xc9, 0xfe, 0x19, 0xe1, 0x13, 0xa0, 0xa7, 0x0a, 0xc7, 0x54, + 0x12, 0xd6, 0x40, 0xdc, 0x08, 0xf0, 0xf5, 0xc0, 0x24, 0x5f, 0xf8, 0x33, + 0xc8, 0xfb, 0x85, 0x9a, 0xaf, 0xe0, 0xc9, 0x0c, 0x91, 0x0a, 0x55, 0x60, + 0xf7, 0xa2, 0x07, 0xec, 0x8b, 0x6d, 0x02, 0x86, 0x43, 0xf3, 0xc2, 0x2e, + 0x87, 0x27, 0xcc, 0x86, 0xf0, 0xe0, 0x26, 0x8f, 0x3d, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, +}; + +// AnnounceMeasure is the announcement's byte count, which is a constant of the +// unit and not a walk. +inline int64_t AnnounceMeasure() { return kTableAnnounceBytes; } + +// Announce writes the announcement into the caller's buffer and answers the +// bytes written — exactly AnnounceMeasure's answer — or -1 when the buffer is +// too small. It allocates nothing and walks nothing. +inline int64_t Announce( uint8_t * buffer, int64_t capacity ) +{ + if ( buffer == NULL || capacity < kTableAnnounceBytes ) { return -1; } + memcpy( buffer, kTableAnnounce, (size_t) kTableAnnounceBytes ); + return kTableAnnounceBytes; +} + +// TableVocabulary is ONE DIRECTION of ONE CONNECTION's id table (§3.3): the +// entries an announcement carried, whole, under one numbering with slot 1 the +// reserved build-version id. +// +// A peer holds TWO of these for a connection, the one it writes with and the +// one it reads with, and neither is the other's. A restart opens a fresh +// connection with empty tables and nothing is cached across connections, so +// its whole life is one connection's. It BORROWS the announcement's bytes rather than +// copying them, so a receiver holds one table a direction and its memory is +// the bound below and nothing else. +struct TableVocabulary +{ + // THE CONFORMING DEFAULT BOUND (§3.3): 32 KiB a direction, eight times the + // 500-id unit that is already a large one. A connection's table is bounded + // by nothing the wire carries, so the receiver declares the maximum and an + // announcement above it is refused by name before an entry is touched. + static const int64_t kDefaultMaxEntries = 4096; + + TableIdTable table; + uint64_t build_version = 0; + bool announced = false; + int64_t max_entries = kDefaultMaxEntries; +}; + +// AnnounceRead reads an announcement into one direction's table (§3.3). +// +// THE BOUND IS CHECKED BEFORE ANYTHING IS ALLOCATED: the entry count is a +// fixed little-endian u64 at the end, so a receiver reads it, compares it and +// refuses without touching an entry. After that it is §3's ordinary FILE read, +// because the announcement IS a file, with EXACTLY ONE STRICT CHECK over its +// body: the reserved build-version field present, exactly once, under kind 9, +// eight bytes wide. Everything else is an ordinary field under §4's tolerance, +// so an unknown one is skipped and counted and the announcement can GAIN a +// field in a later minor without a lockstep redeploy. +// +// The FIRST announcement sets the table and it is the only one that can. A +// SECOND is refused by name: it does not replace the table, it does not amend +// it and it changes nothing. A refused announcement sets NO TABLE. +inline bool AnnounceRead( TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + if ( vocabulary.announced ) + { + to->refused = true; + to->reason = second_announcement; + return false; + } + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireForm ) + { + to->refused = true; + to->reason = buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + return false; + } + if ( bytes < 9 ) { to->malformed = true; return false; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + if ( ( lo | ( hi << 32 ) ) > (uint64_t) vocabulary.max_entries ) + { + to->refused = true; + to->reason = vocabulary_too_large; + return false; + } + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else { to->refused = true; to->reason = newer_form; } + return false; + } + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) { to->malformed = true; return false; } + // the body, under §4's tolerance and this form's one strict check + TableReader r( buffer + 1, body_bytes, to, &table ); + uint64_t version = 0; + int32_t seen = 0; + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { to->malformed = true; return false; } + if ( ref == 0 ) { break; } + if ( ref > (uint64_t) table.count || !r.has( 1 ) ) { to->malformed = true; return false; } + const uint64_t id = table.at( ref ); + const uint8_t kind = r.get8(); + if ( id != kTableBuildVersionFieldId ) + { + to->unknown++; + if ( !r.skip( kind ) ) { to->malformed = true; return false; } + continue; + } + if ( kind != 9 || !r.has( 8 ) ) { to->refused = true; to->reason = no_vocabulary; return false; } + version = r.get64(); + seen++; + } + if ( seen != 1 ) { to->refused = true; to->reason = no_vocabulary; return false; } + vocabulary.table = table; + vocabulary.build_version = version; + vocabulary.announced = true; + return true; +} + +inline float table_bits_to_float( uint32_t bits ) { float f; memcpy( &f, &bits, 4 ); return f; } +inline uint32_t table_float_to_bits( float f ) { uint32_t b; memcpy( &b, &f, 4 ); return b; } +inline double table_bits_to_double( uint64_t bits ) { double d; memcpy( &d, &bits, 8 ); return d; } +inline uint64_t table_double_to_bits( double d ) { uint64_t b; memcpy( &b, &d, 8 ); return b; } + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +#ifndef LISTDEMO_SCHEMA_TABLE_ARENA +#define LISTDEMO_SCHEMA_TABLE_ARENA + +namespace listdemo { + +// ---- variable-length tables: tuning constants (docs/SPEC-TABLES.md) ---- +// +// The segment size and the count multiply to exactly 2^32: the u32 reference +// is the arena's hard ceiling, and these constants saturate it rather than +// leaving address space unreachable. Slab handout costs one atomic per slab, +// so per-node allocation costs no synchronization at all. + +static const uint32_t kTableSegmentBits = 22; // 4 MiB segments +static const uint32_t kTableSegmentSize = 1u << kTableSegmentBits; +static const uint32_t kTableSegmentMask = kTableSegmentSize - 1u; +static const uint32_t kTableMaxSegments = 1u << ( 32 - kTableSegmentBits ); // 1024 -> 4 GiB +static const uint32_t kTableSlabBytes = 64u * 1024u; // one atomic per slab +static const uint32_t kTableAlign = 8; // every node starts 8-aligned +static const uint32_t kTableAllocFailed = 0xFFFFFFFFu; + +// ---- THE CALLER'S ALLOCATOR (docs/SPEC-TABLES.md §6.5) ---- +// +// Every allocation the variable-length runtime makes goes through one of +// these — the arena's segments, the pack walk's identity map, the numbering's +// entry array, the packed region, and the tool path's node directory. There is +// no other call to the C library on this path, so a counting allocator sees +// every byte and a game's own heap can own all of it. +// +// It is the shape TableBlockAllocator already has (§19.1): two function +// pointers and a context the caller carries. What it adds is a CONTRACT ON +// alloc — the bytes come back ZEROED. Lock copies whole nodes, PADDING +// INCLUDED, so anything left uninitialized reaches a packed region; the default +// pair reaches that through calloc, which costs nothing measurable because a +// fresh segment is untouched pages either way. +struct TableAllocator +{ + void * ( *alloc )( void * context, int64_t bytes ); // ZEROED bytes, NULL on failure + void ( *free )( void * context, void * pointer ); + void * context; +}; + +// The default pair, and it is the one every entry point takes when the caller +// names none. It calls schema_allocate / schema_release, so a program with its +// own C-library replacement can move the floor without writing a struct at all. +inline void * table_default_alloc( void * context, int64_t bytes ) { (void) context; return schema_allocate( bytes ); } +inline void table_default_free( void * context, void * pointer ) { (void) context; schema_release( pointer ); } + +inline TableAllocator TableDefaultAllocator() +{ + TableAllocator allocator; + allocator.alloc = table_default_alloc; + allocator.free = table_default_free; + allocator.context = NULL; + return allocator; +} + +// ---- TableRef: a relocatable reference (never a machine pointer) ---- +// +// Two encodings, one slot, and the FORM says which is in force: +// +// in the arena — the node's arena offset (segment index in the high bits) +// in a region — the SELF-RELATIVE byte delta from this slot's own address, +// so a deref is one add, needs no base pointer, and a whole +// region relocates by memcpy with zero fix-up +// +// 0 is null in both, and a slot can never name the node that contains it, so +// zero names nothing real in either form. +// +// A REGION DELTA HAS NO REQUIRED SIGN (§6.3). A region is packed depth-first, +// so a node's FIRST reference points forward; every LATER reference to that +// same node points BACK at the one body it already has, which is exactly what +// makes one node one node in a region. Sharing and a back-reference are the +// same fact, and nothing validates a reference by its sign. +// +// IT IS EIGHT BYTES, SIGNED, so ONE REGION REACHES EVERYTHING (§6.3, §7): a +// four-byte slot bounded a region at 2 GiB, and the scale a cook exists for is +// *"100mbs or many gigabytes of data in Assets.bin"*. +struct TableRef +{ + int64_t value = 0; + bool null() const { return value == 0; } +}; + +// TableSlot is what Alloc hands back: usable as the node pointer (write +// fields through it) AND as the reference to store in a pointer field. +template struct TableSlot +{ + T * ptr = NULL; + TableRef ref; + T * operator->() const { return ptr; } + T & operator*() const { return *ptr; } + operator T *() const { return ptr; } + operator TableRef() const { return ref; } + bool null() const { return ptr == NULL; } +}; + +inline uint32_t TableAlignUp( uint32_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( kTableAlign - 1 ); } +inline int64_t TableAlignUp64( int64_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( int64_t( kTableAlign ) - 1 ); } + +// ---- a BYTE BUFFER's node (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// A *bytes or *string slot is a TableRef like every pointer slot, and it names +// a BLOB NODE: this eight-byte header and then the bytes, at offset eight so +// the data is eight-aligned. A *string blob carries one more zero byte after +// its data, so a region hands back a C string with no copy. The node's extent +// is the header plus its bytes, rounded to the arena's alignment like every +// node's; on the wire it is a record whose body is the bytes (§3.1). +struct TableBlob +{ + uint32_t length; + uint32_t zero; +}; + +static const int64_t kTableBlobHeader = 8; // length (u32), then four zero bytes +static const int64_t kTableBlobMaxLength = 0xFFFFFFFF; // a record's length is a u32 (§3.1) + +// the node's storage: the header, the bytes, a string's terminator, rounded +// to the arena's alignment like every node +inline int64_t TableBlobStorage( int64_t length, bool terminated ) +{ + return TableAlignUp64( kTableBlobHeader + length + ( terminated ? 1 : 0 ) ); +} + +// What a read answers: a pointer INTO the region and the length, NULL and +// zero for a null slot. Off a locked region, a loaded one or an opened cook +// the pointer is one add from the slot, and nothing is copied. +struct TableBytesView +{ + const uint8_t * data; + int64_t length; +}; + +struct TableStringView +{ + const char * data; // zero-terminated + int64_t length; +}; + +// What AllocBytes and AllocString hand back: the bytes to write through, the +// length asked for, and the reference to store in the slot — the three +// answers TableSlot gives for a table node. +struct TableBytesSlot +{ + uint8_t * data = NULL; + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +struct TableStringSlot +{ + char * data = NULL; // room for length bytes and the terminator, already zero + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +// ---- the arena: segmented, slab-handed, lock-free by ownership ---- +// +// Allocation is thread-local inside a worker's slab — no atomics on the node +// path. A worker takes its next slab with ONE compare-exchange, and a new +// segment is published with one more. Nothing ever moves: a segment, once +// allocated, lives untouched until the arena is torn down, so a T* obtained +// from Alloc stays valid while other workers allocate, and an offset stays +// correct while the arena grows. +// +// The model this DELIBERATELY refuses: one buffer under a lock, grown by +// realloc. A realloc moves the buffer under workers mid-write; offsets fix +// identity but not the raw references already resolved from them, and the +// resulting corruption is invisible until much later. Segments never move, so +// that bug class cannot be written here. +// +// Slack: at most one slab tail per worker plus one slab per segment (a slab +// that will not fit is skipped rather than split), i.e. under 2% of a segment +// plus threads x 64 KiB. That is the price of never synchronizing per node. +struct TableArena +{ + std::atomic segments[ kTableMaxSegments ]; + std::atomic cursor; // (segment << kTableSegmentBits) | bytes handed out + bool locked = false; // MONOTONIC: Lock() is one-way, there is no unlock + // THE ARENA CARRIES ITS OWN, so everything downstream of a builder — + // segments, pack map, numbering, region, node directory — allocates through + // the one pair the caller named, with nothing to thread by hand. + TableAllocator allocator; +}; + +inline void TableArenaInit( TableArena & arena, TableAllocator allocator ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + arena.segments[i].store( NULL, std::memory_order_relaxed ); + } + arena.cursor.store( 0, std::memory_order_relaxed ); + arena.locked = false; + arena.allocator = allocator; +} + +inline void TableArenaShutdown( TableArena & arena ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + uint8_t * segment = arena.segments[i].exchange( NULL, std::memory_order_acq_rel ); + if ( segment != NULL ) { arena.allocator.free( arena.allocator.context, segment ); } + } + arena.cursor.store( 0, std::memory_order_relaxed ); +} + +// one L1 load plus an add: the segment table is 8 KiB and stays hot +inline uint8_t * TableArenaAt( const TableArena & arena, uint32_t offset ) +{ + return arena.segments[ offset >> kTableSegmentBits ].load( std::memory_order_relaxed ) + ( offset & kTableSegmentMask ); +} + +// TableArenaGrabSlab hands one worker its next private slab. Returns +// kTableAllocFailed when the arena's address space or the allocator is +// exhausted — a loud refusal, never a silent smaller slab. +inline uint32_t TableArenaGrabSlab( TableArena & arena ) +{ + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t segment = cursor >> kTableSegmentBits; + uint32_t used = cursor & kTableSegmentMask; + // strictly less: a slab is never split across segments, and the tail + // is the documented slack + if ( used + kTableSlabBytes < kTableSegmentSize ) + { + if ( arena.segments[segment].load( std::memory_order_acquire ) == NULL ) + { + // THE SEGMENT COMES BACK ZEROED, which is the allocator's + // contract and not an extra pass here: Lock copies whole nodes, + // PADDING INCLUDED, so anything uninitialized reaches a packed + // region. Value-initializing a node with placement new zeroes + // its MEMBERS and not its padding, so the zeroing has to happen + // at the segment or not at all. It costs nothing measurable: a + // fresh segment is untouched pages either way, and the default + // pair's calloc has the kernel hand them over zeroed. + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, (int64_t) kTableSegmentSize ); + if ( memory == NULL ) { return kTableAllocFailed; } + uint8_t * expected = NULL; + if ( !arena.segments[segment].compare_exchange_strong( expected, memory, std::memory_order_acq_rel ) ) + { + // another worker published this segment first + arena.allocator.free( arena.allocator.context, memory ); + } + } + if ( arena.cursor.compare_exchange_weak( cursor, cursor + kTableSlabBytes, std::memory_order_acq_rel ) ) + { + return ( segment << kTableSegmentBits ) | used; + } + continue; + } + uint32_t next_segment = segment + 1; + if ( next_segment >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + arena.cursor.compare_exchange_weak( cursor, next_segment << kTableSegmentBits, std::memory_order_acq_rel ); + } +} + +// TableArenaGrabSpan reserves a SPAN of the arena's address space for one node +// larger than a slab — a BYTE BUFFER of any size (docs/SPEC-TABLES.md §2.5) — +// and allocates it as one contiguous block. It takes whole segment indices +// from the cursor, starting at the index after the cursor's so nothing else +// is ever handed out inside the span, and publishes the block under the first +// of them; the indices the span covers past that one stay NULL, which is +// enough, because only a node's START is ever resolved through the segment +// table and a blob's bytes follow its header inside the one allocation. The +// unused tail of the segment the cursor was in is slack, like a slab tail. +// Returns kTableAllocFailed when the address space or the allocator is +// exhausted — a loud refusal, never a smaller blob. +inline uint32_t TableArenaGrabSpan( TableArena & arena, int64_t bytes ) +{ + if ( bytes <= 0 || bytes > ( (int64_t) kTableMaxSegments - 2 ) * (int64_t) kTableSegmentSize ) { return kTableAllocFailed; } + const uint32_t spanned = (uint32_t) ( ( bytes + kTableSegmentSize - 1 ) >> kTableSegmentBits ); + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t start = ( cursor >> kTableSegmentBits ) + 1; + if ( start + spanned >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + uint32_t next = ( start + spanned ) << kTableSegmentBits; + if ( !arena.cursor.compare_exchange_weak( cursor, next, std::memory_order_acq_rel ) ) { continue; } + // the span is this worker's now: nothing else can publish under its + // first index, so a plain store suffices, and the block comes back + // ZEROED like every segment — the blob's bytes and its tail are zeros + // until written + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, bytes ); + if ( memory == NULL ) { return kTableAllocFailed; } + arena.segments[start].store( memory, std::memory_order_release ); + return start << kTableSegmentBits; + } +} + +// ---- TableWorker: one thread's allocation front ---- +// +// The threading contract, stated plainly: +// * Alloc on YOUR OWN worker is safe concurrently with any other worker's. +// No locks, no atomics per node. +// * Writing fields of a node ANOTHER worker allocated is your own +// synchronization problem — this runtime does not arbitrate it. +// * Lock and Save are single-threaded: call them after the workers have +// joined. +struct TableWorker +{ + TableArena * arena = NULL; + uint32_t next = 0; + uint32_t end = 0; + + template TableSlot Alloc() + { + static_assert( alignof( T ) <= kTableAlign, "a table node's alignment must fit the arena's" ); + TableSlot slot; + if ( arena == NULL || arena->locked ) { return slot; } + uint32_t bytes = TableAlignUp( (uint32_t) sizeof( T ) ); + if ( bytes > kTableSlabBytes ) { return slot; } // a node larger than a slab: refused, never split + if ( end == 0 || next + bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return slot; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + uint32_t at = next; + next += bytes; + // A NODE IS BORN IN TWO HALVES: start its lifetime in the raw + // storage, then write the declared defaults ONE MEMBER AT A TIME. + // + // It is "T", not "T{}". Value-initialising the whole aggregate says + // the same thing and costs cl O(BYTES) TO COMPILE — it expands element + // by element in its front end — while both halves here cost + // O(declarations). The slab cap below refuses a large node at RUN + // TIME and bounds nothing at compile time: the cost is paid by + // whatever T a caller instantiates this with. + // Padding is not the difference: value-initialisation zeroes MEMBERS + // and not padding either way, which is why the segment is calloc'd. + // + // TableReset is an OVERLOAD SET, one per closure member, reached from + // this template by argument-dependent lookup on T's own namespace — + // Alloc is a template and cannot spell Reset. + // + // The reset is here because ONE DEFINITION SAYS WHAT THE DECLARED + // DEFAULTS ARE, and it is Reset. Default-initialisation lands on + // the same values today, because a member with a non-zero default + // carries a member initializer that says so — but that is the class + // definition agreeing with Reset, not the arena reading it, and #320's + // fix was itself a pass that MOVED initialisation between the two. + // The arena reads the definition. + slot.ptr = new ( TableArenaAt( *arena, at ) ) T; + TableReset( *slot.ptr ); + slot.ref.value = at; + return slot; + } + + // Alloc a BYTE BUFFER's node of exactly length bytes (docs/SPEC-TABLES.md + // §2.5): the blob header and its bytes, zeroed, in this thread's slab when + // it fits and in a span of the arena's own when it does not. NULL is the + // arena locked, a length below zero or past a record's u32, or the + // allocator refusing. The offset comes back for the reference. + TableBlob * AllocBlob( int64_t length, bool terminated, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( length < 0 || length > kTableBlobMaxLength ) { return NULL; } + const int64_t bytes = TableBlobStorage( length, terminated ); + if ( bytes > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, bytes ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + } + else + { + if ( end == 0 || next + (uint32_t) bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) bytes; + } + TableBlob * blob = (TableBlob *) TableArenaAt( *arena, at ); + blob->length = (uint32_t) length; // the bytes after it are the segment's zeros + blob->zero = 0; + return blob; + } + + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries + // no type id, takes no index and has no Reset, so it goes through the same + // slab and span the blob path uses rather than through Alloc. + uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( bytes <= 0 || align > (int64_t) kTableAlign ) { return NULL; } + const int64_t rounded = TableAlignUp64( bytes ); + if ( rounded > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, rounded ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + return TableArenaAt( *arena, at ); + } + if ( end == 0 || next + (uint32_t) rounded > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) rounded; + return TableArenaAt( *arena, at ); // the segment came back zeroed + } + // a *bytes node: the bytes to write through, and the reference to store + TableBytesSlot AllocBytes( int64_t length ) + { + TableBytesSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, false, at ); + if ( blob == NULL ) { return slot; } + slot.data = (uint8_t *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } + + // a *string node: room for length bytes and the zero byte after them + TableStringSlot AllocString( int64_t length ) + { + TableStringSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, true, at ); + if ( blob == NULL ) { return slot; } + slot.data = (char *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } +}; + +// ---- TablePackMap: the pack walk's identity map (docs/SPEC-TABLES.md §3.1, §6.2) ---- +// +// ONE ENTRY PER REACHABLE NODE, and that map IS identity: a node must know +// where it landed to be named a second time, so Lock packs a shared node ONCE +// and every later reference resolves to the one body it already has. That is +// the same first-visit numbering the wire uses, so the pack order and the node +// order are one order. +// +// COLOURING AN ENTRY WHILE ITS DESCENT IS OPEN COSTS ONE BIT, and it is what +// makes a data cycle free to refuse: a reference to an entry still open is a +// cycle, and Lock returns failure rather than recursing away. The ROOT's entry +// is open for the whole walk. +// +// The map is proportional to NODES, never to bytes, and it lives on the +// AUTHORING side, where §6.5 licenses allocation. Nothing on the reading path +// ever builds one. +struct TablePackEntry +{ + const void * key; // the node's address in the graph being packed + int64_t offset; // where that node landed in the region + uint8_t open; // its descent is still open: a reference here is a cycle +}; + +struct TablePackMap +{ + TablePackEntry * entries = NULL; + int64_t capacity = 0; // a power of two, or zero while empty + int64_t count = 0; + TableAllocator allocator; // the caller's, carried from the walk that built it +}; + +inline void TablePackMapInit( TablePackMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TablePackMapShutdown( TablePackMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TablePackMapInit( map, map.allocator ); +} + +// The two walks behind Lock re-derive the SAME map from the same graph — the +// numbering is never carried between them (§3.1) — so the second starts from +// an empty map and keeps the capacity the first paid for. +inline void TablePackMapReset( TablePackMap & map ) +{ + if ( map.entries != NULL ) { memset( map.entries, 0, (size_t) map.capacity * sizeof( TablePackEntry ) ); } + map.count = 0; +} + +// open addressing, linear probing, a multiply-shift hash over the address: a +// node key is a pointer and its low bits are alignment, so the low bits alone +// would collide on every node of one type +inline int64_t TablePackMapSlot( const TablePackMap & map, const void * key ) +{ + uint64_t hash = (uint64_t) (uintptr_t) key; + hash *= 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != NULL && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TablePackEntry * TablePackMapFind( TablePackMap & map, const void * key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TablePackEntry * entry = &map.entries[ TablePackMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +// QUADRUPLING, not doubling, and the reason is measured: growth rehashes every +// entry, and on a graph of 131,071 nodes the doubling schedule spent 45% of +// Lock in rehashing alone. Quadrupling from 1024 buys 1.35x on that graph and +// keeps the map NODE-proportional (§6.2) — under 128 bytes a node at its +// worst, right after a grow, and about 64 on average. +inline bool TablePackMapGrow( TablePackMap & map ) +{ + TablePackMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 1024; + grown.entries = (TablePackEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TablePackEntry ) ); + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == NULL ) { continue; } + grown.entries[ TablePackMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// REACH a node: one probe answers both questions the walk has. A true "taken" +// says this is a FIRST visit, and the entry is now the node's, coloured open +// at "offset"; otherwise the entry is the one the node already has, and its +// open bit says cycle or sharing. NULL is an allocation failure, and it is a +// refusal like any other: Lock fails rather than packing a graph it cannot +// track. +// +// It is one call and not a find followed by an insert because the walk asks +// this question twice per node — once to measure, once to pack — and every +// probe is a miss into a table larger than L2. +inline TablePackEntry * TablePackMapReach( TablePackMap & map, const void * key, int64_t offset, bool & taken, int64_t & slot ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TablePackMapGrow( map ) ) { return NULL; } + } + slot = TablePackMapSlot( map, key ); + TablePackEntry * entry = &map.entries[slot]; + taken = entry->key != key; // an empty slot is a first visit; the key is never NULL + if ( taken ) + { + entry->key = key; + entry->offset = offset; + entry->open = 1; + map.count++; + } + return entry; +} + +// The descent finished: the node keeps its entry — identity outlives the +// descent — and stops being a cycle. The "hint" is the slot Reach returned, and it +// is checked against the key rather than trusted, so a rehash between the two +// costs a second probe instead of correctness. +inline void TablePackMapClose( TablePackMap & map, const void * key, int64_t hint ) +{ + if ( hint >= 0 && hint < map.capacity && map.entries[hint].key == key ) + { + map.entries[hint].open = 0; + return; + } + TablePackEntry * entry = TablePackMapFind( map, key ); + if ( entry != NULL ) { entry->open = 0; } +} + +// ---- resolution contexts: which encoding a walk is reading ---- + +struct TableArenaCtx { const TableArena * arena; }; +struct TableRegionCtx {}; + +// ---- a BYTE BUFFER's resolution (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// The same two encodings a table pointer has, resolved the same way: a +// self-relative delta in a region — one add, no base — and an arena offset +// while the builder is mutable. The blob is reached through its header, and a +// view is the header plus eight and the header's first word. Nothing here +// allocates and nothing copies: off a locked region, a loaded one or an +// opened cook the view points INTO the region. +inline const TableBlob * TableBlobAt( const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableRegionCtx &, const TableRef & ref ) { return TableBlobAt( ref ); } +inline const TableBlob * TableBlobAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} + +inline TableBytesView TableBytesViewOf( const TableBlob * blob ) +{ + TableBytesView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const uint8_t *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} +inline TableStringView TableStringViewOf( const TableBlob * blob ) +{ + TableStringView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const char *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} + +// the const form's hot path: one add, no base +inline TableBytesView TableBytesAt( const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ref ) ); } +inline TableStringView TableStringAt( const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ref ) ); } +// and the context forms a walk uses: a region context, an arena context, or +// the arena itself while the builder is mutable +template inline TableBytesView TableBytesAt( const Ctx & ctx, const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ctx, ref ) ); } +template inline TableStringView TableStringAt( const Ctx & ctx, const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ctx, ref ) ); } + +// allocate a blob in the arena and point the slot at it; the slot holds the +// arena offset, as every slot does while the builder is mutable +inline uint8_t * TableBytesEmplace( TableWorker & worker, TableRef & slot, int64_t length ) +{ + TableBytesSlot allocated = worker.AllocBytes( length ); + slot = allocated.ref; + return allocated.data; +} +// the text is copied in when one is given; a NULL text leaves the zeros for +// the caller to fill +inline char * TableStringEmplace( TableWorker & worker, TableRef & slot, const char * text, int64_t length ) +{ + TableStringSlot allocated = worker.AllocString( length ); + slot = allocated.ref; + if ( allocated.data != NULL && text != NULL && length > 0 ) { memcpy( allocated.data, text, (size_t) length ); } + return allocated.data; +} + +// ---- the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table, and a +// pointer field rides as an INDEX into it under kind 17. The encoding is +// flat: no pointer edge is a nesting level, so a chain's length is not a depth, +// and two references to one node are one node. +// +// THE FIELD RIDES ONCE: an L with sixty-four bits of capability frames a +// numbering of any size, so the whole numbering is one contiguous payload and a +// save's node bodies have no aggregate ceiling. + +static const uint64_t kTableNodeIndexNull = 0; // absence and null are one value +static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts the table + +// The not-materialized sentinel (§6.3): a record whose type id this build could +// not name. Distinct from every real offset including the root's 0, so an index +// resolving through it yields NULL and can never fabricate the root. +static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; + +// ---- the numbering, on the SAVE side ---- +// +// One entry per reachable node in FIRST-VISIT order, so entry k is node index +// k + 2. The two thunks are what let one loop write a table of mixed types: the +// numbering walk knows each target's type STATICALLY at the site it numbers it, +// so it stores the instantiation there and the loop never asks what a node is. +struct TableNumbering; + +struct TableNodeEntry +{ + const void * node; + uint64_t type_id; + // the type id's MESSAGE-FORM SLOT (docs/SPEC-TABLES.md §3.3), stored where + // the numbering walk stores the id itself and for the same reason: the + // target's type is known STATICALLY at the site that numbers it, so a + // form 2 save reads the slot out of the entry instead of looking an id up. + // Every pointer target's type id is an entry of the announcement, which is + // what makes the slot a compile-time fact of a POINTERED message too. + uint64_t type_slot; + int64_t ( * measure )( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ); + bool ( * save )( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ); +}; + +struct TableNumbering +{ + TablePackMap seen; // node -> index; the ROOT is index 1, open for the whole walk + TableNodeEntry * entries = NULL; + int64_t count = 0; + int64_t capacity = 0; +}; + +// The numbering allocates through the map's pair rather than carrying a second +// copy of it: one numbering is one walk, and a walk has one allocator. +inline void TableNumberingInit( TableNumbering & n, TableAllocator allocator ) +{ + TablePackMapInit( n.seen, allocator ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +inline void TableNumberingShutdown( TableNumbering & n ) +{ + TableAllocator allocator = n.seen.allocator; + TablePackMapShutdown( n.seen ); + allocator.free( allocator.context, n.entries ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +// The index a numbered node was given, for the save that writes it into a +// pointer slot. False means the two walks disagree about the graph, which is a +// refusal and never a guess. +inline bool TableNumberingIndex( const TableNumbering & n, const void * node, uint64_t & index ) +{ + if ( n.seen.capacity == 0 ) { return false; } + const TablePackEntry & entry = n.seen.entries[ TablePackMapSlot( n.seen, node ) ]; + if ( entry.key != node ) { return false; } + index = (uint64_t) entry.offset; + return true; +} + +inline bool TableNumberingAppend( TableNumbering & n, const TableNodeEntry & entry ) +{ + if ( n.count == n.capacity ) + { + // GROW BY COPY, never by realloc: the allocator hook is a PAIR, and a + // game's heap is not required to have a resize primitive at all. The + // schedule quadruples, so the copying is amortized to a constant per + // entry and the growth is the same growth it always was. + int64_t capacity = n.capacity != 0 ? n.capacity * 4 : 256; + TableAllocator allocator = n.seen.allocator; + TableNodeEntry * grown = (TableNodeEntry *) allocator.alloc( allocator.context, capacity * (int64_t) sizeof( TableNodeEntry ) ); + if ( grown == NULL ) { return false; } + if ( n.entries != NULL ) + { + memcpy( grown, n.entries, (size_t) n.count * sizeof( TableNodeEntry ) ); + allocator.free( allocator.context, n.entries ); + } + n.entries = grown; + n.capacity = capacity; + } + n.entries[n.count++] = entry; + return true; +} + +// The thunks the numbering stores. Each resolves to the closure member's own +// MeasureBody / SaveBodyFields through an overload set in the member's DECLARING +// file, reached by argument-dependent lookup at instantiation — the same bridge +// the arena's TableReset uses, and the reason a numbering may span the files of +// one unit without any file naming another's members. +template +inline int64_t TableNodeMeasureThunk( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ) +{ + return TableNodeMeasure( *(const Ctx *) ctx, numbering, ids, *(const T *) node ); +} + +template +inline bool TableNodeSaveThunk( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ) +{ + return TableNodeSave( *(const Ctx *) ctx, numbering, w, ids, *(const T *) node ); +} + +// ---- a BYTE BUFFER's record (docs/SPEC-TABLES.md §2.5, §3.1) ---- +// +// A blob rides as a node record under one of two RESERVED type ids — the fold +// a table's name takes, over the keywords "bytes" and "string", which no table +// can be named — with the bytes as its body and nothing framed inside. These +// two thunks are what the numbering stores for a blob, as it stores a +// member's codec for a table: the length, and the bytes verbatim. +static const uint64_t kTableBytesTypeId = 0x2f2ec0474f1c4fe4ull; // fnv1a64( "bytes" ) +static const uint64_t kTableStringTypeId = 0x704be0d8faaffc58ull; // fnv1a64( "string" ) + +template +inline int64_t TableBlobMeasureThunk( const void *, const TableNumbering &, TableIds &, const void * node ) +{ + return (int64_t) ( (const TableBlob *) node )->length; +} + +template +inline bool TableBlobSaveThunk( const void *, const TableNumbering &, TableWriter & w, TableIds &, const void * node ) +{ + const TableBlob * blob = (const TableBlob *) node; + w.raw( (const void *) ( blob + 1 ), (int64_t) blob->length ); + return true; +} + +// TableNodeTableMeasure and TableNodeTableSave are the framing, and they are +// ONE fill rule written twice — measure derives it from the graph and save +// derives the same one, which is what makes measure == save hold across a +// pointer graph (§3.1). +// +// The field rides ONCE, under the reserved id, kind 12: the payload opens with +// the count and then carries the records back to back, each a type id +// REFERENCE, a length and a body. The reserved id is interned BEFORE the +// records, and a record's type id before its body, which is the first-use order +// the trailer is written in (§3). +template +inline int64_t TableNodeTablePayload( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + int64_t payload = TableLebBytes( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + payload += TableLebBytes( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return -1; } + payload += TableLebBytes( (uint64_t) body ) + body; + } + return payload; +} + +template +inline int64_t TableNodeTableMeasure( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return 0; } // a root that reaches no nodes writes none of them + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return -1; } + return TableLebBytes( ref ) + 1 + TableLebBytes( (uint64_t) payload ) + payload; +} + +template +inline bool TableNodeTableSave( const Ctx & ctx, TableWriter & w, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return true; } + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return false; } + w.putleb( ref ); + w.put8( 12 ); // kind 12 is the opaque byte payload: a reader that cannot name the id skips by L + w.putleb( (uint64_t) payload ); + w.putleb( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + w.putleb( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return false; } + w.putleb( (uint64_t) body ); + if ( !n.entries[k].save( (const void *) &ctx, n, w, ids, n.entries[k].node ) ) { return false; } + } + return true; +} + +// ---- the numbering, on the LOAD side: a region's NODE DIRECTORY (§6.3) ---- +// +// The wire's numbering made resident: one entry per numbered node, in index +// order, position i describing node index i + 1 — so position 0 is the ROOT at +// offset 0. It is ATTRIBUTION, and attribution is separable: nothing that reads +// a structure touches it, a deref is one add on a self-relative offset, and a +// caller may release it once Load returns. +struct TableNodeDirEntry +{ + uint64_t offset; + uint64_t type_id; +}; + +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; + +// TableNodeMap is what a pointer slot resolves through while a body decodes. +struct TableNodeMap +{ + uint8_t * base = NULL; + const TableNodeDirEntry * entries = NULL; + int64_t count = 0; // the ROOT's entry included, so it is records + 1 + bool good = false; // the node table read whole; a numbering that failed resolves nothing + // WHERE THE NODES LIVE, and therefore what a resolved slot holds: a region + // takes the SELF-RELATIVE delta so a deref is one add, and the tool's + // builder path takes the node's ARENA OFFSET (§6.3). + bool arena = false; + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. + TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; +}; + +// TableNodeResolve places one node index in a pointer slot, and every failure +// is one of §4's events with the pointer left null. The declared TARGET type id +// is checked at every index, the root's included: the root carries no record +// and therefore no wire type id, so the READER'S OWN root type is what the +// claim is checked against. +inline void TableNodeResolve( const TableNodeMap & map, TableRef & slot, uint64_t index, uint64_t target, TableReport * report ) +{ + slot.value = 0; + if ( index == kTableNodeIndexNull || !map.good ) { return; } + if ( index - 1 >= (uint64_t) map.count ) + { + report->malformed = true; // an index above node_count + 1 + return; + } + const TableNodeDirEntry & entry = map.entries[index - 1]; + if ( entry.offset == kTableNodeAbsent ) + { + // a node whose type id this build could not name KEEPS ITS INDEX, and + // every pointer naming it reads null. The unknown was counted once, at + // the node, not once per pointer. + return; + } + if ( entry.type_id != target ) + { + report->kind_mismatch++; + return; + } + slot.value = map.arena ? (int64_t) entry.offset + : (int64_t) ( ( map.base + entry.offset ) - (const uint8_t *) &slot ); +} + +// ---- the record SCAN, and it is the whole of load's bound (§3.1) ---- +// +// Reading follows no reference. The scan walks the root body's top-level fields, +// finds the ONE under the reserved id, and reads records out of its payload in +// order — the field rides once, so nothing is copied to make a body contiguous +// and the generated body decoder never learns the transport exists. +struct TableNodeScan +{ + TableReader fields; // over the ROOT body, skipping past everything else + const uint8_t * payload; // the node-table field's payload + int64_t payload_size; + int64_t payload_offset; + bool opened; // the root body has been walked for the field + uint64_t declared; + int64_t records; + bool present; // the root body carries a node table at all + bool malformed; + const TableIdTable * ids; +}; + +inline TableNodeScan TableNodeScanBegin( const uint8_t * body, int64_t size, TableReport * report, const TableIdTable * ids ) +{ + TableNodeScan s = { TableReader( body, size, report, ids ), NULL, 0, 0, false, 0, 0, false, false, ids }; + return s; +} + +// find the node-table field, or answer false when the root body has none. A +// body carrying an id more than once is legal input and THE LAST OCCURRENCE +// WINS (docs/SPEC-TABLES.md §3), so the walk runs to the terminator and keeps +// the last rather than stopping at the first. +inline bool TableNodeScanOpen( TableNodeScan & s ) +{ + if ( s.opened ) { return false; } + s.opened = true; + for ( ;; ) + { + uint64_t ref = 0; + if ( !s.fields.getleb( ref ) ) { break; } + if ( ref == 0 ) { break; } // the terminator + if ( s.ids == NULL || ref > (uint64_t) s.ids->count ) { break; } + const uint64_t id = s.ids->at( ref ); + if ( !s.fields.has( 1 ) ) { break; } + const uint8_t kind = s.fields.get8(); + if ( id == kTableNodeTableFieldId ) + { + s.present = true; + if ( kind != 12 ) { s.malformed = true; return false; } + uint64_t length = 0; + if ( !s.fields.getleb( length ) || !s.fields.room( length ) ) { s.malformed = true; return false; } + s.payload = s.fields.buffer + s.fields.offset; + s.payload_size = (int64_t) length; + s.fields.offset += (int64_t) length; + continue; + } + if ( !s.fields.skip( kind ) ) { break; } + } + if ( s.payload == NULL ) { return false; } + TableReader head( s.payload, s.payload_size, s.fields.report, s.ids ); + if ( !head.getleb( s.declared ) ) { s.malformed = true; return false; } + s.payload_offset = head.offset; + return true; +} + +// the next record, or false at the end of the table — s.malformed says whether +// the end was the end or the framing giving out +inline bool TableNodeScanNext( TableNodeScan & s, uint64_t & type_id, const uint8_t * & body, int64_t & length ) +{ + if ( !s.opened && !TableNodeScanOpen( s ) ) { return false; } + if ( s.payload == NULL || s.payload_offset >= s.payload_size ) { return false; } + TableReader rec( s.payload, s.payload_size, s.fields.report, s.ids ); + rec.offset = s.payload_offset; + uint64_t ref = 0; + if ( !rec.getleb( ref ) || ref == 0 || s.ids == NULL || ref > (uint64_t) s.ids->count ) + { + s.malformed = true; // a type id reference of 0, or one past the table + return false; + } + type_id = s.ids->at( ref ); + uint64_t declared_length = 0; + if ( !rec.getleb( declared_length ) ) + { + s.malformed = true; // a record whose length is damaged + return false; + } + if ( declared_length > (uint64_t) ( s.payload_size - rec.offset ) ) + { + s.malformed = true; // a record whose length runs past its field + return false; + } + body = s.payload + rec.offset; + length = (int64_t) declared_length; + s.payload_offset = rec.offset + length; + s.records++; + return true; +} + +// The record scan is AUTHORITATIVE: node_count is data from the wire, and a +// count that disagrees with the scan is malformed. Nothing is sized from it +// before the scan has confirmed it. +inline bool TableNodeScanWhole( TableNodeScan & s ) +{ + if ( s.malformed ) { return false; } + if ( !s.present ) { return true; } // no node table at all is not a broken one + return s.declared == (uint64_t) s.records; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_ARENA + +#ifndef LISTDEMO_SCHEMA_TABLE_EXTENT +#define LISTDEMO_SCHEMA_TABLE_EXTENT + +namespace listdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_EXTENT + +#ifndef LISTDEMO_SCHEMA_TABLE_MAP +#define LISTDEMO_SCHEMA_TABLE_MAP + +namespace listdemo { + +// ---- a MAP: a sorted entry array, and the lookup over it (§2.8) ---- +// +// On the wire, in a region and in a cook a map is an array of one generated +// ENTRY table held in ascending key order. What this adds is Find — a binary +// search over that array where it lies — and a builder that inserts, replaces +// and erases by key. Nothing here is stored: a region and a cook carry the +// array and the count, and not one byte about a hash or a probe. + +// entries carved from ONE call to the allocator pair; a new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableMapSegmentEntries = 32; + +// TableDeclRef names a type in an unevaluated context and is never defined — +// what 's declval is for, without the include the generated corpus +// refuses to pay for (the iterator_traits note, §13.9). +template T & TableDeclRef(); + +// THE ORDER IS TOTAL, AND IT IS THE SAME IN NINE LANGUAGES (§2.8). Integers +// compare by VALUE, signed for the signed kinds and unsigned for the unsigned. +// Strings compare by BYTES, unsigned, a shorter string that is a prefix of a +// longer one first: memcmp over the common length, then the lengths. Never a +// locale, never a code point, never a case fold. +inline int TableKeyOrder( uint64_t a, uint64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( int64_t a, int64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( const char * a, int32_t a_length, const char * b, int32_t b_length ) +{ + const int32_t common = a_length < b_length ? a_length : b_length; + if ( common > 0 ) + { + const int order = memcmp( (const void *) a, (const void *) b, (size_t) common ); + if ( order != 0 ) { return order < 0 ? -1 : 1; } + } + return a_length < b_length ? -1 : ( a_length > b_length ? 1 : 0 ); +} + +// the length of a NUL-terminated key at a call site, bounded by the storage it +// has to fit: a key one byte longer than the bound is refused, never truncated +inline int32_t TableKeyLength( const char * key, int32_t bound ) +{ + if ( key == NULL ) { return 0; } + for ( int32_t i = 0; i <= bound; i++ ) { if ( key[i] == 0 ) { return i; } } + return bound + 1; // longer than the bound: the caller refuses it +} + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.8, §7.2) ---- +// +// An int64 self-relative reference to the entry array and an int32 count, then +// padding to eight. The reference is a TableRef like a pointer's: in the arena +// it names the builder's HEAD, in a region it is the delta from the slot to +// the first entry, and 0 is the empty map in both. +template struct TableMap +{ + TableRef entries; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Entry * Entries() const + { + return entries.value != 0 ? (const Entry *) ( (const uint8_t *) &entries + entries.value ) : NULL; + } + int32_t size() const { return count; } + + // FIND: floor( log2 n ) + 1 key compares, in place, no allocation. NULL + // when absent, and on a map[K]*T the RESOLVED pointer, which is what a + // pointer field's accessor answers. + template const Entry * FindEntry( Key key ) const + { + const Entry * base = Entries(); + int32_t low = 0, high = count; + while ( low < high ) + { + const int32_t mid = low + ( high - low ) / 2; + const int order = TableEntryOrder( base[mid], key ); + if ( order == 0 ) { return base + mid; } + if ( order < 0 ) { low = mid + 1; } else { high = mid; } + } + return NULL; + } + // the return type is DEDUCED, so it is worked out when a call site + // instantiates Find and not when the holder's record declares the slot — + // which is what lets the entry's own overloads be declared after it + template auto Find( Key key ) const + { + return TableEntryFound( FindEntry( key ) ); + } + + // ---- iteration: ASCENDING key order, the key beside the value ---- + // + // A proxy BY VALUE, the keyed array's shape (§2.4): for ( auto [ key, + // value ] : map ). It carries no iterator_traits, for the reason + // TableKeyed's does not (§13.9). + struct ConstEntry + { + decltype( TableEntryKey( TableDeclRef() ) ) key; + decltype( TableEntryFound( (const Entry *) NULL ) ) value; + }; + + struct ConstIterator + { + const Entry * at; + ConstEntry operator*() const { return ConstEntry{ TableEntryKey( *at ), TableEntryFound( at ) }; } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Entries() }; } + ConstIterator end() const { return ConstIterator{ Entries() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.8, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first entry is inserted. Each +// segment is a fixed number of entries carved from one call to the allocator +// pair. An entry's address is stable for the arena's life, so a value handed +// back by an insert stays valid while other entries arrive. +struct TableMapHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an insert appends into + int32_t live; + int32_t dead; +}; + +template struct TableMapSegment +{ + TableRef next; + int32_t used; // entries carved from this segment + int32_t padding; + uint32_t dead[ ( kTableMapSegmentEntries + 31 ) / 32 ]; // Erase marks one bit, never the entry + Entry entries[ kTableMapSegmentEntries ]; +}; + +inline bool TableMapSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// ---- the ORDERED CURSOR the four writing walks read (§2.8) ---- +// +// Measure, Save, Lock and Cook each write a map's entries in ascending key +// order with no key twice, deriving the order from the builder's entries as +// each walk derives the numbering (§3.1). Nothing passes between them, so +// measure == save over a map is a real check on two sorts agreeing. +// +// A REGION is already sorted, so its cursor is the array in place and +// allocates nothing. The BUILDER's is the sort: an array of entry pointers +// allocated through the pair and released before the walk returns, because +// sorting the segments themselves would move entries whose addresses a caller +// holds. +template struct TableMapCursor +{ + const Entry * const * order = NULL; // the builder's form: sorted pointers + const Entry * entries = NULL; // the region's form: the array in place + int32_t count = 0; + TableAllocator allocator; + bool ok = false; + const Entry * operator[]( int32_t index ) const + { + return order != NULL ? order[index] : entries + index; + } +}; + +// heapsort: O( n log n ) once per map, no recursion, no allocation past the +// pointer array the caller already paid for +template inline void TableMapSort( const Entry ** order, int32_t count ) +{ + for ( int32_t start = count / 2 - 1; start >= 0; start-- ) + { + int32_t root = start; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= count ) { break; } + if ( child + 1 < count && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * swap = order[root]; order[root] = order[child]; order[child] = swap; + root = child; + } + } + for ( int32_t end = count - 1; end > 0; end-- ) + { + const Entry * swap = order[0]; order[0] = order[end]; order[end] = swap; + int32_t root = 0; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= end ) { break; } + if ( child + 1 < end && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * hold = order[root]; order[root] = order[child]; order[child] = hold; + root = child; + } + } +} + +// the REGION form: the array is already sorted, so the cursor is the array +template +inline TableMapCursor TableMapOrder( const TableRegionCtx &, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.entries = map.Entries(); + cursor.count = map.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: gather the LIVE entries out of the segment chain in +// insertion order, then sort. A dead entry costs nothing on any wire (§2.8). +template +inline TableMapCursor TableMapOrder( const TableArena & arena, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.allocator = arena.allocator; + cursor.count = map.count; + if ( map.entries.value == 0 || map.count <= 0 ) { cursor.ok = map.count == 0; cursor.count = 0; return cursor; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + if ( head->live != map.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + const Entry ** order = (const Entry **) arena.allocator.alloc( arena.allocator.context, (int64_t) map.count * (int64_t) sizeof( const Entry * ) ); + if ( order == NULL ) { return cursor; } + int32_t at = 0; + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 && at < map.count ) + { + const TableMapSegment * segment = (const TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used && at < map.count; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + order[at++] = segment->entries + i; + } + segment_ref = segment->next; + } + if ( at != map.count ) + { + arena.allocator.free( arena.allocator.context, order ); + return cursor; + } + TableMapSort( order, map.count ); + cursor.order = order; + cursor.ok = true; + return cursor; +} + +template +inline TableMapCursor TableMapOrder( const TableArenaCtx & ctx, const TableMap & map ) +{ + return TableMapOrder( *ctx.arena, map ); +} + +template inline void TableMapRelease( TableMapCursor & cursor ) +{ + if ( cursor.order != NULL ) { cursor.allocator.free( cursor.allocator.context, (void *) cursor.order ); } + cursor.order = NULL; +} + +// ---- the builder's five (§2.8) ---- +// +// Insert APPENDS after one LINEAR SCAN of the live entries for the key it may +// replace, Find is that same scan, and Erase is the scan and one bit. The +// builder builds NO INDEX, and that is a rule: the sort happens once, at Lock, +// Save or Cook, and every lookup that matters runs over the sorted region. + +// the head, allocated when the first entry is inserted +template +inline TableMapHead * TableMapReach( TableWorker & worker, TableMap & map ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( map.entries.value != 0 ) { return (TableMapHead *) TableArenaAt( *worker.arena, (uint32_t) map.entries.value ); } + uint32_t at = 0; + TableMapHead * head = (TableMapHead *) worker.AllocRaw( (int64_t) sizeof( TableMapHead ), (int64_t) alignof( TableMapHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + map.entries.value = (int64_t) at; + return head; +} + +// one entry's storage, appended: the current segment when it has room, a new +// one carved from one call to the pair when it does not +template +inline Entry * TableMapAppend( TableWorker & worker, TableMapHead * head, TableMap & map ) +{ + TableMapSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableMapSegmentEntries ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableMapSegment *) worker.AllocRaw( (int64_t) sizeof( TableMapSegment ), (int64_t) alignof( TableMapSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableMapSegment * previous = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Entry * entry = segment->entries + segment->used; + segment->used++; + head->live++; + map.count++; + return entry; +} + +// the LINEAR SCAN: the live entries in insertion order, O( n ) key compares +template +inline Entry * TableMapScan( const TableArena & arena, const TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) == 0 ) { return segment->entries + i; } + } + segment_ref = segment->next; + } + return NULL; +} + +// ERASE marks the entry DEAD, one bit in the segment's slot and not in the +// entry table, and decrements the live count. Its storage is reclaimed at +// RESET and never reused mid-build, because reusing a slot would make "an +// entry's address is stable" false for exactly one case. +template +inline bool TableMapErase( TableArena & arena, TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return false; } + TableMapHead * head = (TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) != 0 ) { continue; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + map.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INSERTION order, live entries only (§2.8) ---- +template struct TableMapEach +{ + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableMapSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableMapSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + auto operator*() const { return TableEntryEach( segment->entries + index ); } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableMapEach TableMapEachOf( const TableArena & arena, const TableMap & map ) +{ + TableMapEach each = { &arena, TableRef() }; + if ( map.entries.value != 0 ) + { + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + each.first = head->first; + } + return each; +} + +// ---- the LOAD side: where a decoded entry lands (§2.8) ---- +// +// THE READER TRUSTS NOTHING and spends one compare per entry. Every load path +// applies the same rules and produces one report (§4), so the region load of +// §6.5 and LoadBuilder never disagree about a wire. These two shapes are what +// makes that true with one generated decoder: a REGION carves the entry array +// out of the holder node's own extent, and the TOOL's path appends into the +// builder's arena, and the decoder above them cannot tell which it has. + +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. + +// TableMapFill is one map field being decoded: where the next entry lands, and +// the entry that last LANDED, which is what the ascending check compares +// against. +template struct TableMapFill +{ + TableMap * map = NULL; + Entry * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; +}; + +template +inline TableMapFill TableMapFillBegin( const TableNodeMap & nodes, TableMap & map, uint32_t n ) +{ + TableMapFill fill; + fill.map = ↦ + map.entries.value = 0; + map.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Entry ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Entry ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Entry *) base; + fill.capacity = (int32_t) n; + map.entries.value = (int64_t) ( base - (const uint8_t *) &map.entries ); + fill.ok = true; + return fill; +} + +// the entry that last LANDED — NULL before the first +template inline Entry * TableMapFillLast( TableMapFill & fill ) +{ + if ( fill.map->count <= 0 ) { return NULL; } + if ( fill.array != NULL ) { return fill.array + ( fill.map->count - 1 ); } + return TableMapLive( *fill.worker->arena, *fill.map, fill.map->count - 1 ); +} + +// the next slot, at the entry type's declared defaults +template inline Entry * TableMapFillNext( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + if ( fill.map->count >= fill.capacity ) { return NULL; } + Entry * entry = fill.array + fill.map->count; + TableReset( *entry ); + fill.map->count++; + return entry; + } + TableMapHead * head = TableMapReach( *fill.worker, *fill.map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( *fill.worker, head, *fill.map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// A MAP WITH HALF ITS KEYS IS NOT A MAP (§2.8): at the first entry whose key +// kind disagrees with the reader's declaration the map resets to EMPTY, one +// kind_mismatch is counted for the map, and its remaining bytes are skipped. +template inline void TableMapFillReset( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + fill.map->entries.value = 0; + fill.map->count = 0; + return; + } + if ( fill.map->entries.value != 0 ) + { + TableMapHead * head = (TableMapHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.map->entries.value ); + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + } + fill.map->count = 0; +} + +// an EMPTY map's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableMapFillEnd( TableMapFill & fill ) +{ + if ( fill.array != NULL && fill.map->count == 0 ) { fill.map->entries.value = 0; } +} + +// the k-th LIVE entry of a builder map, in insertion order — what the tool +// path's ascending check compares against +template +inline Entry * TableMapLive( const TableArena & arena, const TableMap & map, int32_t index ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + int32_t at = 0; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( at == index ) { return segment->entries + i; } + at++; + } + segment_ref = segment->next; + } + return NULL; +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.8, §6.5) ---- +// +// LoadMeasure's term for a map is N x sizeof( Entry ) rounded to +// alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this +// reads no field: it walks the map's own header and, where an entry's value +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. +// A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its +// own L and the body's terminator, and under this form's variable lengths that +// footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a +// map's L can carry, and therefore what a LoadMeasure may be asked for. +static const int64_t kTableMapEntryFloor = 2; + +inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry + at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); + at += (int64_t) n * entry_size; + if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// ---- the TEXT form's placement (docs/SPEC-TABLES.md §2.8, §16) ---- +// +// The text is a plain JSON object keyed by the KEY, and the generic walk fills +// it through the ENTRY'S OWN descriptor — so all it needs from here is one +// entry at one key, handed back at its defaults. It is the builder's Insert +// with the ENTRY returned rather than its value, because the walk writes the +// value through a field row and not through a typed pointer. +template +inline Entry * TableMapPlace( TableWorker & worker, TableMap & map, Key key ) +{ + if ( worker.arena == NULL ) { return NULL; } + Entry * found = TableMapScan( *worker.arena, map, key ); + if ( found != NULL ) + { + TableResetMapValue( *found ); // a repeated key is LAST-WINS, whole + return found; + } + TableMapHead * head = TableMapReach( worker, map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( worker, head, map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// ---- the OPTIONAL RUNTIME INDEX (§2.8) ---- +// +// Open addressing with LINEAR PROBING over the sorted array, built AT LOAD for +// a map large enough that log n compares over a cold array cost more than one +// hash and a probe. IT IS NEVER STORED: the caller measures it, owns its +// storage, builds it in one pass and releases it whenever. +// +// ITS HASH AND ITS LOAD FACTOR ARE NOT A CROSS-PORT CONTRACT, and that is a +// rule. What a port is held to is the CONTRACT of the lookup: the same value +// the sorted array's Find returns for the same key, and no allocation past the +// storage the caller handed in. +struct TableMapIndex +{ + int32_t * slots = NULL; // entry indices, +1; 0 is an empty slot + int32_t capacity = 0; + bool good = false; +}; + +// this runtime's own, and no port reproduces it: fnv1a64 over the key's bytes +inline uint64_t TableMapHash( const void * bytes, int32_t length ) +{ + uint64_t hash = 0xCBF29CE484222325ull; + const uint8_t * at = (const uint8_t *) bytes; + for ( int32_t i = 0; i < length; i++ ) { hash ^= (uint64_t) at[i]; hash *= 0x100000001B3ull; } + return hash; +} +inline uint64_t TableMapHash( uint64_t key ) { return TableMapHash( (const void *) &key, (int32_t) sizeof( key ) ); } + +// this runtime's own load factor, and no port reproduces it either: the next +// power of two at or above twice the count, so a probe run stays short +inline int32_t TableMapIndexSlots( int32_t count ) +{ + int32_t slots = 8; + while ( slots < count * 2 ) { slots *= 2; } + return slots; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_MAP + +#ifndef LISTDEMO_SCHEMA_TABLE_LIST +#define LISTDEMO_SCHEMA_TABLE_LIST + +namespace listdemo { + +// ---- an UNBOUNDED ARRAY: a counted array whose count the data decides (§2.9) ---- +// +// On the wire, in a region and in a cook a list is the kind 14 body a [..N]T +// writes, its elements by-value records inside the holder's node extent. What +// this adds is the slot, a builder that appends into segments that never +// move, and a const surface that indexes and iterates in place. There is no +// sort, no key and no lookup: the order is INSERTION order, and it is +// identity the way position is identity in a fixed array. + +// elements carved from ONE call to the allocator pair. A new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableListSegmentElements = 32; + +// THE ELEMENT STORAGE: T itself, and a TableRef slot for a []*T, whose +// elements are references exactly as a pointer field's slot is (§2.1) +template struct TableListStorage { typedef T Element; }; +template struct TableListStorage { typedef TableRef Element; }; + +// WHAT THE CONST FORM ANSWERS: the element by reference, and on a []*T the +// RESOLVED pointer, one add on the self-relative delta, NULL for a null slot, +// exactly as At answers it (§6.2, §6.3) +template struct TableListConst +{ + typedef const T & Result; + static Result At( const T * element ) { return *element; } +}; +template struct TableListConst +{ + typedef const T * Result; + static Result At( const TableRef * element ) + { + return element->value != 0 ? (const T *) ( (const uint8_t *) element + element->value ) : NULL; + } +}; + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.9, §7.2) ---- +// +// An int64 self-relative reference to the element array and an int32 count, +// then padding to eight. The reference is a TableRef like a pointer's: in the +// arena it names the builder's HEAD, in a region it is the delta from the slot +// to the first element, and 0 is the empty list in both. It is the map's slot +// exactly, because it is the same two facts. +template struct TableList +{ + typedef typename TableListStorage::Element Element; + + TableRef elements; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Element * Elements() const + { + return elements.value != 0 ? (const Element *) ( (const uint8_t *) &elements + elements.value ) : NULL; + } + int32_t size() const { return count; } + + // INDEXING IS BOUNDS-CHECKED IN EVERY BUILD (§2.4, §2.9): the extent is a + // number that CAME FROM A FILE, so an index past it is not a mistake a + // release build gets to make cheaply. There is no undefined-behavior path + // here in any configuration. The assert carries the message where a + // debugger can read it and NDEBUG removes that. The fatal is what stands + // after it. Both go through the hooks: define schema_assert and + // schema_fatal and this refusal lands in your own handler. + void RefuseIndex( int32_t index ) const + { + if ( (uint32_t) index >= (uint32_t) count ) + { + schema_assert( false && "an unbounded array is indexed inside its count, which came from a file" ); + schema_fatal(); + } + } + typename TableListConst::Result operator[]( int32_t index ) const + { + RefuseIndex( index ); + return TableListConst::At( Elements() + index ); + } + + // ---- iteration: INDEX order, the element and no key ---- + // + // It carries no iterator_traits, for the reason TableKeyed's does not + // (§13.9). + struct ConstIterator + { + const Element * at; + typename TableListConst::Result operator*() const { return TableListConst::At( at ); } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Elements() }; } + ConstIterator end() const { return ConstIterator{ Elements() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.9, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first element is added. Each +// segment is a fixed number of elements carved from one call to the allocator +// pair. An element's address is stable for the arena's life, so a T * handed +// back by Add stays valid while other elements arrive. +struct TableListHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an Add appends into + int32_t live; + int32_t dead; +}; + +template struct TableListSegment +{ + TableRef next; + int32_t used; // elements carved from this segment + int32_t padding; + uint32_t dead[ ( kTableListSegmentElements + 31 ) / 32 ]; // Erase marks one bit, never the element + Element elements[ kTableListSegmentElements ]; +}; + +inline bool TableListSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// the head, allocated when the first element is added +template +inline TableListHead * TableListReach( TableWorker & worker, TableList & list ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( list.elements.value != 0 ) { return (TableListHead *) TableArenaAt( *worker.arena, (uint32_t) list.elements.value ); } + uint32_t at = 0; + TableListHead * head = (TableListHead *) worker.AllocRaw( (int64_t) sizeof( TableListHead ), (int64_t) alignof( TableListHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + list.elements.value = (int64_t) at; + return head; +} + +// one element's storage, appended: the current segment when it has room, a +// new one carved from one call to the pair when it does not. NULL means NOT +// ADDED: an arena that cannot carve another segment, or a count at the int32 +// cap (§2.2, §2.9). +template +inline typename TableList::Element * TableListAppend( TableWorker & worker, TableListHead * head, TableList & list ) +{ + typedef typename TableList::Element Element; + if ( list.count >= INT32_MAX ) { return NULL; } // the int32 storage cap + TableListSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableListSegmentElements ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableListSegment *) worker.AllocRaw( (int64_t) sizeof( TableListSegment ), (int64_t) alignof( TableListSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableListSegment * previous = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Element * element = segment->elements + segment->used; + segment->used++; + head->live++; + list.count++; + return element; +} + +// ADD, whole: the head, the append, and the element at its declared defaults +// (§2.9). The text form's placement is this same call, because a list has no +// key to place under (§16). +template +inline typename TableList::Element * TableListPlace( TableWorker & worker, TableList & list ) +{ + typedef typename TableList::Element Element; + TableListHead * head = TableListReach( worker, list ); + if ( head == NULL ) { return NULL; } + Element * element = TableListAppend( worker, head, list ); + if ( element == NULL ) { return NULL; } + new ( element ) Element(); // value-init: the declared defaults, and null for a slot + return element; +} + +// ERASE, ADDRESSED BY THE POINTER (§2.9): the element Add handed back is the +// handle, because a list has no key and the address is the one thing the +// builder promises never moves (§6.4). It marks the element DEAD, one bit in +// the segment's slot and not in the element storage, and decrements the live +// count. False when the pointer is not this list's. Its storage is reclaimed +// at RESET and never reused mid-build, the map's rule for the map's reason. +template +inline bool TableListErase( TableArena & arena, TableList & list, const typename TableList::Element * element ) +{ + typedef typename TableList::Element Element; + if ( list.elements.value == 0 || element == NULL ) { return false; } + TableListHead * head = (TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableListSegment * segment = (TableListSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + if ( element >= segment->elements && element < segment->elements + segment->used ) + { + const int32_t i = (int32_t) ( element - segment->elements ); + if ( TableListSegmentDead( segment->dead, i ) ) { return false; } // already erased + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + list.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INDEX order, live elements only (§2.9) ---- +template struct TableListEach +{ + typedef typename TableList::Element Element; + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableListSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableListSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + Element * operator*() const { return segment->elements + index; } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableListEach TableListEachOf( const TableArena & arena, const TableList & list ) +{ + TableListEach each = { &arena, TableRef() }; + if ( list.elements.value != 0 ) + { + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + each.first = head->first; + } + return each; +} + +// ---- the INDEX-ORDER CURSOR the four writing walks read (§2.9) ---- +// +// Measure, Save, Lock and Cook each visit a list's live elements in the order +// they were added, and they allocate nothing to do it: a region's cursor is +// the array in place, and the builder's walks the segment chain. Indexing the +// builder's form is SEQUENTIAL by construction, every walk steps i, i + 1, +// i + 2, so the cursor remembers where the last access landed and moves one +// live slot per step. An access behind the memo restarts from the first +// segment, which no walk here does. +template struct TableListCursor +{ + const Element * elements = NULL; // the region's form: the array in place + const TableArena * arena = NULL; // the builder's form: the segments + TableRef first; + int32_t count = 0; + bool ok = false; + // the memo: the segment and slot the last access landed on, and the live + // index that slot holds + mutable const TableListSegment * segment = NULL; + mutable int32_t within = -1; + mutable int32_t logical = -1; + + const Element * At( int32_t index ) const + { + if ( elements != NULL ) { return elements + index; } + if ( segment == NULL || index < logical ) + { + segment = first.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL; + within = -1; + logical = -1; + } + while ( logical < index ) + { + for ( ;; ) + { + within++; + while ( segment != NULL && within >= segment->used ) + { + segment = segment->next.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + within = 0; + } + if ( segment == NULL ) { return NULL; } // the slot and the head disagree + if ( !TableListSegmentDead( segment->dead, within ) ) { break; } + } + logical++; + } + return segment->elements + within; + } + const Element & operator[]( int32_t index ) const { return *At( index ); } +}; + +// the REGION form: the array is the cursor +template +inline TableListCursor::Element> TableListElements( const TableRegionCtx &, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.elements = list.Elements(); + cursor.count = list.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: the live elements out of the segment chain, in the +// order they were added. A dead element costs nothing on any wire (§2.9). +template +inline TableListCursor::Element> TableListElements( const TableArena & arena, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.arena = &arena; + cursor.count = list.count; + if ( list.elements.value == 0 || list.count <= 0 ) { cursor.ok = list.count == 0; cursor.count = 0; return cursor; } + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + if ( head->live != list.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + cursor.first = head->first; + cursor.ok = true; + return cursor; +} + +template +inline TableListCursor::Element> TableListElements( const TableArenaCtx & ctx, const TableList & list ) +{ + return TableListElements( *ctx.arena, list ); +} + +// ---- the LOAD side: where a decoded element lands (§2.9) ---- +// +// The same two shapes the map's fill takes, because the decoder above them +// cannot tell which it has: a REGION carves the element array out of the +// holder node's own extent, PRE-ORDER, and the TOOL's path appends into the +// builder's arena. +template struct TableListFill +{ + typedef typename TableList::Element Element; + TableList * list = NULL; + Element * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; + bool refused = false; // a count above the int32 cap on the tool's path: LoadBuilder answers NULL +}; + +template +inline TableListFill TableListFillBegin( const TableNodeMap & nodes, TableList & list, uint64_t n ) +{ + typedef typename TableList::Element Element; + TableListFill fill; + fill.list = &list; + list.elements.value = 0; + list.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( n > (uint64_t) INT32_MAX ) + { + // A COUNT ABOVE THE int32 STORAGE CAP (§2.2, §2.9): into a region it was + // refused by LoadMeasure before this ran, and into a builder it is the + // refusal LoadBuilder answers NULL for, moving no counter + fill.refused = nodes.carve->worker != NULL; + return fill; + } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Element ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Element ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Element *) base; + fill.capacity = (int32_t) n; + list.elements.value = (int64_t) ( base - (const uint8_t *) &list.elements ); + fill.ok = true; + return fill; +} + +// the next slot, at the element's declared defaults. NULL when the arena +// could not carve, which the decoder reports as framing damage +template inline typename TableList::Element * TableListFillNext( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count >= fill.capacity ) { return NULL; } + Element * element = fill.array + fill.list->count; + new ( element ) Element(); + fill.list->count++; + return element; + } + return TableListPlace( *fill.worker, *fill.list ); +} + +// A SLOT WHOSE ELEMENT NEVER LANDED is given back (§2.9, §4): the array keeps +// what it decoded, and an element whose own framing gave out before one byte +// of it decoded was not decoded. The region's form uncounts it, and the builder's +// marks it dead, which is what the storage rule allows mid-build. +template inline void TableListFillDrop( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count > 0 ) { fill.list->count--; } + return; + } + if ( fill.list->elements.value == 0 ) { return; } + TableListHead * head = (TableListHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.list->elements.value ); + if ( head->last.value == 0 ) { return; } + TableListSegment * segment = (TableListSegment *) TableArenaAt( *fill.worker->arena, (uint32_t) head->last.value ); + if ( segment->used <= 0 ) { return; } + const int32_t i = segment->used - 1; + if ( TableListSegmentDead( segment->dead, i ) ) { return; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + fill.list->count--; +} + +// an EMPTY list's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableListFillEnd( TableListFill & fill ) +{ + if ( fill.array != NULL && fill.list->count == 0 ) { fill.list->elements.value = 0; } +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.9, §6.5) ---- +// +// N x sizeof( T ) rounded to alignof( T ), AT EVERY DEPTH. N is framing and +// not a value, so this reads no field: it walks the list's own header and, +// where a table element holds a list or a map of its own, the elements' +// headers under it. Every -1 carries its REASON (§6.5): the int32 cap first, +// because a count past it cannot fit any body, and then the body's own L. +inline bool TableListWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t elem_size, int64_t elem_align, uint8_t elem_kind, int64_t elem_floor, + TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != elem_kind ) { return true; } // another element kind: §4's ordinary kind mismatch, the field reads empty + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; } // an N the list's L cannot carry + at = ( at + elem_align - 1 ) & ~( elem_align - 1 ); + at += (int64_t) n * elem_size; + if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_LIST + +#ifndef LISTDEMO_SCHEMA_BUILD_VERSION +#define LISTDEMO_SCHEMA_BUILD_VERSION + +namespace listdemo { + +// THE BUILD VERSION (docs/SPEC-TABLES.md §20): one digest over every fact the bytes +// this build produces depend on — the type wire's protocol id, every record's +// layout as the compiler's own C ABI model computes it, and the facts that +// decide what a load PUTS in those slots. It is the number a cook's header +// carries and the number Open compares, and the number a block's prologue +// carries and BlockOpen compares: a build version answers "which build?" and +// not "which form?", and what separates the two forms is their MAGIC. +// +// There are TWO ids in the design and they are not interchangeable: the +// PROTOCOL ID is the type wire's and nothing else, and the BUILD VERSION is +// what everything cooked or blocked is keyed by. A table edit moves this and +// never the protocol id; a type edit moves both. +static const uint64_t BuildVersion = 0x8d7c0edaca4571c7ull; + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_BUILD_VERSION + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK +#define LISTDEMO_SCHEMA_TABLE_COOK + +namespace listdemo { + +// ---- the cooked form (docs/SPEC-TABLES.md §7) ---- +// +// A cooked file is a HEADER, a DATA part and an ATTRIBUTION part, in that +// order. Every word of the header is a u64 written in the byte order the cook +// was produced in, and the header is 64 bytes: +// +// 0 magic 0x4b4f4f434d484353, read BYTEWISE before anything else +// 8 build_version the unit's id (docs/SPEC-TABLES.md §20) +// 16 byte_order 1 little, 2 big — the order that WROTE the file +// 24 data_length the region's bytes, rounded up to alignment +// 32 attribution_length the directory's bytes, or 0 +// 40 alignment the region's alignment, never below eight +// 48 reserved zero +// 56 reserved zero +// +// The DATA part is Lock's region written verbatim (§7.2) — the root at its +// base — and it is what a runtime points at. The ATTRIBUTION part is the node +// directory (§6.3), and NOTHING THAT READS THE STRUCTURE TOUCHES IT: it is +// written beside the data for schema cook-check, so a build that ships no +// tooling need not carry it at all. +static const int64_t kTableCookHeaderBytes = 64; + +// THE MAGIC'S VALUE, and a consumer written from the page needs the constant +// rather than a description of one. It is "SCHMCOOK" read as ASCII in the byte +// order a little-endian store produces — the same shape the block form's +// SCHMABLK takes, so a hex dump of a little-endian cook is legible and the two +// accelerators sit in one vocabulary. +// +// IT IS STORED IN THE PRODUCER'S ORDER, which is what makes it the byte-order +// check as well as the form check: a consumer reads back this build's +// constant, or that constant byte-reversed — which identifies a cook of the +// OTHER order — or something that is not a cook. All three answers but the +// first refuse, and a cook and a BLOCK are separated here too, because a +// form's identity belongs in its magic rather than in a second digest. +static const uint64_t TableCookMagic = 0x4b4f4f434d484353ull; + +// THIS BUILD's byte order, as the header's own word carries it. The magic is +// what REFUSES a foreign order; this word is what RECORDS which order wrote +// the file, so a refusal names the order rather than inferring it and a tool +// dumping a cook reads the fact. A file whose magic matched and whose order +// word did not is corrupt, and there is no reading that recovers it. +// +// The BUILD VERSION cannot do either job: §20.1 digests byteorder as a +// GENERATION input, little for every target schema generates for today, so +// two builds of one schema for two orders emit the same id. +#if defined( __BYTE_ORDER__ ) && defined( __ORDER_BIG_ENDIAN__ ) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__ +static const uint64_t TableCookByteOrder = 2; // big +#else +static const uint64_t TableCookByteOrder = 1; // little +#endif + +// The greatest region alignment a cooked file may name. The DATA part begins +// at align_up( 64, alignment ), which is 64 for every unit this language can +// declare — the largest alignment it has is sixteen — so a word past this cap +// describes a file no build of this schema wrote (docs/SPEC-TABLES.md §7.1). +static const uint64_t TableCookMaxAlign = 64; + +// The header read, BYTEWISE. memcpy is the portable spelling of "these eight +// bytes, in this machine's order"; every compiler this repo builds under folds +// it to one load, and it is the only read in the whole of Open that is not a +// comparison. +inline uint64_t table_cook_read64( const uint8_t * p ) +{ + uint64_t v; + memcpy( &v, p, sizeof( v ) ); + return v; +} + +// TableCookOpen: THE WHOLE CHECK, in one place, because §7 states the +// enumeration once and every generated Open is that one enumeration plus +// its own root's two layout facts. +// +// THE CHECK, in order: the magic read bytewise, the byte order it establishes, +// the build version against this build's own, both RESERVED words zero, the +// region alignment the header names, the two part lengths against the length +// the caller passed — a truncated file and a file with trailing bytes are the +// same refusal — the root's own storage inside the data part, and the +// alignment of the base. +// +// AND THAT IS ALL OF IT. On a match the bytes ARE what this build wrote, in +// this build's layout and this build's byte order, so there is nothing to +// validate and nothing to fix up: the caller gets the root. Nothing per node +// happens here, which is what makes open O(1) in the file's size; a walk of +// any shape would forfeit that, and validating an untrusted file is schema +// cook-check's job and a person's decision (§7.4). +// +// EVERY NUMBER BELOW COMES OUT OF THE FILE, so the arithmetic is unsigned and +// each term is BOUNDED BEFORE IT IS ADDED: a forged length near 2^64 must +// refuse, and an addition that wrapped would be the defect the comparison +// after it was supposed to catch. Nothing past length is read on any path, +// including every refusing one. +inline const uint8_t * TableCookOpen( const void * bytes, uint64_t length, uint64_t root_size, uint64_t root_align ) +{ + if ( bytes == NULL ) { return NULL; } + if ( length < (uint64_t) kTableCookHeaderBytes ) { return NULL; } + const uint8_t * raw = (const uint8_t *) bytes; + // the MAGIC, bytewise and first: it is what establishes the byte order + // every other header word is read in, so nothing else may be read before + // it. A byte-reversed constant is a cook of the other order and refuses + // here, which is why the order never reaches a fix-up pass. + if ( table_cook_read64( raw ) != TableCookMagic ) { return NULL; } + if ( table_cook_read64( raw + 16 ) != TableCookByteOrder ) { return NULL; } + if ( table_cook_read64( raw + 8 ) != BuildVersion ) { return NULL; } + // the RESERVED words: a non-zero one means a writer used a form this build + // does not understand, and Open refuses rather than ignoring it. + if ( table_cook_read64( raw + 48 ) != 0 ) { return NULL; } + if ( table_cook_read64( raw + 56 ) != 0 ) { return NULL; } + const uint64_t data_length = table_cook_read64( raw + 24 ); + const uint64_t attribution_length = table_cook_read64( raw + 32 ); + const uint64_t alignment = table_cook_read64( raw + 40 ); + // THE ALIGNMENT WORD IS DATA, and it is the one header field the rest of + // the check does arithmetic WITH rather than only comparison against. A + // region's alignment is a power of two, never below eight (the floor that + // puts the attribution part on an eight-byte boundary without a second + // padding rule) and never past the cap above; a word that is none of those + // rounds nothing and aligns nothing, so it is refused before it is used. + if ( alignment < 8 || alignment > TableCookMaxAlign ) { return NULL; } + if ( ( alignment & ( alignment - 1 ) ) != 0 ) { return NULL; } + // and it must be an alignment THE ROOT CAN SIT AT, since the root is at + // the region's base: both are powers of two, so "at least the root's" + // is one division. + if ( ( alignment % root_align ) != 0 ) { return NULL; } + // The DATA part begins at align_up( 64, alignment ). It is DERIVED and not + // a header field, because a fact a reader computes is a fact two writers + // cannot disagree about. + const uint64_t data_offset = ( (uint64_t) kTableCookHeaderBytes + alignment - 1 ) & ~( alignment - 1 ); + if ( length < data_offset ) { return NULL; } + // the two part lengths against the length the caller passed. The whole + // file is data_offset + data_length + attribution_length, and a length + // that is not EXACTLY that refuses — truncation and trailing bytes are one + // refusal, and both terms are subtracted rather than added so no sum can + // carry. + if ( data_length > length - data_offset ) { return NULL; } + if ( attribution_length != length - data_offset - data_length ) { return NULL; } + // the ROOT sits at the region's base, so the region has to hold it: a + // shorter data part describes a root partly outside the file, which is the + // one way a match-and-point reader could hand back storage it never + // received. + if ( data_length < root_size ) { return NULL; } + const uint8_t * base = raw + data_offset; + // the alignment of the BASE. The header pads the data part to the region's + // alignment, so a base an allocator or mmap gave you is already aligned — + // mmap gives page alignment for free — and a base that is not is a caller's + // buffer this form cannot be read out of. + if ( ( (uintptr_t) base % (uintptr_t) alignment ) != 0 ) { return NULL; } + return base; +} + +// ---- the cooked form, the WRITE side (docs/SPEC-TABLES.md §7.6) ---- +// +// THE BYTE ORDER IS THE TARGET'S, NOT THE HOST'S. A cook is produced in the +// byte order of the build that will read it (§7), so the fixing happens here — +// offline, once, on the writing side — and never at Open. Passing +// TableByteOrder::Big on a little-endian machine produces a big-endian build's +// file, and nothing about the writing host reaches the bytes. +enum class TableByteOrder +{ + Little = 1, // the header's byte_order word, and the order every scalar is written in + Big = 2, +}; + +// One store, width as an argument. Every call site passes a literal width, so +// the loop folds to a store (and a byte swap on the foreign order); a name per +// width would claim four §11 names to save nothing. +inline void table_cook_put( uint8_t * at, uint64_t value, int32_t width, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * i ) ); } + } + else + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * ( width - 1 - i ) ) ); } + } +} + +// A 128-bit store as two lanes: sixteen bytes, the low lane first in the +// little order and the high lane first — each lane big-endian — in the big +// order, exactly as a u64 is one lane of eight (docs/SPEC-TABLES.md §7.2). +inline void table_cook_put128( uint8_t * at, uint64_t lo, uint64_t hi, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) { table_cook_put( at, lo, 8, order ); table_cook_put( at + 8, hi, 8, order ); } + else { table_cook_put( at, hi, 8, order ); table_cook_put( at + 8, lo, 8, order ); } +} + +// A buffer piece: the USED bytes and nothing else. The tail is already zero — +// the whole extent was zeroed before any field was written — so this copies the +// used prefix and leaves the rest, which is what makes a string's unused tail a +// consequence of one memset rather than a rule per buffer. A used length past +// the buffer, or below zero, is a value no reader could have produced and it is +// clamped rather than trusted: this writes inside the caller's buffer on every +// input. +inline void table_cook_bytes( uint8_t * at, const void * source, int64_t used, int64_t capacity ) +{ + if ( used <= 0 ) { return; } + const int64_t n = used < capacity ? used : capacity; + memcpy( at, source, (size_t) n ); +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE +#define LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// ---- the cooked form's WRITE side for a POINTERED root (docs/SPEC-TABLES.md §7.6) ---- +// +// A pointered root's cook is the region of §7.2: every node the numbering +// reached (§3.1), once, at its own type's alignment, in index order, the root +// at offset zero. This is that region while it is being laid out and written — +// the tool's own Layout and Write, in one struct. +// +// The OFFSETS are one per node, the root's zero at position 0 and node index k +// at position k - 1, which is the directory's own order (§6.3); they are the +// one allocation the write makes beyond the numbering, and they go through the +// same pair. A measure needs no offsets and leaves the pointer NULL. +struct TableCookRegion +{ + const TableNumbering * numbering = NULL; // node -> index, from the walk that placed it + int64_t * offsets = NULL; // index - 1 -> the node's region offset; NULL while measuring + int64_t count = 0; // nodes, the root included + int64_t bytes = 0; // the data part's length, rounded to align + int64_t align = 0; // the region's alignment: the nodes' greatest, never below eight + uint8_t * base = NULL; // where the data part is being written; NULL while measuring +}; + +// A reference slot: the SELF-RELATIVE delta from the slot's own address to the +// node's start (§6.3), and zero for null. The node is found by the address the +// numbering keyed it under, which is the same address the walk resolved through +// the same context — so a reference the numbering does not carry is a slot the +// walk never reached (a counted array's slot past its count, an absent +// optional's value) holding a node the region will not hold, and it is refused +// rather than written as a delta to nowhere. +inline bool table_cook_ref( const TableCookRegion & region, uint8_t * at, const void * pointee, TableByteOrder order ) +{ + if ( pointee == NULL ) { table_cook_put( at, 0, 8, order ); return true; } + uint64_t index = 0; + if ( !TableNumberingIndex( *region.numbering, pointee, index ) ) { return false; } + if ( index == 0 || index > (uint64_t) region.count ) { return false; } + const int64_t delta = region.offsets[index - 1] - (int64_t) ( at - region.base ); + table_cook_put( at, (uint64_t) delta, 8, order ); + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// table Sample — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Sample { + int32_t v = 0; +}; + +// table Row — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Row { + TableList items; // Sample: the element array, empty until an Add + int32_t label = 0; +}; + +// table Sheet — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Sheet { + TableList rows; // Row: the element array, empty until an Add + TableRef pinned; // *Row — null until assigned +}; + +// table Item — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Item { + int32_t count = 0; +}; + +// table SquadRosterEntry — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct SquadRosterEntry { + uint8_t key = 0; + Item value; +}; + +// table Squad — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Squad { + TableMap roster; // map[uint8]Item — the sorted entry array, empty until an insert + int32_t name = 0; +}; + +// table Army — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Army { + TableList squads; // Squad: the element array, empty until an Add + int32_t after = 0; +}; + +// table Deck — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Deck { + Row hands[3]; // used count beside it; count in [0, 3] + int32_t hands_count = 0; + int32_t after = 0; +}; + +// ---- prefill: the declared defaults, in place (docs/SPEC-TABLES.md) ---- + +inline void SampleReset( Sample & value ); +inline void RowReset( Row & value ); +inline void SheetReset( Sheet & value ); +inline void ItemReset( Item & value ); +inline void SquadRosterEntryReset( SquadRosterEntry & value ); +inline void SquadReset( Squad & value ); +inline void ArmyReset( Army & value ); +inline void DeckReset( Deck & value ); + +inline void SampleReset( Sample & value ) +{ + value.v = 0; +} + +inline void RowReset( Row & value ) +{ + value.items.elements.value = 0; // Sample: empty + value.items.count = 0; + value.items.padding = 0; + value.label = 0; +} + +inline void SheetReset( Sheet & value ) +{ + value.rows.elements.value = 0; // Row: empty + value.rows.count = 0; + value.rows.padding = 0; + value.pinned.value = 0; // *Row — null +} + +inline void ItemReset( Item & value ) +{ + value.count = 0; +} + +inline void SquadRosterEntryReset( SquadRosterEntry & value ) +{ + value.key = 0; + ItemReset( value.value ); +} + +inline void SquadReset( Squad & value ) +{ + value.roster.entries.value = 0; // map[uint8]Item: empty + value.roster.count = 0; + value.roster.padding = 0; + value.name = 0; +} + +inline void ArmyReset( Army & value ) +{ + value.squads.elements.value = 0; // Squad: empty + value.squads.count = 0; + value.squads.padding = 0; + value.after = 0; +} + +inline void DeckReset( Deck & value ) +{ + RowReset( value.hands[0] ); + for ( int32_t i = 1; i < 3; i++ ) { value.hands[i] = value.hands[0]; } + value.hands_count = 0; + value.after = 0; +} + +// ---- the arena's reset hook (docs/SPEC-TABLES.md §6) ---- +// +// TableWorker::Alloc is a template and cannot name a member's Reset, so +// the arena reaches it through this overload set by argument-dependent +// lookup. It is how a node born in raw arena storage comes to hold the +// declared defaults without value-initialising the whole aggregate. + +inline void TableReset( Sample & value ) { SampleReset( value ); } +inline void TableReset( Row & value ) { RowReset( value ); } +inline void TableReset( Sheet & value ) { SheetReset( value ); } +inline void TableReset( Item & value ) { ItemReset( value ); } +inline void TableReset( SquadRosterEntry & value ) { SquadRosterEntryReset( value ); } +inline void TableReset( Squad & value ) { SquadReset( value ); } +inline void TableReset( Army & value ) { ArmyReset( value ); } +inline void TableReset( Deck & value ) { DeckReset( value ); } + +// ---- pointer targets: allocation and resolution (docs/SPEC-TABLES.md §2) ---- +// +// A reference resolves differently in the two forms, and the CONTEXT says +// which: in the arena it is an offset; in a region it is a self-relative +// delta, so the const deref below is one add and needs no base pointer. + +// Row is a pointer target. +inline const Row * RowAt( const TableRef & ref ) // the const form's hot path: one add, no base +{ + return ref.value != 0 ? (const Row *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline Row * RowAt( TableRef & ref ) +{ + return ref.value != 0 ? (Row *) ( (uint8_t *) &ref + ref.value ) : NULL; +} +inline const Row * RowAt( const TableRegionCtx &, const TableRef & ref ) { return RowAt( ref ); } +inline const Row * RowAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const Row *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +// while the builder is mutable, resolve against the arena itself +inline Row * RowAt( TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (Row *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} +inline const Row * RowAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const Row *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} +// allocate one Row in the arena; the slot holds the arena offset +inline Row * RowEmplace( TableWorker & worker, TableRef & slot ) +{ + TableSlot allocated = worker.Alloc(); + slot = allocated.ref; + return allocated.ptr; +} + +// ---- codecs: measure/save/load per closure member ---- + +inline int64_t SampleMeasureBody( TableIds & ids, const Sample & value ); +LISTDEMO_TABLE_INLINE bool SampleSaveBody( TableWriter & w, TableIds & ids, const Sample & value ); +LISTDEMO_TABLE_INLINE bool SampleLoadBody( TableReader & r, Sample & value ); +template inline int64_t RowMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Row & value ); +template inline bool RowSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Row & value ); +template inline bool RowSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Row & value ); +inline bool RowLoadBody( TableReader & r, const TableNodeMap & nodes, Row & value ); +template inline int64_t SheetMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Sheet & value ); +template inline bool SheetSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Sheet & value ); +template inline bool SheetSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Sheet & value ); +inline bool SheetLoadBody( TableReader & r, const TableNodeMap & nodes, Sheet & value ); +inline int64_t ItemMeasureBody( TableIds & ids, const Item & value ); +LISTDEMO_TABLE_INLINE bool ItemSaveBody( TableWriter & w, TableIds & ids, const Item & value ); +LISTDEMO_TABLE_INLINE bool ItemLoadBody( TableReader & r, Item & value ); +inline int64_t SquadRosterEntryMeasureBody( TableIds & ids, const SquadRosterEntry & value ); +LISTDEMO_TABLE_INLINE bool SquadRosterEntrySaveBody( TableWriter & w, TableIds & ids, const SquadRosterEntry & value ); +LISTDEMO_TABLE_INLINE bool SquadRosterEntryLoadBody( TableReader & r, SquadRosterEntry & value ); +template inline int64_t SquadMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Squad & value ); +template inline bool SquadSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Squad & value ); +template inline bool SquadSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Squad & value ); +inline bool SquadLoadBody( TableReader & r, const TableNodeMap & nodes, Squad & value ); +template inline int64_t ArmyMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Army & value ); +template inline bool ArmySaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Army & value ); +template inline bool ArmySaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Army & value ); +inline bool ArmyLoadBody( TableReader & r, const TableNodeMap & nodes, Army & value ); +template inline int64_t DeckMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Deck & value ); +template inline bool DeckSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Deck & value ); +template inline bool DeckSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Deck & value ); +inline bool DeckLoadBody( TableReader & r, const TableNodeMap & nodes, Deck & value ); + +// ---- pointer-graph walkers: number (measure/save), pack (Lock) ---- + +template inline bool RowNumber( const Ctx & ctx, TableNumbering & numbering, const Row & value ); +template inline int64_t RowPackMeasure( const Ctx & ctx, TablePackMap & seen, const Row & value ); +template inline bool RowPack( const Ctx & ctx, TablePackMap & seen, const Row & src, Row & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool SheetNumber( const Ctx & ctx, TableNumbering & numbering, const Sheet & value ); +template inline int64_t SheetPackMeasure( const Ctx & ctx, TablePackMap & seen, const Sheet & value ); +template inline bool SheetPack( const Ctx & ctx, TablePackMap & seen, const Sheet & src, Sheet & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool SquadNumber( const Ctx & ctx, TableNumbering & numbering, const Squad & value ); +template inline int64_t SquadPackMeasure( const Ctx & ctx, TablePackMap & seen, const Squad & value ); +template inline bool SquadPack( const Ctx & ctx, TablePackMap & seen, const Squad & src, Squad & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool ArmyNumber( const Ctx & ctx, TableNumbering & numbering, const Army & value ); +template inline int64_t ArmyPackMeasure( const Ctx & ctx, TablePackMap & seen, const Army & value ); +template inline bool ArmyPack( const Ctx & ctx, TablePackMap & seen, const Army & src, Army & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool DeckNumber( const Ctx & ctx, TableNumbering & numbering, const Deck & value ); +template inline int64_t DeckPackMeasure( const Ctx & ctx, TablePackMap & seen, const Deck & value ); +template inline bool DeckPack( const Ctx & ctx, TablePackMap & seen, const Deck & src, Deck & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +// ---- the numbering's bridge to each member's codec (docs/SPEC-TABLES.md §3.1) ---- + +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Row & value ) { return RowMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Row & value ) { return RowSaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Sheet & value ) { return SheetMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Sheet & value ) { return SheetSaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Squad & value ) { return SquadMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Squad & value ) { return SquadSaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Army & value ) { return ArmyMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Army & value ) { return ArmySaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Deck & value ) { return DeckMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Deck & value ) { return DeckSaveBody( ctx, numbering, w, ids, value ); } + +// ---- SquadRosterEntry: the order, the key and the value (docs/SPEC-TABLES.md §2.8) ---- +// +// The four overloads the map runtime's templates reach by argument-dependent +// lookup. Nothing outside this file names them. +static_assert( alignof( SquadRosterEntry ) <= kTableAlign, "a map entry's alignment must fit the arena's" ); + +inline int TableEntryOrder( const SquadRosterEntry & a, const SquadRosterEntry & b ) +{ + return TableKeyOrder( (uint64_t) a.key, (uint64_t) b.key ); // integers compare by VALUE, unsigned here +} +inline int TableEntryOrder( const SquadRosterEntry & entry, uint8_t key ) +{ + return TableKeyOrder( (uint64_t) entry.key, (uint64_t) key ); +} +inline uint8_t TableEntryKey( const SquadRosterEntry & entry ) { return entry.key; } +inline void TableEntrySetKey( SquadRosterEntry & entry, uint8_t key ) { entry.key = key; } +inline const Item * TableEntryFound( const SquadRosterEntry * entry ) { return entry != NULL ? &entry->value : NULL; } +inline Item * TableEntryValue( SquadRosterEntry * entry ) { return &entry->value; } +struct SquadRosterEntryEach { uint8_t key; decltype( TableEntryValue( (SquadRosterEntry *) NULL ) ) value; }; +inline SquadRosterEntryEach TableEntryEach( SquadRosterEntry * entry ) { return SquadRosterEntryEach{ TableEntryKey( *entry ), TableEntryValue( entry ) }; } +inline void TableResetMapValue( SquadRosterEntry & value ) +{ + ItemReset( value.value ); +} + +// SquadRosterEntryReadKey: the key, before the slot is chosen (docs/SPEC-TABLES.md §2.8). +// Field order inside a body is not contractual (§3), so this scans rather +// than assuming a position — and this implementation writes the key first, +// so on any wire it wrote the scan ends at the first header. +struct SquadRosterEntryKeyRead +{ + uint8_t key; + bool found; // the body carried the key's id + bool kind_bad; // it carried it under another kind: the MAP's event + bool over; // longer than this reader's bound: the ENTRY is dropped + bool malformed; // the entry's framing gave out +}; + +inline SquadRosterEntryKeyRead SquadRosterEntryReadKey( const uint8_t * body, int64_t length, const TableIdTable * ids ) +{ + SquadRosterEntryKeyRead out = { 0, false, false, false, false }; + TableReport scratch; // the scan's own framing damage is the MAP's, raised by the caller + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { out.malformed = true; return out; } + if ( field_ref == 0 ) { return out; } // the terminator: no key field is the key's DEFAULT + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { out.malformed = true; return out; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { out.malformed = true; return out; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x3dc94a19365b10ecull ) // `key`, the ordinary hash of an ordinary name + { + out.kind_bad = field_kind != 6; // THE KEY KIND IS THE READER'S DECLARATION + out.found = !out.kind_bad; + if ( !out.kind_bad ) + { + if ( !r.has( 1 ) ) { out.malformed = true; return out; } + out.key = (uint8_t) r.get8(); + continue; // the LAST occurrence is the one §3 keeps + } + } + if ( !r.skip( field_kind ) ) { out.malformed = true; return out; } + } +} + +inline int64_t SampleMeasureBody( TableIds & ids, const Sample & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.v != 0 ) { bytes += TableLebBytes( ids.ref( 0xaf63eb4c86020609ull, 23 ) ) + 1 + 4; } // v + return bytes; +} + +inline int64_t SampleMeasure( const Sample & value ) +{ + TableIds ids; + const int64_t body = SampleMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool SampleSaveBody( TableWriter & w, TableIds & ids, const Sample & value ) +{ + if ( value.v != 0 ) + { + w.putleb( ids.ref( 0xaf63eb4c86020609ull, 23 ) ); w.put8( 4 ); // v + w.put32( uint32_t( value.v ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t SampleSave( const Sample & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !SampleSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == SampleMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool SampleLoadBody( TableReader & r, Sample & value ) +{ + SampleReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xaf63eb4c86020609ull: // v + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.v = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict SampleLoadVerdict( Sample & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + SampleReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + SampleReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !SampleLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool SampleLoad( Sample & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return SampleLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t SampleMeasureMessage( const Sample & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = SampleMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t SampleSaveMessage( const Sample & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !SampleSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == SampleMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool SampleLoadMessage( Sample & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + SampleReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return SampleLoadBody( r, value ); +} + +template +inline int64_t RowMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Row & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // items: a kind 14 array of kind 13 elements, INDEX order (§2.9) + TableListCursor cursor_items = TableListElements( ctx, value.items ); + if ( !cursor_items.ok ) { return -1; } // the slot and the head disagree + if ( cursor_items.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_items = ids.ref( 0x3e7884bf4f412c6full, 6 ); + int64_t body_items = 0; + body_items += 1 + TableLebBytes( (uint64_t) ( cursor_items.count ) ); // the element kind byte and the count + for ( int32_t elem_i_items = 0; elem_i_items < cursor_items.count; elem_i_items++ ) + { + const int64_t elem_bytes_items = SampleMeasureBody( ids, cursor_items[elem_i_items] ); + if ( elem_bytes_items < 0 ) { return -1; } + body_items += TableLebBytes( (uint64_t) ( elem_bytes_items ) ) + ( elem_bytes_items ); + } + bytes += TableLebBytes( ref_items ) + 1 + TableLebBytes( (uint64_t) ( body_items ) ) + ( body_items ); + } + } + if ( value.label != 0 ) { bytes += TableLebBytes( ids.ref( 0x39f7fcec8fcb623dull, 22 ) ) + 1 + 4; } // label + return bytes; +} + +template +inline bool RowSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Row & value ) +{ + (void) ctx; (void) numbering; + { + TableListCursor cursor_items = TableListElements( ctx, value.items ); // items + if ( !cursor_items.ok ) { return false; } + if ( cursor_items.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_items = ids.ref( 0x3e7884bf4f412c6full, 6 ); + int64_t body_items = 0; + body_items += 1 + TableLebBytes( (uint64_t) ( cursor_items.count ) ); // the element kind byte and the count + for ( int32_t elem_i_items = 0; elem_i_items < cursor_items.count; elem_i_items++ ) + { + const int64_t elem_bytes_items = SampleMeasureBody( ids, cursor_items[elem_i_items] ); + if ( elem_bytes_items < 0 ) { return false; } + body_items += TableLebBytes( (uint64_t) ( elem_bytes_items ) ) + ( elem_bytes_items ); + } + w.putleb( ref_items ); w.put8( 14 ); w.putleb( (uint64_t) body_items ); // items + w.put8( 13 ); w.putleb( (uint64_t) ( cursor_items.count ) ); + for ( int32_t elem_i_items = 0; elem_i_items < cursor_items.count; elem_i_items++ ) + { + { + const int64_t elem_len_items = SampleMeasureBody( ids, cursor_items[elem_i_items] ); + if ( elem_len_items < 0 ) return false; + w.putleb( (uint64_t) elem_len_items ); + if ( !SampleSaveBody( w, ids, cursor_items[elem_i_items] ) ) return false; + } + } + } + } + if ( value.label != 0 ) + { + w.putleb( ids.ref( 0x39f7fcec8fcb623dull, 22 ) ); w.put8( 4 ); // label + w.put32( uint32_t( value.label ) ); + } + return !w.overflow; +} + +template +inline bool RowSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Row & value ) +{ + if ( !RowSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool RowLoadBody( TableReader & r, const TableNodeMap & nodes, Row & value ) +{ + (void) nodes; + RowReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x3e7884bf4f412c6full: // items + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.items, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Sample * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + uint64_t elem_len_items = 0; + if ( !sub.getleb( elem_len_items ) || !sub.room( elem_len_items ) ) { r.report->malformed = true; break; } + { + TableReader elem_items( sub.buffer + sub.offset, (int64_t) elem_len_items, r.report, r.ids ); + SampleLoadBody( elem_items, ( *slot ) ); + } + sub.offset += (int64_t) elem_len_items; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x39f7fcec8fcb623dull: // label + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.label = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +template +inline int64_t SheetMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Sheet & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // rows: a kind 14 array of kind 13 elements, INDEX order (§2.9) + TableListCursor cursor_rows = TableListElements( ctx, value.rows ); + if ( !cursor_rows.ok ) { return -1; } // the slot and the head disagree + if ( cursor_rows.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_rows = ids.ref( 0xa3a7061ff10a8138ull, 27 ); + int64_t body_rows = 0; + body_rows += 1 + TableLebBytes( (uint64_t) ( cursor_rows.count ) ); // the element kind byte and the count + for ( int32_t elem_i_rows = 0; elem_i_rows < cursor_rows.count; elem_i_rows++ ) + { + const int64_t elem_bytes_rows = RowMeasureBody( ctx, numbering, ids, cursor_rows[elem_i_rows] ); + if ( elem_bytes_rows < 0 ) { return -1; } + body_rows += TableLebBytes( (uint64_t) ( elem_bytes_rows ) ) + ( elem_bytes_rows ); + } + bytes += TableLebBytes( ref_rows ) + 1 + TableLebBytes( (uint64_t) ( body_rows ) ) + ( body_rows ); + } + } + { + const Row * pointee_pinned = RowAt( ctx, value.pinned ); // *Row + // A POINTER RIDES AS A NODE INDEX (docs/SPEC-TABLES.md §3.1): the + // header and the index and nothing below it, because the pointee's + // body is in the node table and not here. NULL IS ELIDED — absence + // and null are one value — and a non-null pointer ALWAYS rides, even + // when its node's body is entirely default. + if ( pointee_pinned != NULL ) + { + uint64_t index_pinned = 0; + if ( !TableNumberingIndex( numbering, (const void *) pointee_pinned, index_pinned ) ) { return -1; } + bytes += TableLebBytes( ids.ref( 0x5f82477707ad620full, 28 ) ) + 1 + TableLebBytes( index_pinned ); + } + } + return bytes; +} + +template +inline bool SheetSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Sheet & value ) +{ + { + TableListCursor cursor_rows = TableListElements( ctx, value.rows ); // rows + if ( !cursor_rows.ok ) { return false; } + if ( cursor_rows.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_rows = ids.ref( 0xa3a7061ff10a8138ull, 27 ); + int64_t body_rows = 0; + body_rows += 1 + TableLebBytes( (uint64_t) ( cursor_rows.count ) ); // the element kind byte and the count + for ( int32_t elem_i_rows = 0; elem_i_rows < cursor_rows.count; elem_i_rows++ ) + { + const int64_t elem_bytes_rows = RowMeasureBody( ctx, numbering, ids, cursor_rows[elem_i_rows] ); + if ( elem_bytes_rows < 0 ) { return false; } + body_rows += TableLebBytes( (uint64_t) ( elem_bytes_rows ) ) + ( elem_bytes_rows ); + } + w.putleb( ref_rows ); w.put8( 14 ); w.putleb( (uint64_t) body_rows ); // rows + w.put8( 13 ); w.putleb( (uint64_t) ( cursor_rows.count ) ); + for ( int32_t elem_i_rows = 0; elem_i_rows < cursor_rows.count; elem_i_rows++ ) + { + { + const int64_t elem_len_rows = RowMeasureBody( ctx, numbering, ids, cursor_rows[elem_i_rows] ); + if ( elem_len_rows < 0 ) return false; + w.putleb( (uint64_t) elem_len_rows ); + if ( !RowSaveBody( ctx, numbering, w, ids, cursor_rows[elem_i_rows] ) ) return false; + } + } + } + } + { + const Row * pointee_pinned = RowAt( ctx, value.pinned ); // *Row + if ( pointee_pinned != NULL ) + { + uint64_t index_pinned = 0; + if ( !TableNumberingIndex( numbering, (const void *) pointee_pinned, index_pinned ) ) { return false; } + w.putleb( ids.ref( 0x5f82477707ad620full, 28 ) ); w.put8( 17 ); // pinned — a NODE INDEX into the flat node table + w.putleb( index_pinned ); + } + } + return !w.overflow; +} + +template +inline bool SheetSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Sheet & value ) +{ + if ( !SheetSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool SheetLoadBody( TableReader & r, const TableNodeMap & nodes, Sheet & value ) +{ + SheetReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xa3a7061ff10a8138ull: // rows + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.rows, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Row * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + uint64_t elem_len_rows = 0; + if ( !sub.getleb( elem_len_rows ) || !sub.room( elem_len_rows ) ) { r.report->malformed = true; break; } + { + TableReader elem_rows( sub.buffer + sub.offset, (int64_t) elem_len_rows, r.report, r.ids ); + RowLoadBody( elem_rows, nodes, ( *slot ) ); + } + sub.offset += (int64_t) elem_len_rows; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x5f82477707ad620full: // pinned + { + if ( kind != 17 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + // A POINTER FIELD'S PAYLOAD IS A NUMBER (docs/SPEC-TABLES.md §3.1): it is + // bounds-checked and resolved through the numbering, never FOLLOWED, so + // there is no traversal here and therefore no traversal bound. + { + uint64_t node_index = 0; + if ( !r.getleb( node_index ) ) { r.report->malformed = true; return false; } + TableNodeResolve( nodes, value.pinned, node_index, 0xa013e119fec906fbull, r.report ); // *Row + } + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +inline int64_t ItemMeasureBody( TableIds & ids, const Item & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.count != 0 ) { bytes += TableLebBytes( ids.ref( 0xb1e5e28e4479a274ull, 11 ) ) + 1 + 4; } // count + return bytes; +} + +inline int64_t ItemMeasure( const Item & value ) +{ + TableIds ids; + const int64_t body = ItemMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool ItemSaveBody( TableWriter & w, TableIds & ids, const Item & value ) +{ + if ( value.count != 0 ) + { + w.putleb( ids.ref( 0xb1e5e28e4479a274ull, 11 ) ); w.put8( 4 ); // count + w.put32( uint32_t( value.count ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t ItemSave( const Item & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !ItemSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == ItemMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool ItemLoadBody( TableReader & r, Item & value ) +{ + ItemReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xb1e5e28e4479a274ull: // count + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.count = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict ItemLoadVerdict( Item & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + ItemReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + ItemReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !ItemLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool ItemLoad( Item & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return ItemLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t ItemMeasureMessage( const Item & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = ItemMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t ItemSaveMessage( const Item & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !ItemSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == ItemMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool ItemLoadMessage( Item & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + ItemReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return ItemLoadBody( r, value ); +} + +inline int64_t SquadRosterEntryMeasureBody( TableIds & ids, const SquadRosterEntry & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.key != 0 ) { bytes += TableLebBytes( ids.ref( 0x3dc94a19365b10ecull, 31 ) ) + 1 + 1; } // key + { + const int32_t mark_value = ids.count; + const uint64_t ref_value = ids.ref( 0x7ce4fd9430e80ceaull, 32 ); + const int64_t body_value = ItemMeasureBody( ids, value.value ); + if ( body_value < 0 ) { return -1; } + if ( body_value > 1 ) { bytes += TableLebBytes( ref_value ) + 1 + TableLebBytes( (uint64_t) ( body_value ) ) + ( body_value ); } // value + else { ids.truncate( mark_value ); } // an all-default nested table elides, and costs no entry + } + return bytes; +} + +LISTDEMO_TABLE_INLINE bool SquadRosterEntrySaveBody( TableWriter & w, TableIds & ids, const SquadRosterEntry & value ) +{ + if ( value.key != 0 ) + { + w.putleb( ids.ref( 0x3dc94a19365b10ecull, 31 ) ); w.put8( 6 ); // key + w.put8( uint8_t( value.key ) ); + } + { + const int32_t mark_value = ids.count; + const uint64_t ref_value = ids.ref( 0x7ce4fd9430e80ceaull, 32 ); + const int64_t body_value = ItemMeasureBody( ids, value.value ); + if ( body_value < 0 ) return false; // storage invariant, refused as measure refuses it + if ( body_value > 1 ) // all-default nested elides + { + w.putleb( ref_value ); w.put8( 13 ); w.putleb( (uint64_t) body_value ); // value + if ( !ItemSaveBody( w, ids, value.value ) ) return false; + } + else { ids.truncate( mark_value ); } + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +LISTDEMO_TABLE_INLINE bool SquadRosterEntryLoadBody( TableReader & r, SquadRosterEntry & value ) +{ + SquadRosterEntryReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x3dc94a19365b10ecull: // key + { + if ( kind != 6 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t decoded_v = uint8_t( r.get8( ) ); + value.key = decoded_v; + break; + } + case 0x7ce4fd9430e80ceaull: // value + { + if ( kind != 13 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + { + TableReader sub( r.buffer + r.offset, (int64_t) body_len, r.report, r.ids ); + ItemLoadBody( sub, value.value ); + if ( sub.offset != sub.size ) + { + r.report->malformed = true; + ItemReset( value.value ); + } + } + r.offset += (int64_t) body_len; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +template +inline int64_t SquadMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Squad & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // roster: a kind 14 array of kind 13 elements, ASCENDING (§2.8) + TableMapCursor order_roster = TableMapOrder( ctx, value.roster ); + if ( !order_roster.ok ) { return -1; } // the sort could not run + if ( order_roster.count > 0 ) + { + const uint64_t ref_roster = ids.ref( 0x1c84390d304f4f42ull, 29 ); + int64_t body_roster = 1 + TableLebBytes( (uint64_t) order_roster.count ); // the element kind byte and the count + for ( int32_t i = 0; i < order_roster.count; i++ ) + { + const int64_t elem_roster = SquadRosterEntryMeasureBody( ids, *order_roster[i] ); + if ( elem_roster < 0 ) { TableMapRelease( order_roster ); return -1; } + body_roster += TableLebBytes( (uint64_t) ( elem_roster ) ) + ( elem_roster ); // BUT THE ENTRY ALWAYS RIDES: identity here is the key + } + bytes += TableLebBytes( ref_roster ) + 1 + TableLebBytes( (uint64_t) ( body_roster ) ) + ( body_roster ); + } + TableMapRelease( order_roster ); + } + if ( value.name != 0 ) { bytes += TableLebBytes( ids.ref( 0xc4bcadba8e631b86ull, 30 ) ) + 1 + 4; } // name + return bytes; +} + +template +inline bool SquadSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Squad & value ) +{ + (void) ctx; (void) numbering; + { + TableMapCursor order_roster = TableMapOrder( ctx, value.roster ); // roster + if ( !order_roster.ok ) { return false; } + if ( order_roster.count > 0 ) // an EMPTY map elides, the by-value rule (§3) + { + const uint64_t ref_roster = ids.ref( 0x1c84390d304f4f42ull, 29 ); + int64_t body_roster = 1 + TableLebBytes( (uint64_t) order_roster.count ); + for ( int32_t i = 0; i < order_roster.count; i++ ) + { + const int64_t elem_roster = SquadRosterEntryMeasureBody( ids, *order_roster[i] ); + if ( elem_roster < 0 ) { TableMapRelease( order_roster ); return false; } + body_roster += TableLebBytes( (uint64_t) ( elem_roster ) ) + ( elem_roster ); + } + w.putleb( ref_roster ); w.put8( 14 ); w.putleb( (uint64_t) body_roster ); + w.put8( 13 ); w.putleb( (uint64_t) order_roster.count ); + for ( int32_t i = 0; i < order_roster.count; i++ ) + { + const int64_t elem_len_roster = SquadRosterEntryMeasureBody( ids, *order_roster[i] ); + if ( elem_len_roster < 0 ) { TableMapRelease( order_roster ); return false; } + w.putleb( (uint64_t) elem_len_roster ); + if ( !SquadRosterEntrySaveBody( w, ids, *order_roster[i] ) ) { TableMapRelease( order_roster ); return false; } + } + } + TableMapRelease( order_roster ); + } + if ( value.name != 0 ) + { + w.putleb( ids.ref( 0xc4bcadba8e631b86ull, 30 ) ); w.put8( 4 ); // name + w.put32( uint32_t( value.name ) ); + } + return !w.overflow; +} + +template +inline bool SquadSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Squad & value ) +{ + if ( !SquadSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool SquadLoadBody( TableReader & r, const TableNodeMap & nodes, Squad & value ) +{ + (void) nodes; + SquadReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x1c84390d304f4f42ull: // roster + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + if ( !r.getleb( count ) ) { r.report->malformed = true; r.offset = body_end; break; } + // A MAP HEADER WHOSE ELEMENT KIND IS NOT 13 is the ordinary array + // kind mismatch of §4, and nothing about a map is special-cased + if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + TableMapFill fill = TableMapFillBegin( nodes, value.roster, (uint32_t) count ); + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + uint8_t last_key = 0; + bool landed = false; + for ( uint64_t i = 0; i < count; i++ ) + { + uint64_t elem_len = 0; + if ( !sub.getleb( elem_len ) || !sub.room( elem_len ) ) { r.report->malformed = true; break; } + const uint8_t * elem_body = sub.buffer + sub.offset; + sub.offset += (int64_t) elem_len; + SquadRosterEntryKeyRead read = SquadRosterEntryReadKey( elem_body, (int64_t) elem_len, r.ids ); + // THE KEY KIND IS CHECKED FIRST: a key read under another kind + // desynchronizes the rest of the scan, and the honest answer to a + // body whose key is not this reader's kind is the KIND, not the + // framing damage that follows from it. + if ( read.kind_bad ) + { + // A MAP WITH HALF ITS KEYS IS NOT A MAP (§2.8): the map resets + // to EMPTY, ONE kind_mismatch is counted for it, and the rest + // is skipped. Events counted inside earlier entries stand. + r.report->kind_mismatch++; + TableMapFillReset( fill ); + break; + } + if ( read.malformed ) { r.report->malformed = true; break; } + if ( read.over ) { r.report->clamped++; continue; } // skipped by its L, one count per entry + const int order = landed ? TableKeyOrder( (uint64_t) last_key, (uint64_t) read.key ) : -1; + if ( order > 0 ) + { + // DESCENDING: not a body any conforming writer produced. The map + // keeps the ascending prefix it has, the rest skips by the map's + // L, and the PARENT reads on past the field's length (§4). + r.report->malformed = true; + break; + } + SquadRosterEntry * slot = NULL; + if ( order == 0 ) + { + // EQUAL: a DUPLICATE. The slot that entry took is reset to the + // entry's defaults by the decode below, so LAST WINS WHOLE and an + // elided field of the repeat reads as its default. The map's + // count excludes it. + slot = TableMapFillLast( fill ); + r.report->duplicate++; + } + else + { + slot = TableMapFillNext( fill ); // ASCENDING: the next slot + } + if ( slot == NULL ) { r.report->malformed = true; break; } + { + TableReader elem( elem_body, (int64_t) elem_len, r.report, r.ids ); + SquadRosterEntryLoadBody( elem, *slot ); + } + last_key = read.key; // the WIRE keys of the entries that LAND + landed = true; + } + TableMapFillEnd( fill ); + } + r.offset = body_end; // the remaining entries skip by the map's L + break; + } + case 0xc4bcadba8e631b86ull: // name + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.name = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +template +inline int64_t ArmyMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Army & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // squads: a kind 14 array of kind 13 elements, INDEX order (§2.9) + TableListCursor cursor_squads = TableListElements( ctx, value.squads ); + if ( !cursor_squads.ok ) { return -1; } // the slot and the head disagree + if ( cursor_squads.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_squads = ids.ref( 0x7848019b0c02a926ull, 4 ); + int64_t body_squads = 0; + body_squads += 1 + TableLebBytes( (uint64_t) ( cursor_squads.count ) ); // the element kind byte and the count + for ( int32_t elem_i_squads = 0; elem_i_squads < cursor_squads.count; elem_i_squads++ ) + { + const int64_t elem_bytes_squads = SquadMeasureBody( ctx, numbering, ids, cursor_squads[elem_i_squads] ); + if ( elem_bytes_squads < 0 ) { return -1; } + body_squads += TableLebBytes( (uint64_t) ( elem_bytes_squads ) ) + ( elem_bytes_squads ); + } + bytes += TableLebBytes( ref_squads ) + 1 + TableLebBytes( (uint64_t) ( body_squads ) ) + ( body_squads ); + } + } + if ( value.after != 0 ) { bytes += TableLebBytes( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ) + 1 + 4; } // after + return bytes; +} + +template +inline bool ArmySaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Army & value ) +{ + { + TableListCursor cursor_squads = TableListElements( ctx, value.squads ); // squads + if ( !cursor_squads.ok ) { return false; } + if ( cursor_squads.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_squads = ids.ref( 0x7848019b0c02a926ull, 4 ); + int64_t body_squads = 0; + body_squads += 1 + TableLebBytes( (uint64_t) ( cursor_squads.count ) ); // the element kind byte and the count + for ( int32_t elem_i_squads = 0; elem_i_squads < cursor_squads.count; elem_i_squads++ ) + { + const int64_t elem_bytes_squads = SquadMeasureBody( ctx, numbering, ids, cursor_squads[elem_i_squads] ); + if ( elem_bytes_squads < 0 ) { return false; } + body_squads += TableLebBytes( (uint64_t) ( elem_bytes_squads ) ) + ( elem_bytes_squads ); + } + w.putleb( ref_squads ); w.put8( 14 ); w.putleb( (uint64_t) body_squads ); // squads + w.put8( 13 ); w.putleb( (uint64_t) ( cursor_squads.count ) ); + for ( int32_t elem_i_squads = 0; elem_i_squads < cursor_squads.count; elem_i_squads++ ) + { + { + const int64_t elem_len_squads = SquadMeasureBody( ctx, numbering, ids, cursor_squads[elem_i_squads] ); + if ( elem_len_squads < 0 ) return false; + w.putleb( (uint64_t) elem_len_squads ); + if ( !SquadSaveBody( ctx, numbering, w, ids, cursor_squads[elem_i_squads] ) ) return false; + } + } + } + } + if ( value.after != 0 ) + { + w.putleb( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ); w.put8( 4 ); // after + w.put32( uint32_t( value.after ) ); + } + return !w.overflow; +} + +template +inline bool ArmySaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Army & value ) +{ + if ( !ArmySaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool ArmyLoadBody( TableReader & r, const TableNodeMap & nodes, Army & value ) +{ + ArmyReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x7848019b0c02a926ull: // squads + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.squads, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Squad * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + uint64_t elem_len_squads = 0; + if ( !sub.getleb( elem_len_squads ) || !sub.room( elem_len_squads ) ) { r.report->malformed = true; break; } + { + TableReader elem_squads( sub.buffer + sub.offset, (int64_t) elem_len_squads, r.report, r.ids ); + SquadLoadBody( elem_squads, nodes, ( *slot ) ); + } + sub.offset += (int64_t) elem_len_squads; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xbf82010f6f71eae9ull: // after + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.after = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +template +inline int64_t DeckMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Deck & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.hands_count < 0 || value.hands_count > 3 ) { return -1; } // storage invariant + if ( value.hands_count > 0 ) + { + const uint64_t ref_hands = ids.ref( 0x81b46a69304ee2c9ull, 9 ); + int64_t body_hands = 0; + body_hands += 1 + TableLebBytes( (uint64_t) ( value.hands_count ) ); // the element kind byte and the count + for ( int32_t elem_i = 0; elem_i < value.hands_count; elem_i++ ) + { + const int64_t elem_bytes = RowMeasureBody( ctx, numbering, ids, value.hands[elem_i] ); + if ( elem_bytes < 0 ) { return -1; } + body_hands += TableLebBytes( (uint64_t) ( elem_bytes ) ) + ( elem_bytes ); + } + bytes += TableLebBytes( ref_hands ) + 1 + TableLebBytes( (uint64_t) ( body_hands ) ) + ( body_hands ); // hands + } + if ( value.after != 0 ) { bytes += TableLebBytes( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ) + 1 + 4; } // after + return bytes; +} + +template +inline bool DeckSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Deck & value ) +{ + if ( value.hands_count < 0 || value.hands_count > 3 ) { return false; } // storage invariant + if ( value.hands_count > 0 ) + { + const uint64_t ref_hands = ids.ref( 0x81b46a69304ee2c9ull, 9 ); + int64_t body_hands = 0; + body_hands += 1 + TableLebBytes( (uint64_t) ( value.hands_count ) ); // the element kind byte and the count + for ( int32_t elem_i = 0; elem_i < value.hands_count; elem_i++ ) + { + const int64_t elem_bytes = RowMeasureBody( ctx, numbering, ids, value.hands[elem_i] ); + if ( elem_bytes < 0 ) { return false; } + body_hands += TableLebBytes( (uint64_t) ( elem_bytes ) ) + ( elem_bytes ); + } + w.putleb( ref_hands ); w.put8( 14 ); w.putleb( (uint64_t) body_hands ); // hands + w.put8( 13 ); w.putleb( (uint64_t) ( value.hands_count ) ); + for ( int32_t elem_i = 0; elem_i < value.hands_count; elem_i++ ) + { + { + const int64_t elem_len = RowMeasureBody( ctx, numbering, ids, value.hands[elem_i] ); + if ( elem_len < 0 ) return false; + w.putleb( (uint64_t) elem_len ); + if ( !RowSaveBody( ctx, numbering, w, ids, value.hands[elem_i] ) ) return false; + } + } + } + if ( value.after != 0 ) + { + w.putleb( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ); w.put8( 4 ); // after + w.put32( uint32_t( value.after ) ); + } + return !w.overflow; +} + +template +inline bool DeckSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Deck & value ) +{ + if ( !DeckSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool DeckLoadBody( TableReader & r, const TableNodeMap & nodes, Deck & value ) +{ + DeckReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x81b46a69304ee2c9ull: // hands + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER — the element kind byte and the + // count, so fewer than two bytes — is INERT (§4): the field keeps the + // value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + // A DAMAGED COUNT stops the elements and nothing else: the field + // RODE, so an optional is still PRESENT (§2.3) — only a foreign + // ELEMENT KIND says the payload is not this array's at all. + if ( !counted_ok ) { r.report->malformed = true; } + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + uint64_t keep = count; + if ( keep > 3 ) { keep = 3; r.report->clamped++; } + // elements are BOUNDED by the field body: a count the length + // cannot cover keeps the decoded prefix, flags malformed, and + // the parent continues at the next field — following fields' + // bytes are never fabricated into elements + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + uint64_t decoded = 0; + for ( uint64_t i = 0; i < keep; i++ ) + { + uint64_t elem_len = 0; + if ( !sub.getleb( elem_len ) || !sub.room( elem_len ) ) { r.report->malformed = true; break; } + { + TableReader elem( sub.buffer + sub.offset, (int64_t) elem_len, r.report, r.ids ); + RowLoadBody( elem, nodes, value.hands[(int32_t) i] ); + } + sub.offset += (int64_t) elem_len; + decoded = i + 1; + } + value.hands_count = (int32_t) decoded; + } + } + r.offset = body_end; // excess elements and slack skip via the length + break; + } + case 0xbf82010f6f71eae9ull: // after + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.after = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// RowWireExtent: the extent Row's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool RowWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x3e7884bf4f412c6full && field_kind == 14 ) // items: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Sample ), (int64_t) alignof( Sample ), 13, 2, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// RowExtentAt: the node extent Row's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as RowExtentPack advances it (§2.8, §2.9). +template +inline bool RowExtentAt( const Ctx & ctx, const Row & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.items ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Sample ) - 1 ) & ~( (int64_t) alignof( Sample ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Sample ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t RowExtent( const Ctx & ctx, const Row & value ) +{ + int64_t at = 0; + if ( !RowExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// RowExtentPack: carve Row's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset RowExtentAt advances (§2.8, §2.9). +template +inline bool RowExtentPack( const Ctx & ctx, const Row & src, Row & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.items ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Sample ) - 1 ) & ~( (int64_t) alignof( Sample ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Sample ); + if ( at + bytes > capacity ) { return false; } + Sample * placed = (Sample *) ( extent + at ); + at += bytes; + dst.items.count = cursor.count; + dst.items.padding = 0; + dst.items.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.items.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Sample ) ); // trivially copyable, by construction + } + } + return true; +} + +// SheetWireExtent: the extent Sheet's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool SheetWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0xa3a7061ff10a8138ull && field_kind == 14 ) // rows: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Row ), (int64_t) alignof( Row ), 13, 2, &RowWireExtent, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// SheetExtentAt: the node extent Sheet's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as SheetExtentPack advances it (§2.8, §2.9). +template +inline bool SheetExtentAt( const Ctx & ctx, const Sheet & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.rows ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Row ) - 1 ) & ~( (int64_t) alignof( Row ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Row ); // the whole array FIRST + for ( int32_t i = 0; i < cursor.count; i++ ) // then, element by element in index order + { + if ( !RowExtentAt( ctx, cursor[i], at ) ) { return false; } + } + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t SheetExtent( const Ctx & ctx, const Sheet & value ) +{ + int64_t at = 0; + if ( !SheetExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// SheetExtentPack: carve Sheet's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset SheetExtentAt advances (§2.8, §2.9). +template +inline bool SheetExtentPack( const Ctx & ctx, const Sheet & src, Sheet & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.rows ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Row ) - 1 ) & ~( (int64_t) alignof( Row ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Row ); + if ( at + bytes > capacity ) { return false; } + Row * placed = (Row *) ( extent + at ); + at += bytes; + dst.rows.count = cursor.count; + dst.rows.padding = 0; + dst.rows.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.rows.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Row ) ); // trivially copyable, by construction + } + for ( int32_t i = 0; i < cursor.count; i++ ) + { + if ( !RowExtentPack( ctx, cursor[i], placed[i], extent, at, capacity ) ) { return false; } + } + } + return true; +} + +// SquadWireExtent: the extent Squad's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool SquadWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x1c84390d304f4f42ull && field_kind == 14 ) // roster + { + uint64_t map_len = 0; + if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } + const uint8_t * map_body = r.buffer + r.offset; + r.offset += (int64_t) map_len; + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( SquadRosterEntry ), (int64_t) alignof( SquadRosterEntry ), NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// SquadExtentAt: the node extent Squad's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as SquadExtentPack advances it (§2.8, §2.9). +template +inline bool SquadExtentAt( const Ctx & ctx, const Squad & value, int64_t & at ) +{ + { + TableMapCursor cursor = TableMapOrder( ctx, value.roster ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( SquadRosterEntry ) + at += (int64_t) cursor.count * (int64_t) sizeof( SquadRosterEntry ); // the whole array FIRST + TableMapRelease( cursor ); + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t SquadExtent( const Ctx & ctx, const Squad & value ) +{ + int64_t at = 0; + if ( !SquadExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// SquadExtentPack: carve Squad's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset SquadExtentAt advances (§2.8, §2.9). +template +inline bool SquadExtentPack( const Ctx & ctx, const Squad & src, Squad & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableMapCursor cursor = TableMapOrder( ctx, src.roster ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( SquadRosterEntry ); + if ( at + bytes > capacity ) { TableMapRelease( cursor ); return false; } + SquadRosterEntry * placed = (SquadRosterEntry *) ( extent + at ); + at += bytes; + dst.roster.count = cursor.count; + dst.roster.padding = 0; + dst.roster.entries.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.roster.entries ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) + { + memcpy( (void *) ( placed + i ), (const void *) cursor[i], sizeof( SquadRosterEntry ) ); // trivially copyable, by construction + } + TableMapRelease( cursor ); + } + return true; +} + +// ArmyWireExtent: the extent Army's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool ArmyWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x7848019b0c02a926ull && field_kind == 14 ) // squads: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Squad ), (int64_t) alignof( Squad ), 13, 2, &SquadWireExtent, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// ArmyExtentAt: the node extent Army's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as ArmyExtentPack advances it (§2.8, §2.9). +template +inline bool ArmyExtentAt( const Ctx & ctx, const Army & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.squads ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Squad ) - 1 ) & ~( (int64_t) alignof( Squad ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Squad ); // the whole array FIRST + for ( int32_t i = 0; i < cursor.count; i++ ) // then, element by element in index order + { + if ( !SquadExtentAt( ctx, cursor[i], at ) ) { return false; } + } + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t ArmyExtent( const Ctx & ctx, const Army & value ) +{ + int64_t at = 0; + if ( !ArmyExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// ArmyExtentPack: carve Army's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset ArmyExtentAt advances (§2.8, §2.9). +template +inline bool ArmyExtentPack( const Ctx & ctx, const Army & src, Army & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.squads ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Squad ) - 1 ) & ~( (int64_t) alignof( Squad ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Squad ); + if ( at + bytes > capacity ) { return false; } + Squad * placed = (Squad *) ( extent + at ); + at += bytes; + dst.squads.count = cursor.count; + dst.squads.padding = 0; + dst.squads.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.squads.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Squad ) ); // trivially copyable, by construction + } + for ( int32_t i = 0; i < cursor.count; i++ ) + { + if ( !SquadExtentPack( ctx, cursor[i], placed[i], extent, at, capacity ) ) { return false; } + } + } + return true; +} + +// DeckWireExtent: the extent Deck's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool DeckWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x81b46a69304ee2c9ull && field_kind == 14 ) // hands: a nesting that holds a list or a map + { + uint64_t nested_len = 0; + if ( !r.getleb( nested_len ) || !r.room( nested_len ) ) { return true; } + const uint8_t * nested_body = r.buffer + r.offset; + r.offset += (int64_t) nested_len; + if ( !TableWireExtentElements( nested_body, (int64_t) nested_len, at, &RowWireExtent, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// DeckExtentAt: the node extent Deck's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as DeckExtentPack advances it (§2.8, §2.9). +template +inline bool DeckExtentAt( const Ctx & ctx, const Deck & value, int64_t & at ) +{ + for ( int32_t i = 0; i < value.hands_count && i < 3; i++ ) // hands + { + if ( !RowExtentAt( ctx, value.hands[i], at ) ) { return false; } + } + for ( int32_t i = value.hands_count; i < 3; i++ ) // hands: the slots the walk does not reach (§7.6) + { + if ( !TableExtentUnreachedEmpty( RowExtent( ctx, value.hands[i] ) ) ) { return false; } + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t DeckExtent( const Ctx & ctx, const Deck & value ) +{ + int64_t at = 0; + if ( !DeckExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// DeckExtentPack: carve Deck's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset DeckExtentAt advances (§2.8, §2.9). +template +inline bool DeckExtentPack( const Ctx & ctx, const Deck & src, Deck & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + for ( int32_t i = 0; i < src.hands_count && i < 3; i++ ) // hands + { + if ( !RowExtentPack( ctx, src.hands[i], dst.hands[i], extent, at, capacity ) ) { return false; } + } + for ( int32_t i = src.hands_count; i < 3; i++ ) // hands: the slots the walk does not reach (§7.6) + { + if ( !TableExtentUnreachedEmpty( RowExtent( ctx, src.hands[i] ) ) ) { return false; } + } + return true; +} + +// ---- Squad.roster: the builder's five and the side index (§2.8) ---- + +// INSERT: the key is copied, the value is handed back at its defaults to +// fill. A DUPLICATE key REPLACES — the value is reset and the same entry +// handed back, key and address unchanged — so a caller that wants to know +// writes Find first. NULL is NOT INSERTED: a key longer than the bound, +// because a truncated key would be a merged entry, and an arena that +// cannot carve another segment, alike. +inline Item * SquadRosterInsert( TableWorker & worker, TableMap & map, uint8_t key ) +{ + if ( worker.arena == NULL ) { return NULL; } + SquadRosterEntry * found = TableMapScan( *worker.arena, map, key ); // one LINEAR SCAN of the live entries + if ( found != NULL ) + { + TableResetMapValue( *found ); // a duplicate REPLACES: the value goes back to its defaults + return TableEntryValue( found ); + } + TableMapHead * head = TableMapReach( worker, map ); + if ( head == NULL ) { return NULL; } + SquadRosterEntry * entry = TableMapAppend( worker, head, map ); // APPENDS; nothing ever moves (§6.4) + if ( entry == NULL ) { return NULL; } + TableReset( *entry ); + TableEntrySetKey( *entry, key ); + return TableEntryValue( entry ); +} + +// FIND on the builder: the same linear scan, O( n ) key compares over the +// segments in insertion order. NULL when absent. The builder builds NO +// INDEX, and that is a rule — the sort happens once, at Lock, Save or +// Cook, and every lookup that matters runs over the sorted region. +inline Item * SquadRosterFind( TableArena & arena, TableMap & map, uint8_t key ) +{ + SquadRosterEntry * found = TableMapScan( arena, map, key ); + return found != NULL ? TableEntryValue( found ) : NULL; +} + +// ERASE: marks the entry DEAD, one bit in the segment's slot and not in the +// entry table. False when absent. Its storage is held until the builder +// resets and never reused mid-build, because reusing a slot would make "an +// entry's address is stable" false for exactly one case. +inline bool SquadRosterErase( TableArena & arena, TableMap & map, uint8_t key ) +{ + return TableMapErase( arena, map, key ); +} + +// EACH on the builder: INSERTION order, live entries only. +inline TableMapEach SquadRosterEach( const TableArena & arena, const TableMap & map ) +{ + return TableMapEachOf( arena, map ); +} + +// ---- the OPTIONAL INDEX: caller-owned, built at load, never stored ---- +// +// Open addressing with linear probing over the sorted array, for a map large +// enough that log n compares over a cold array cost more than one hash and a +// probe. ITS HASH AND ITS LOAD FACTOR ARE NOT A CROSS-PORT CONTRACT: the +// index is never stored, so no golden, no cook-check rule and no +// build-version line ever names either. What a port is held to is the +// CONTRACT of the lookup — the same value the sorted array's Find returns +// for the same key, and no allocation past the storage the caller handed in. +inline int64_t SquadRosterIndexMeasure( const TableMap & map ) +{ + return (int64_t) TableMapIndexSlots( map.count ) * (int64_t) sizeof( int32_t ); +} + +inline TableMapIndex SquadRosterIndex( const TableMap & map, void * storage, int64_t bytes ) +{ + TableMapIndex index; + const int32_t slots = TableMapIndexSlots( map.count ); + if ( storage == NULL || bytes < (int64_t) slots * (int64_t) sizeof( int32_t ) ) { return index; } + index.slots = (int32_t *) storage; + index.capacity = slots; + for ( int32_t i = 0; i < slots; i++ ) { index.slots[i] = 0; } + const SquadRosterEntry * entries = map.Entries(); + for ( int32_t i = 0; i < map.count; i++ ) // ONE PASS over the sorted array + { + int32_t at = (int32_t) ( TableMapHash( (uint64_t) entries[i].key ) & (uint64_t) ( slots - 1 ) ); + while ( index.slots[at] != 0 ) { at = ( at + 1 ) & ( slots - 1 ); } + index.slots[at] = i + 1; // slots are ENTRY INDICES; 0 is an empty slot + } + index.good = true; + return index; +} + +inline const Item * SquadRosterIndexFind( const TableMapIndex & index, const TableMap & map, uint8_t key ) +{ + if ( !index.good ) { return map.Find( key ); } // an index that did not build is not a wrong answer + const SquadRosterEntry * entries = map.Entries(); + int32_t at = (int32_t) ( TableMapHash( (uint64_t) key ) & (uint64_t) ( index.capacity - 1 ) ); + for ( int32_t probe = 0; probe < index.capacity; probe++ ) + { + const int32_t slot = index.slots[at]; + if ( slot == 0 ) { return NULL; } + if ( TableEntryOrder( entries[slot - 1], key ) == 0 ) { return TableEntryFound( entries + slot - 1 ); } + at = ( at + 1 ) & ( index.capacity - 1 ); + } + return NULL; +} + +// ---- Row.items: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Sample * RowItemsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool RowItemsErase( TableArena & arena, TableList & list, const Sample * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach RowItemsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Sheet.rows: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Row * SheetRowsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool SheetRowsErase( TableArena & arena, TableList & list, const Row * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach SheetRowsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Army.squads: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Squad * ArmySquadsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool ArmySquadsErase( TableArena & arena, TableList & list, const Squad * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach ArmySquadsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// RowNumber: number everything Row POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool RowNumber( const Ctx & ctx, TableNumbering & numbering, const Row & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// RowPackMeasure: the packed region bytes of everything Row POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t RowPackMeasure( const Ctx & ctx, TablePackMap & seen, const Row & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// RowPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool RowPackEdges( const Ctx & ctx, TablePackMap & seen, const Row & src, Row & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool RowPack( const Ctx & ctx, TablePackMap & seen, const Row & src, Row & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Row ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Row ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !RowExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return RowPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool RowPackEdges( const Ctx & ctx, TablePackMap & seen, const Row & src, Row & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// SheetNumber: number everything Sheet POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool SheetNumber( const Ctx & ctx, TableNumbering & numbering, const Sheet & value ) +{ + { // rows: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_rows = TableListElements( ctx, value.rows ); + if ( !cursor_rows.ok ) { return false; } + for ( int32_t i = 0; i < cursor_rows.count; i++ ) + { + if ( !RowNumber( ctx, numbering, cursor_rows[i] ) ) { return false; } + } + } + { + const Row * pointee = RowAt( ctx, value.pinned ); // pinned + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( numbering.seen, (const void *) pointee, + (int64_t) ( numbering.count + 2 ), taken, slot ); // its index, if this is its first visit + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + } + else + { + TableNodeEntry node; + node.node = (const void *) pointee; + node.type_id = 0xa013e119fec906fbull; // fnv1a64( "Row" ) + node.type_slot = 55; // its slot in the unit's vocabulary (§3.3) + node.measure = &TableNodeMeasureThunk; + node.save = &TableNodeSaveThunk; + if ( !TableNumberingAppend( numbering, node ) ) { return false; } + if ( !RowNumber( ctx, numbering, *pointee ) ) { return false; } + TablePackMapClose( numbering.seen, (const void *) pointee, slot ); + } + } + } + return true; +} + +// SheetPackMeasure: the packed region bytes of everything Sheet POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t SheetPackMeasure( const Ctx & ctx, TablePackMap & seen, const Sheet & value ) +{ + int64_t bytes = 0; + { // rows: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_rows = TableListElements( ctx, value.rows ); + if ( !cursor_rows.ok ) { return -1; } + for ( int32_t i = 0; i < cursor_rows.count; i++ ) + { + int64_t inner = RowPackMeasure( ctx, seen, cursor_rows[i] ); + if ( inner < 0 ) { return -1; } + bytes += inner; + } + } + { + const Row * pointee = RowAt( ctx, value.pinned ); // pinned + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, 0, taken, slot ); + if ( entry == NULL ) { return -1; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return -1; } // a data cycle + } + else + { + int64_t inner = RowPackMeasure( ctx, seen, *pointee ); + if ( inner < 0 ) { return -1; } + TablePackMapClose( seen, (const void *) pointee, slot ); + int64_t node_extent = RowExtent( ctx, *pointee ); + if ( node_extent < 0 ) { return -1; } + bytes += TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + node_extent ) + inner; + } + } + } + return bytes; +} + +// SheetPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool SheetPackEdges( const Ctx & ctx, TablePackMap & seen, const Sheet & src, Sheet & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool SheetPack( const Ctx & ctx, TablePackMap & seen, const Sheet & src, Sheet & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Sheet ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Sheet ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !SheetExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return SheetPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool SheetPackEdges( const Ctx & ctx, TablePackMap & seen, const Sheet & src, Sheet & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + { // rows: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_rows = TableListElements( ctx, src.rows ); + if ( !cursor_rows.ok ) { return false; } + Row * placed_rows = (Row *) ( dst.rows.elements.value != 0 ? ( (uint8_t *) &dst.rows.elements + dst.rows.elements.value ) : NULL ); + for ( int32_t i = 0; i < cursor_rows.count; i++ ) + { + if ( !RowPackEdges( ctx, seen, cursor_rows[i], placed_rows[i], base, capacity, used ) ) { return false; } + } + } + { + dst.pinned.value = 0; // pinned + const Row * pointee = RowAt( ctx, src.pinned ); + if ( pointee != NULL ) + { + int64_t at = TableAlignUp64( used ); // where it WOULD land, if this is its first visit + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, at, taken, slot ); + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + dst.pinned.value = (int64_t) ( ( base + entry->offset ) - (const uint8_t *) &dst.pinned ); // the one body it already has + } + else + { + int64_t node_extent = RowExtent( ctx, *pointee ); + if ( node_extent < 0 ) { return false; } + const int64_t node_bytes = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + node_extent ); + if ( at + node_bytes > capacity ) { return false; } + used = at + node_bytes; + Row * child = new ( base + at ) Row; // lifetime only: the Pack below memcpy's the whole node over it + dst.pinned.value = (int64_t) ( ( base + at ) - (const uint8_t *) &dst.pinned ); + if ( !RowPack( ctx, seen, *pointee, *child, base, capacity, used ) ) { return false; } + TablePackMapClose( seen, (const void *) pointee, slot ); + } + } + } + return true; +} + +// SquadNumber: number everything Squad POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool SquadNumber( const Ctx & ctx, TableNumbering & numbering, const Squad & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// SquadPackMeasure: the packed region bytes of everything Squad POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t SquadPackMeasure( const Ctx & ctx, TablePackMap & seen, const Squad & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// SquadPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool SquadPackEdges( const Ctx & ctx, TablePackMap & seen, const Squad & src, Squad & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool SquadPack( const Ctx & ctx, TablePackMap & seen, const Squad & src, Squad & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Squad ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Squad ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !SquadExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return SquadPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool SquadPackEdges( const Ctx & ctx, TablePackMap & seen, const Squad & src, Squad & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// ArmyNumber: number everything Army POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool ArmyNumber( const Ctx & ctx, TableNumbering & numbering, const Army & value ) +{ + { // squads: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_squads = TableListElements( ctx, value.squads ); + if ( !cursor_squads.ok ) { return false; } + for ( int32_t i = 0; i < cursor_squads.count; i++ ) + { + if ( !SquadNumber( ctx, numbering, cursor_squads[i] ) ) { return false; } + } + } + return true; +} + +// ArmyPackMeasure: the packed region bytes of everything Army POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t ArmyPackMeasure( const Ctx & ctx, TablePackMap & seen, const Army & value ) +{ + int64_t bytes = 0; + { // squads: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_squads = TableListElements( ctx, value.squads ); + if ( !cursor_squads.ok ) { return -1; } + for ( int32_t i = 0; i < cursor_squads.count; i++ ) + { + int64_t inner = SquadPackMeasure( ctx, seen, cursor_squads[i] ); + if ( inner < 0 ) { return -1; } + bytes += inner; + } + } + return bytes; +} + +// ArmyPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool ArmyPackEdges( const Ctx & ctx, TablePackMap & seen, const Army & src, Army & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool ArmyPack( const Ctx & ctx, TablePackMap & seen, const Army & src, Army & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Army ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Army ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !ArmyExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return ArmyPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool ArmyPackEdges( const Ctx & ctx, TablePackMap & seen, const Army & src, Army & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + { // squads: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_squads = TableListElements( ctx, src.squads ); + if ( !cursor_squads.ok ) { return false; } + Squad * placed_squads = (Squad *) ( dst.squads.elements.value != 0 ? ( (uint8_t *) &dst.squads.elements + dst.squads.elements.value ) : NULL ); + for ( int32_t i = 0; i < cursor_squads.count; i++ ) + { + if ( !SquadPackEdges( ctx, seen, cursor_squads[i], placed_squads[i], base, capacity, used ) ) { return false; } + } + } + return true; +} + +// DeckNumber: number everything Deck POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool DeckNumber( const Ctx & ctx, TableNumbering & numbering, const Deck & value ) +{ + for ( int32_t i = 0; i < value.hands_count && i < 3; i++ ) // hands + { + if ( !RowNumber( ctx, numbering, value.hands[i] ) ) { return false; } + } + return true; +} + +// DeckPackMeasure: the packed region bytes of everything Deck POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t DeckPackMeasure( const Ctx & ctx, TablePackMap & seen, const Deck & value ) +{ + int64_t bytes = 0; + for ( int32_t i = 0; i < value.hands_count && i < 3; i++ ) // hands + { + int64_t inner = RowPackMeasure( ctx, seen, value.hands[i] ); + if ( inner < 0 ) { return -1; } + bytes += inner; + } + return bytes; +} + +// DeckPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool DeckPackEdges( const Ctx & ctx, TablePackMap & seen, const Deck & src, Deck & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool DeckPack( const Ctx & ctx, TablePackMap & seen, const Deck & src, Deck & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Deck ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Deck ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !DeckExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return DeckPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool DeckPackEdges( const Ctx & ctx, TablePackMap & seen, const Deck & src, Deck & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + for ( int32_t i = 0; i < src.hands_count && i < 3; i++ ) // hands + { + if ( !RowPackEdges( ctx, seen, src.hands[i], dst.hands[i], base, capacity, used ) ) { return false; } + } + return true; +} + +// ---- Row: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: RowBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Row is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct RowBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + RowBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~RowBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + RowBuilder( const RowBuilder & ) = delete; + RowBuilder & operator=( const RowBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Row * GetRoot() { return arena.locked ? NULL : (Row *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Row * AsConst() const { return (const Row *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool RowBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Row & root = *(const Row *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = RowPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = RowExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + Row * destination = new ( packed ) Row; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !RowPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Row on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// RowNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t RowNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// RowNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void RowNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// RowNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t RowNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// RowNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t RowNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// RowNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void RowNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = RowNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? RowNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool RowNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Row & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return RowNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t RowMeasureWire( const Ctx & ctx, const Row & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( RowNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = RowMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t RowSaveWire( const Ctx & ctx, const Row & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !RowNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = RowSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == RowMeasure( root ) +} + +inline int64_t RowMeasure( const Row * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return RowMeasureWire( ctx, *root, allocator ); +} + +inline int64_t RowSave( const Row * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return RowSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t RowMeasure( const RowBuilder & builder ) +{ + if ( builder.region != NULL ) { return RowMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return RowMeasureWire( ctx, *(const Row *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t RowSave( const RowBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return RowSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return RowSaveWire( ctx, *(const Row *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t RowMeasureMessage( const Row * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return RowMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t RowSaveMessage( const Row * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return RowSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t RowMeasureMessage( const RowBuilder & builder ) +{ + if ( builder.region != NULL ) { return RowMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return RowMeasureWire( ctx, *(const Row *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t RowSaveMessage( const RowBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return RowSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return RowSaveWire( ctx, *(const Row *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// RowLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t RowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = RowNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// RowLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Row * RowLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Row ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = RowNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xa013e119fec906fbull; + Row * root = new ( region ) Row; // lifetime only: LoadBody's first act is RowReset + RowReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = RowNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + RowNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + RowNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Row ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + RowLoadBody( r, nodes, *root ); + return root; +} + +// RowLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t RowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = RowNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// RowLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Row * RowLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Row ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = RowNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xa013e119fec906fbull; + Row * root = new ( region ) Row; // lifetime only: LoadBody's first act is RowReset + RowReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = RowNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + RowNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + RowNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Row ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + RowLoadBody( r, nodes, *root ); + return root; +} + +// RowLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool RowLoadBuilder( RowBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Row * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0xa013e119fec906fbull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = RowNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + RowNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = RowLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Sheet: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: SheetBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Sheet is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct SheetBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + SheetBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~SheetBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + SheetBuilder( const SheetBuilder & ) = delete; + SheetBuilder & operator=( const SheetBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Sheet * GetRoot() { return arena.locked ? NULL : (Sheet *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Sheet * AsConst() const { return (const Sheet *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool SheetBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Sheet & root = *(const Sheet *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = SheetPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = SheetExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + Sheet * destination = new ( packed ) Sheet; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !SheetPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Sheet on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// SheetNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t SheetNodeStorage( uint64_t type_id, const uint8_t * body, int64_t length, const TableIdTable * ids, TableRefuseReason & reason ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + case 0xa013e119fec906fbull: // Row + { + int64_t extent = 0; + if ( !RowWireExtent( body, length, extent, ids, reason ) ) { return kTableNodeRefused; } + return TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + extent ); + } + default: break; + } + return -1; +} + +// SheetNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void SheetNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + switch ( type_id ) + { + case 0xa013e119fec906fbull: { Row * node = new ( at ) Row; RowReset( *node ); break; } // Row + default: break; + } +} + +// SheetNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t SheetNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + case 0xa013e119fec906fbull: return TableAlignUp64( (int64_t) sizeof( Row ) ); // Row + default: break; + } + return 0; +} + +// SheetNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t SheetNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + switch ( type_id ) + { + case 0xa013e119fec906fbull: return (uint32_t) worker.Alloc().ref.value; // Row + default: break; + } + return 0; +} + +// SheetNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void SheetNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + TableRefuseReason reason = count_over_length; // pass one already refused what this could refuse + const int64_t storage = SheetNodeStorage( type_id, r.buffer, r.size, r.ids, reason ); + const int64_t record = storage > 0 ? SheetNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + switch ( type_id ) + { + case 0xa013e119fec906fbull: RowLoadBody( r, nodes, *(Row *) at ); break; // Row + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool SheetNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Sheet & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return SheetNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t SheetMeasureWire( const Ctx & ctx, const Sheet & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( SheetNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = SheetMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t SheetSaveWire( const Ctx & ctx, const Sheet & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !SheetNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = SheetSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == SheetMeasure( root ) +} + +inline int64_t SheetMeasure( const Sheet * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SheetMeasureWire( ctx, *root, allocator ); +} + +inline int64_t SheetSave( const Sheet * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SheetSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t SheetMeasure( const SheetBuilder & builder ) +{ + if ( builder.region != NULL ) { return SheetMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SheetMeasureWire( ctx, *(const Sheet *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t SheetSave( const SheetBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return SheetSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SheetSaveWire( ctx, *(const Sheet *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t SheetMeasureMessage( const Sheet * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SheetMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t SheetSaveMessage( const Sheet * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SheetSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t SheetMeasureMessage( const SheetBuilder & builder ) +{ + if ( builder.region != NULL ) { return SheetMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return SheetMeasureWire( ctx, *(const Sheet *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t SheetSaveMessage( const SheetBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return SheetSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return SheetSaveWire( ctx, *(const Sheet *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// SheetLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t SheetLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !SheetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SheetNodeStorage( type_id, body, length, &ids_table, reason ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// SheetLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Sheet * SheetLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Sheet ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !SheetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SheetNodeStorage( type_id, body, length, &ids_table, reason ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x0cc9e0af9a85fbc8ull; + Sheet * root = new ( region ) Sheet; // lifetime only: LoadBody's first act is SheetReset + SheetReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = SheetNodeStorage( type_id, body, length, &ids_table, reason ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + SheetNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SheetNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Sheet ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + SheetLoadBody( r, nodes, *root ); + return root; +} + +// SheetLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t SheetLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !SheetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SheetNodeStorage( type_id, body, length, &ids_table, reason ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// SheetLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Sheet * SheetLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Sheet ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !SheetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SheetNodeStorage( type_id, body, length, &ids_table, reason ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x0cc9e0af9a85fbc8ull; + Sheet * root = new ( region ) Sheet; // lifetime only: LoadBody's first act is SheetReset + SheetReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Sheet ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = SheetNodeStorage( type_id, body, length, &ids_table, reason ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + SheetNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SheetNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Sheet ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + SheetLoadBody( r, nodes, *root ); + return root; +} + +// SheetLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool SheetLoadBuilder( SheetBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Sheet * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0x0cc9e0af9a85fbc8ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = SheetNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SheetNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = SheetLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Squad: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: SquadBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Squad is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct SquadBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + SquadBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~SquadBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + SquadBuilder( const SquadBuilder & ) = delete; + SquadBuilder & operator=( const SquadBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Squad * GetRoot() { return arena.locked ? NULL : (Squad *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Squad * AsConst() const { return (const Squad *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool SquadBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Squad & root = *(const Squad *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = SquadPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = SquadExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + Squad * destination = new ( packed ) Squad; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !SquadPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Squad on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// SquadNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t SquadNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// SquadNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void SquadNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// SquadNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t SquadNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// SquadNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t SquadNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// SquadNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void SquadNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = SquadNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? SquadNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool SquadNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Squad & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return SquadNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t SquadMeasureWire( const Ctx & ctx, const Squad & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( SquadNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = SquadMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t SquadSaveWire( const Ctx & ctx, const Squad & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !SquadNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = SquadSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == SquadMeasure( root ) +} + +inline int64_t SquadMeasure( const Squad * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SquadMeasureWire( ctx, *root, allocator ); +} + +inline int64_t SquadSave( const Squad * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SquadSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t SquadMeasure( const SquadBuilder & builder ) +{ + if ( builder.region != NULL ) { return SquadMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SquadMeasureWire( ctx, *(const Squad *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t SquadSave( const SquadBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return SquadSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SquadSaveWire( ctx, *(const Squad *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t SquadMeasureMessage( const Squad * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SquadMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t SquadSaveMessage( const Squad * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SquadSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t SquadMeasureMessage( const SquadBuilder & builder ) +{ + if ( builder.region != NULL ) { return SquadMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return SquadMeasureWire( ctx, *(const Squad *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t SquadSaveMessage( const SquadBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return SquadSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return SquadSaveWire( ctx, *(const Squad *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// SquadLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t SquadLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SquadNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// SquadLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Squad * SquadLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Squad ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SquadNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xec07a2f760550a91ull; + Squad * root = new ( region ) Squad; // lifetime only: LoadBody's first act is SquadReset + SquadReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = SquadNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + SquadNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SquadNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Squad ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + SquadLoadBody( r, nodes, *root ); + return root; +} + +// SquadLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t SquadLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SquadNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// SquadLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Squad * SquadLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Squad ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SquadNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xec07a2f760550a91ull; + Squad * root = new ( region ) Squad; // lifetime only: LoadBody's first act is SquadReset + SquadReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = SquadNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + SquadNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SquadNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Squad ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + SquadLoadBody( r, nodes, *root ); + return root; +} + +// SquadLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool SquadLoadBuilder( SquadBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Squad * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0xec07a2f760550a91ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = SquadNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SquadNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = SquadLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Army: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: ArmyBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Army is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct ArmyBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + ArmyBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~ArmyBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + ArmyBuilder( const ArmyBuilder & ) = delete; + ArmyBuilder & operator=( const ArmyBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Army * GetRoot() { return arena.locked ? NULL : (Army *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Army * AsConst() const { return (const Army *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool ArmyBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Army & root = *(const Army *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = ArmyPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = ArmyExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + Army * destination = new ( packed ) Army; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !ArmyPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Army on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// ArmyNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t ArmyNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// ArmyNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void ArmyNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// ArmyNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t ArmyNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// ArmyNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t ArmyNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// ArmyNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void ArmyNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = ArmyNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? ArmyNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool ArmyNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Army & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return ArmyNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t ArmyMeasureWire( const Ctx & ctx, const Army & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( ArmyNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = ArmyMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t ArmySaveWire( const Ctx & ctx, const Army & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !ArmyNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = ArmySaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == ArmyMeasure( root ) +} + +inline int64_t ArmyMeasure( const Army * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return ArmyMeasureWire( ctx, *root, allocator ); +} + +inline int64_t ArmySave( const Army * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return ArmySaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t ArmyMeasure( const ArmyBuilder & builder ) +{ + if ( builder.region != NULL ) { return ArmyMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return ArmyMeasureWire( ctx, *(const Army *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t ArmySave( const ArmyBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return ArmySave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return ArmySaveWire( ctx, *(const Army *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t ArmyMeasureMessage( const Army * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return ArmyMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t ArmySaveMessage( const Army * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return ArmySaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t ArmyMeasureMessage( const ArmyBuilder & builder ) +{ + if ( builder.region != NULL ) { return ArmyMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return ArmyMeasureWire( ctx, *(const Army *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t ArmySaveMessage( const ArmyBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return ArmySaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return ArmySaveWire( ctx, *(const Army *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// ArmyLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t ArmyLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !ArmyWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = ArmyNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// ArmyLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Army * ArmyLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Army ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !ArmyWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = ArmyNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x06e2378b553a9e84ull; + Army * root = new ( region ) Army; // lifetime only: LoadBody's first act is ArmyReset + ArmyReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = ArmyNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + ArmyNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + ArmyNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Army ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + ArmyLoadBody( r, nodes, *root ); + return root; +} + +// ArmyLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t ArmyLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !ArmyWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = ArmyNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// ArmyLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Army * ArmyLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Army ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !ArmyWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = ArmyNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x06e2378b553a9e84ull; + Army * root = new ( region ) Army; // lifetime only: LoadBody's first act is ArmyReset + ArmyReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Army ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = ArmyNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + ArmyNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + ArmyNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Army ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + ArmyLoadBody( r, nodes, *root ); + return root; +} + +// ArmyLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool ArmyLoadBuilder( ArmyBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Army * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0x06e2378b553a9e84ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = ArmyNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + ArmyNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = ArmyLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Deck: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: DeckBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Deck is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct DeckBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + DeckBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~DeckBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + DeckBuilder( const DeckBuilder & ) = delete; + DeckBuilder & operator=( const DeckBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Deck * GetRoot() { return arena.locked ? NULL : (Deck *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Deck * AsConst() const { return (const Deck *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool DeckBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Deck & root = *(const Deck *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = DeckPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = DeckExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + Deck * destination = new ( packed ) Deck; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !DeckPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Deck on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// DeckNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t DeckNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// DeckNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void DeckNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// DeckNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t DeckNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// DeckNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t DeckNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// DeckNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void DeckNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = DeckNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? DeckNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool DeckNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Deck & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return DeckNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t DeckMeasureWire( const Ctx & ctx, const Deck & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( DeckNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = DeckMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t DeckSaveWire( const Ctx & ctx, const Deck & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !DeckNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = DeckSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == DeckMeasure( root ) +} + +inline int64_t DeckMeasure( const Deck * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return DeckMeasureWire( ctx, *root, allocator ); +} + +inline int64_t DeckSave( const Deck * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return DeckSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t DeckMeasure( const DeckBuilder & builder ) +{ + if ( builder.region != NULL ) { return DeckMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return DeckMeasureWire( ctx, *(const Deck *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t DeckSave( const DeckBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return DeckSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return DeckSaveWire( ctx, *(const Deck *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t DeckMeasureMessage( const Deck * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return DeckMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t DeckSaveMessage( const Deck * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return DeckSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t DeckMeasureMessage( const DeckBuilder & builder ) +{ + if ( builder.region != NULL ) { return DeckMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return DeckMeasureWire( ctx, *(const Deck *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t DeckSaveMessage( const DeckBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return DeckSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return DeckSaveWire( ctx, *(const Deck *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// DeckLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t DeckLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !DeckWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = DeckNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// DeckLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Deck * DeckLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Deck ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !DeckWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = DeckNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xd043187343bfcfe8ull; + Deck * root = new ( region ) Deck; // lifetime only: LoadBody's first act is DeckReset + DeckReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = DeckNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + DeckNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + DeckNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Deck ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + DeckLoadBody( r, nodes, *root ); + return root; +} + +// DeckLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t DeckLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !DeckWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = DeckNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// DeckLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Deck * DeckLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Deck ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !DeckWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = DeckNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xd043187343bfcfe8ull; + Deck * root = new ( region ) Deck; // lifetime only: LoadBody's first act is DeckReset + DeckReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Deck ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = DeckNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + DeckNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + DeckNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Deck ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + DeckLoadBody( r, nodes, *root ); + return root; +} + +// DeckLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool DeckLoadBuilder( DeckBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Deck * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0xd043187343bfcfe8ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = DeckNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + DeckNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = DeckLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- the cooked form: point at a cook (docs/SPEC-TABLES.md §7) ---- + +// SampleOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Sample IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Sample * SampleOpen( const void * bytes, uint64_t length ) +{ + return (const Sample *) TableCookOpen( bytes, length, (uint64_t) sizeof( Sample ), (uint64_t) alignof( Sample ) ); +} + +// RowOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH RowAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Row * RowOpen( const void * bytes, uint64_t length ) +{ + return (const Row *) TableCookOpen( bytes, length, (uint64_t) sizeof( Row ), (uint64_t) alignof( Row ) ); +} + +// SheetOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH SheetAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Sheet * SheetOpen( const void * bytes, uint64_t length ) +{ + return (const Sheet *) TableCookOpen( bytes, length, (uint64_t) sizeof( Sheet ), (uint64_t) alignof( Sheet ) ); +} + +// ItemOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Item IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Item * ItemOpen( const void * bytes, uint64_t length ) +{ + return (const Item *) TableCookOpen( bytes, length, (uint64_t) sizeof( Item ), (uint64_t) alignof( Item ) ); +} + +// SquadOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH SquadAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Squad * SquadOpen( const void * bytes, uint64_t length ) +{ + return (const Squad *) TableCookOpen( bytes, length, (uint64_t) sizeof( Squad ), (uint64_t) alignof( Squad ) ); +} + +// ArmyOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH ArmyAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Army * ArmyOpen( const void * bytes, uint64_t length ) +{ + return (const Army *) TableCookOpen( bytes, length, (uint64_t) sizeof( Army ), (uint64_t) alignof( Army ) ); +} + +// DeckOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH DeckAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Deck * DeckOpen( const void * bytes, uint64_t length ) +{ + return (const Deck *) TableCookOpen( bytes, length, (uint64_t) sizeof( Deck ), (uint64_t) alignof( Deck ) ); +} + +// ---- the cooked form: WRITE a cook (docs/SPEC-TABLES.md §7.6) ---- +// +// The bytes are `schema cook`'s, and the tool stays the reference: the two +// writers are held to one file, byte for byte, in both byte orders. A cook is +// content-addressed by (asset hash, build version), so two writers of one +// instance produce ONE artifact or the pair means nothing. + +inline void SampleCookBody( uint8_t * at, const Sample & value, TableByteOrder order ); +template inline bool RowCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Row & value, TableByteOrder order ); +template inline bool SheetCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Sheet & value, TableByteOrder order ); +inline void ItemCookBody( uint8_t * at, const Item & value, TableByteOrder order ); +inline void SquadRosterEntryCookBody( uint8_t * at, const SquadRosterEntry & value, TableByteOrder order ); +template inline bool SquadCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Squad & value, TableByteOrder order ); +template inline bool ArmyCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Army & value, TableByteOrder order ); +template inline bool DeckCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Deck & value, TableByteOrder order ); + +inline void SampleCookBody( uint8_t * at, const Sample & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.v, 4, order ); +} + +template inline bool RowCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Row & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // items: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.label, 4, order ); + return true; +} + +template inline bool SheetCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Sheet & value, TableByteOrder order ) +{ + table_cook_put( at + 0, 0, 8, order ); // rows: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + if ( !table_cook_ref( region, at + 16, (const void *) RowAt( ctx, value.pinned ), order ) ) { return false; } // pinned + return true; +} + +inline void ItemCookBody( uint8_t * at, const Item & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.count, 4, order ); +} + +inline void SquadRosterEntryCookBody( uint8_t * at, const SquadRosterEntry & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.key, 1, order ); + ItemCookBody( at + 4, value.value, order ); +} + +template inline bool SquadCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Squad & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // roster: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.name, 4, order ); + return true; +} + +template inline bool ArmyCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Army & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // squads: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.after, 4, order ); + return true; +} + +template inline bool DeckCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Deck & value, TableByteOrder order ) +{ + // all 3 slots: the storage is allocate-max, and a slot past the count rides as it lies (§7.2) + for ( int32_t i = 0; i < 3; i++ ) + { + if ( !RowCookBody( ctx, region, at + 0 + i * 24, value.hands[ i ], order ) ) { return false; } + } + table_cook_put( at + 72, (uint64_t) (uint32_t) value.hands_count, 4, order ); + table_cook_put( at + 76, (uint64_t) value.after, 4, order ); + return true; +} + +template inline bool SampleCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Sample & value, TableByteOrder order ); +template inline bool RowCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Row & value, TableByteOrder order ); +template inline bool SheetCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Sheet & value, TableByteOrder order ); +template inline bool ItemCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Item & value, TableByteOrder order ); +template inline bool SquadRosterEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const SquadRosterEntry & value, TableByteOrder order ); +template inline bool SquadCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Squad & value, TableByteOrder order ); +template inline bool ArmyCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Army & value, TableByteOrder order ); +template inline bool DeckCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Deck & value, TableByteOrder order ); + +// SampleCookExtent: Sample's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SampleCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Sample & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// RowCookExtent: Row's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool RowCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Row & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // items: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.items ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( Sample ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 4; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + SampleCookBody( array + i * 4, cursor[i], order ); + } + } + return true; +} + +// SheetCookExtent: Sheet's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SheetCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Sheet & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // rows: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.rows ); + if ( !cursor.ok ) { return false; } + at = ( at + 7 ) & ~(int64_t) 7; // at alignof( Row ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 24; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + if ( !RowCookBody( ctx, region, array + i * 24, cursor[i], order ) ) { return false; } + } + for ( int32_t i = 0; i < cursor.count; i++ ) // then, element by element in index order + { + if ( !RowCookExtent( ctx, region, extent, at, array + i * 24, cursor[i], order ) ) { return false; } + } + } + return true; +} + +// ItemCookExtent: Item's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool ItemCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Item & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// SquadRosterEntryCookExtent: SquadRosterEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SquadRosterEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const SquadRosterEntry & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// SquadCookExtent: Squad's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SquadCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Squad & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // roster + TableMapCursor cursor = TableMapOrder( ctx, value.roster ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( SquadRosterEntry ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 8; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) + { + SquadRosterEntryCookBody( array + i * 8, *cursor[i], order ); + } + TableMapRelease( cursor ); + } + return true; +} + +// ArmyCookExtent: Army's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool ArmyCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Army & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // squads: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.squads ); + if ( !cursor.ok ) { return false; } + at = ( at + 7 ) & ~(int64_t) 7; // at alignof( Squad ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 24; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + if ( !SquadCookBody( ctx, region, array + i * 24, cursor[i], order ) ) { return false; } + } + for ( int32_t i = 0; i < cursor.count; i++ ) // then, element by element in index order + { + if ( !SquadCookExtent( ctx, region, extent, at, array + i * 24, cursor[i], order ) ) { return false; } + } + } + return true; +} + +// DeckCookExtent: Deck's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool DeckCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Deck & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + for ( int32_t i = 0; i < ( value.hands_count < 3 ? value.hands_count : 3 ); i++ ) // hands + { + if ( !RowCookExtent( ctx, region, extent, at, record + 0 + i * 24, value.hands[i], order ) ) { return false; } + } + return true; +} + +// SampleCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool SampleCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Sample & value, TableByteOrder order ) +{ + SampleCookBody( at, value, order ); + int64_t extent_at = 0; + return SampleCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// RowCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool RowCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Row & value, TableByteOrder order ) +{ + if ( !RowCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return RowCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// SheetCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool SheetCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Sheet & value, TableByteOrder order ) +{ + if ( !SheetCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return SheetCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// ItemCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool ItemCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Item & value, TableByteOrder order ) +{ + ItemCookBody( at, value, order ); + int64_t extent_at = 0; + return ItemCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// SquadRosterEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool SquadRosterEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const SquadRosterEntry & value, TableByteOrder order ) +{ + SquadRosterEntryCookBody( at, value, order ); + int64_t extent_at = 0; + return SquadRosterEntryCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// SquadCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool SquadCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Squad & value, TableByteOrder order ) +{ + if ( !SquadCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return SquadCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// ArmyCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool ArmyCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Army & value, TableByteOrder order ) +{ + if ( !ArmyCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return ArmyCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// DeckCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool DeckCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Deck & value, TableByteOrder order ) +{ + if ( !DeckCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return DeckCookExtent( ctx, region, at + 80, extent_at, at, value, order ); +} + +// SampleCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Sample IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t SampleCookMeasure( const Sample & value ) +{ + (void) value; + return 88; // 64 header + 8 data + 16 attribution +} + +// SampleCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract SampleMeasure/SampleSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool SampleCook( const Sample & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) SampleCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 8, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + SampleCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 72, 0, 8, order ); + table_cook_put( raw + 80, 0xdc40d61254c70aa7ull, 8, order ); + return true; +} + +// RowCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool RowCookLayout( const Ctx & ctx, const Row & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = RowExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// RowCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t RowCookMeasureFrom( const Ctx & ctx, const Row & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( RowNumberFrom( ctx, numbering, root ) && RowCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// RowCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool RowCookFrom( const Ctx & ctx, const Row & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = RowNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && RowCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = RowCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0xa013e119fec906fbull, 8, order ); // the root: fnv1a64( "Row" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// RowCookMeasure / RowCook over a REGION root — a locked builder's AsConst, a +// region RowLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t RowCookMeasure( const Row * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return RowCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool RowCook( const Row * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return RowCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t RowCookMeasure( const RowBuilder & builder ) +{ + if ( builder.region != NULL ) { return RowCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return RowCookMeasureFrom( ctx, *(const Row *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool RowCook( const RowBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return RowCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return RowCookFrom( ctx, *(const Row *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// SheetCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool SheetCookLayout( const Ctx & ctx, const Sheet & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = SheetExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + case 0xa013e119fec906fbull: // Row + { + const int64_t extent = RowExtent( ctx, *(const Row *) numbering.entries[k].node ); + if ( extent < 0 ) { return false; } + size = 24 + extent; node_align = 8; + } + break; + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// SheetCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t SheetCookMeasureFrom( const Ctx & ctx, const Sheet & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( SheetNumberFrom( ctx, numbering, root ) && SheetCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// SheetCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool SheetCookFrom( const Ctx & ctx, const Sheet & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = SheetNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && SheetCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = SheetCookNode( ctx, region, region.base, root, order ); + for ( int64_t k = 0; ok && k < numbering.count; k++ ) + { + uint8_t * at = region.base + region.offsets[k + 1]; + const void * node = numbering.entries[k].node; + switch ( numbering.entries[k].type_id ) + { + case 0xa013e119fec906fbull: ok = RowCookNode( ctx, region, at, *(const Row *) node, order ); break; // Row + default: ok = false; break; + } + } + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0x0cc9e0af9a85fbc8ull, 8, order ); // the root: fnv1a64( "Sheet" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// SheetCookMeasure / SheetCook over a REGION root — a locked builder's AsConst, a +// region SheetLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t SheetCookMeasure( const Sheet * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SheetCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool SheetCook( const Sheet * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return SheetCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t SheetCookMeasure( const SheetBuilder & builder ) +{ + if ( builder.region != NULL ) { return SheetCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SheetCookMeasureFrom( ctx, *(const Sheet *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool SheetCook( const SheetBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return SheetCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SheetCookFrom( ctx, *(const Sheet *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ItemCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Item IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t ItemCookMeasure( const Item & value ) +{ + (void) value; + return 88; // 64 header + 8 data + 16 attribution +} + +// ItemCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract ItemMeasure/ItemSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool ItemCook( const Item & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) ItemCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 8, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + ItemCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 72, 0, 8, order ); + table_cook_put( raw + 80, 0x52cfa1d198476806ull, 8, order ); + return true; +} + +// SquadCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool SquadCookLayout( const Ctx & ctx, const Squad & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = SquadExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// SquadCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t SquadCookMeasureFrom( const Ctx & ctx, const Squad & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( SquadNumberFrom( ctx, numbering, root ) && SquadCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// SquadCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool SquadCookFrom( const Ctx & ctx, const Squad & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = SquadNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && SquadCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = SquadCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0xec07a2f760550a91ull, 8, order ); // the root: fnv1a64( "Squad" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// SquadCookMeasure / SquadCook over a REGION root — a locked builder's AsConst, a +// region SquadLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t SquadCookMeasure( const Squad * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SquadCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool SquadCook( const Squad * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return SquadCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t SquadCookMeasure( const SquadBuilder & builder ) +{ + if ( builder.region != NULL ) { return SquadCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SquadCookMeasureFrom( ctx, *(const Squad *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool SquadCook( const SquadBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return SquadCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SquadCookFrom( ctx, *(const Squad *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ArmyCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool ArmyCookLayout( const Ctx & ctx, const Army & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = ArmyExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// ArmyCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t ArmyCookMeasureFrom( const Ctx & ctx, const Army & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( ArmyNumberFrom( ctx, numbering, root ) && ArmyCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// ArmyCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool ArmyCookFrom( const Ctx & ctx, const Army & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = ArmyNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && ArmyCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = ArmyCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0x06e2378b553a9e84ull, 8, order ); // the root: fnv1a64( "Army" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// ArmyCookMeasure / ArmyCook over a REGION root — a locked builder's AsConst, a +// region ArmyLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t ArmyCookMeasure( const Army * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return ArmyCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool ArmyCook( const Army * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return ArmyCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t ArmyCookMeasure( const ArmyBuilder & builder ) +{ + if ( builder.region != NULL ) { return ArmyCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return ArmyCookMeasureFrom( ctx, *(const Army *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool ArmyCook( const ArmyBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return ArmyCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return ArmyCookFrom( ctx, *(const Army *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// DeckCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool DeckCookLayout( const Ctx & ctx, const Deck & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = DeckExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 80 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// DeckCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t DeckCookMeasureFrom( const Ctx & ctx, const Deck & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( DeckNumberFrom( ctx, numbering, root ) && DeckCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// DeckCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool DeckCookFrom( const Ctx & ctx, const Deck & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = DeckNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && DeckCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = DeckCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0xd043187343bfcfe8ull, 8, order ); // the root: fnv1a64( "Deck" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// DeckCookMeasure / DeckCook over a REGION root — a locked builder's AsConst, a +// region DeckLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t DeckCookMeasure( const Deck * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return DeckCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool DeckCook( const Deck * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return DeckCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t DeckCookMeasure( const DeckBuilder & builder ) +{ + if ( builder.region != NULL ) { return DeckCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return DeckCookMeasureFrom( ctx, *(const Deck *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool DeckCook( const DeckBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return DeckCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return DeckCookFrom( ctx, *(const Deck *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ---- relocatability, enforced: the wire is a pure length-prefixed +// stream AND the decoded storage is pointer-free — every closure type +// must stay trivially copyable and standard-layout, so instances can be +// memcpy'd, mmap'd, shared across processes, and walked through +// descriptor offsets. A failure here means a pointer, virtual or +// non-trivial member crept into generated storage. +// +// They ask the COMPILER ITSELF, which is what every C++ standard library +// answers the same two questions with — and it costs this header no +// include at all. +// A pointer FIELD is a TableRef — eight bytes and no address — so the +// property holds in BOTH forms: a fixed-size table is one relocatable +// struct, and a packed region is one relocatable block whose references +// are self-relative and therefore survive a plain memcpy. +static_assert( __is_trivially_copyable( Sample ), "Sample must stay relocatable" ); +static_assert( __is_standard_layout( Sample ), "Sample must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Row ), "Row must stay relocatable" ); +static_assert( __is_standard_layout( Row ), "Row must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Sheet ), "Sheet must stay relocatable" ); +static_assert( __is_standard_layout( Sheet ), "Sheet must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Item ), "Item must stay relocatable" ); +static_assert( __is_standard_layout( Item ), "Item must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( SquadRosterEntry ), "SquadRosterEntry must stay relocatable" ); +static_assert( __is_standard_layout( SquadRosterEntry ), "SquadRosterEntry must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Squad ), "Squad must stay relocatable" ); +static_assert( __is_standard_layout( Squad ), "Squad must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Army ), "Army must stay relocatable" ); +static_assert( __is_standard_layout( Army ), "Army must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Deck ), "Deck must stay relocatable" ); +static_assert( __is_standard_layout( Deck ), "Deck must stay standard-layout for offsetof" ); + +// ---- the cook's layout contract (docs/SPEC-TABLES.md §20.3) ---- +// +// The compiler derived every number below from the declaration and folded it +// into the BUILD VERSION; these asserts are this compiler saying whether it +// agrees. The model is not self-evidently right — on 32-bit System V +// alignof(uint64_t) is 4, not 8 — which is precisely why it is asserted +// rather than assumed. +static_assert( sizeof( Sample ) == 4, "Sample's sizeof moved: the build version was taken over 4, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Sample ) == 4, "Sample's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Sample, v ) == 0, "Sample's field v moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Row ) == 24, "Row's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Row ) == 8, "Row's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Row, items ) == 0, "Row's field items moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Row, label ) == 16, "Row's field label moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Sheet ) == 24, "Sheet's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Sheet ) == 8, "Sheet's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Sheet, rows ) == 0, "Sheet's field rows moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Sheet, pinned ) == 16, "Sheet's field pinned moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Item ) == 4, "Item's sizeof moved: the build version was taken over 4, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Item ) == 4, "Item's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Item, count ) == 0, "Item's field count moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( SquadRosterEntry ) == 8, "SquadRosterEntry's sizeof moved: the build version was taken over 8, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( SquadRosterEntry ) == 4, "SquadRosterEntry's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( SquadRosterEntry, key ) == 0, "SquadRosterEntry's field key moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( SquadRosterEntry, value ) == 4, "SquadRosterEntry's field value moved: the build version was taken over offset 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Squad ) == 24, "Squad's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Squad ) == 8, "Squad's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Squad, roster ) == 0, "Squad's field roster moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Squad, name ) == 16, "Squad's field name moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Army ) == 24, "Army's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Army ) == 8, "Army's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Army, squads ) == 0, "Army's field squads moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Army, after ) == 16, "Army's field after moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Deck ) == 80, "Deck's sizeof moved: the build version was taken over 80, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Deck ) == 8, "Deck's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Deck, hands ) == 0, "Deck's field hands moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Deck, after ) == 76, "Deck's field after moved: the build version was taken over offset 76 (docs/SPEC-TABLES.md §20.3)" ); + +static_assert( alignof( Sample ) <= kTableAlign, "Row.items: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( Row ) <= kTableAlign, "Sheet.rows: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( Squad ) <= kTableAlign, "Army.squads: an unbounded array's element alignment must fit the arena's" ); + +// ---- reflection descriptors (tables only, docs/SPEC-TABLES.md) ---- + +inline const TableTypeInfo * SampleTableType(); +inline const TableTypeInfo * RowTableType(); +inline const TableTypeInfo * SheetTableType(); +inline const TableTypeInfo * ItemTableType(); +inline const TableTypeInfo * SquadRosterEntryTableType(); +inline const TableTypeInfo * SquadTableType(); +inline const TableTypeInfo * ArmyTableType(); +inline const TableTypeInfo * DeckTableType(); +// The descriptors are CONSTANT-INITIALISED data, and a field's target is +// the ADDRESS of another descriptor. These declarations are what let a +// self- or mutually-referential graph — Node naming itself through *Node — +// be expressed as constant data instead of a lazy link, which could not +// have been written race-free OR recursion-safe. The whole reflection +// surface is therefore immutable: read it from any thread, any time. +extern const TableTypeInfo SampleTableInfo; +extern const TableTypeInfo RowTableInfo; +extern const TableTypeInfo SheetTableInfo; +extern const TableTypeInfo ItemTableInfo; +extern const TableTypeInfo SquadRosterEntryTableInfo; +extern const TableTypeInfo SquadTableInfo; +extern const TableTypeInfo ArmyTableInfo; +extern const TableTypeInfo DeckTableInfo; + +inline const TableFieldInfo SampleTableFields[] = { + { "v", "v", "int32", 0xaf63eb4c86020609ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Sample, v ), (uint32_t) sizeof( Sample::v ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo SampleTableInfo = { "Sample", (uint32_t) sizeof( Sample ), 1, SampleTableFields, +[]( void * p ) { SampleReset( *(Sample *) p ); }, false }; +inline const TableTypeInfo * SampleTableType() { return &SampleTableInfo; } + +inline const TableFieldInfo RowTableFields[] = { + { "items", "items", "Sample", 0x3e7884bf4f412c6full, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Row, items ), (uint32_t) sizeof( Sample ), (uint32_t) offsetof( Row, items.count ), 0xffffffffu, &SampleTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "label", "label", "int32", 0x39f7fcec8fcb623dull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Row, label ), (uint32_t) sizeof( Row::label ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo RowTableInfo = { "Row", (uint32_t) sizeof( Row ), 2, RowTableFields, +[]( void * p ) { RowReset( *(Row *) p ); }, true }; +inline const TableTypeInfo * RowTableType() { return &RowTableInfo; } + +inline const TableFieldInfo SheetTableFields[] = { + { "rows", "rows", "Row", 0xa3a7061ff10a8138ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Sheet, rows ), (uint32_t) sizeof( Row ), (uint32_t) offsetof( Sheet, rows.count ), 0xffffffffu, &RowTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "pinned", "pinned", "Row", 0x5f82477707ad620full, 17, false, true, []( const void * slot ) -> const void * { return (const void *) RowAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) RowEmplace( worker, *(TableRef *) slot ); }, false, false, 0, (uint32_t) offsetof( Sheet, pinned ), (uint32_t) sizeof( TableRef ), 0xffffffffu, 0xffffffffu, &RowTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo SheetTableInfo = { "Sheet", (uint32_t) sizeof( Sheet ), 2, SheetTableFields, +[]( void * p ) { SheetReset( *(Sheet *) p ); }, true }; +inline const TableTypeInfo * SheetTableType() { return &SheetTableInfo; } + +inline const TableFieldInfo ItemTableFields[] = { + { "count", "count", "int32", 0xb1e5e28e4479a274ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Item, count ), (uint32_t) sizeof( Item::count ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo ItemTableInfo = { "Item", (uint32_t) sizeof( Item ), 1, ItemTableFields, +[]( void * p ) { ItemReset( *(Item *) p ); }, false }; +inline const TableTypeInfo * ItemTableType() { return &ItemTableInfo; } + +inline const TableFieldInfo SquadRosterEntryTableFields[] = { + { "key", "key", "uint8", 0x3dc94a19365b10ecull, 6, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( SquadRosterEntry, key ), (uint32_t) sizeof( SquadRosterEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( SquadRosterEntry, value ), (uint32_t) sizeof( SquadRosterEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo SquadRosterEntryTableInfo = { "SquadRosterEntry", (uint32_t) sizeof( SquadRosterEntry ), 2, SquadRosterEntryTableFields, +[]( void * p ) { SquadRosterEntryReset( *(SquadRosterEntry *) p ); }, false }; +inline const TableTypeInfo * SquadRosterEntryTableType() { return &SquadRosterEntryTableInfo; } + +inline const TableFieldInfo SquadTableFields[] = { + { "roster", "roster", "map[uint8]Item", 0x1c84390d304f4f42ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Squad, roster ), (uint32_t) sizeof( SquadRosterEntry ), (uint32_t) offsetof( Squad, roster.count ), 0xffffffffu, &SquadRosterEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { SquadRosterEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint8_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint8_t) key_value ); } return (void *) placed; }, "" }, + { "name", "name", "int32", 0xc4bcadba8e631b86ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Squad, name ), (uint32_t) sizeof( Squad::name ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo SquadTableInfo = { "Squad", (uint32_t) sizeof( Squad ), 2, SquadTableFields, +[]( void * p ) { SquadReset( *(Squad *) p ); }, true }; +inline const TableTypeInfo * SquadTableType() { return &SquadTableInfo; } + +inline const TableFieldInfo ArmyTableFields[] = { + { "squads", "squads", "Squad", 0x7848019b0c02a926ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Army, squads ), (uint32_t) sizeof( Squad ), (uint32_t) offsetof( Army, squads.count ), 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Army, after ), (uint32_t) sizeof( Army::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo ArmyTableInfo = { "Army", (uint32_t) sizeof( Army ), 2, ArmyTableFields, +[]( void * p ) { ArmyReset( *(Army *) p ); }, true }; +inline const TableTypeInfo * ArmyTableType() { return &ArmyTableInfo; } + +inline const TableFieldInfo DeckTableFields[] = { + { "hands", "hands", "Row", 0x81b46a69304ee2c9ull, 13, true, false, NULL, NULL, true, false, 3, (uint32_t) offsetof( Deck, hands ), (uint32_t) sizeof( Deck::hands[0] ), (uint32_t) offsetof( Deck, hands_count ), 0xffffffffu, &RowTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Deck, after ), (uint32_t) sizeof( Deck::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo DeckTableInfo = { "Deck", (uint32_t) sizeof( Deck ), 2, DeckTableFields, +[]( void * p ) { DeckReset( *(Deck *) p ); }, true }; +inline const TableTypeInfo * DeckTableType() { return &DeckTableInfo; } + +// ---- the text form (docs/SPEC-TABLES.md §16) ---- + +// Sample in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// HoldersTable.cpp; link it to use them. +bool SampleFromJson( Sample & value, const char * text, int64_t bytes, TableReport * report ); +int64_t SampleToJsonMeasure( const Sample & value ); +int64_t SampleToJson( const Sample & value, char * buffer, int64_t capacity ); + +// Row in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in HoldersTable.cpp; link it to use them. +bool RowFromJson( RowBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t RowToJsonMeasure( const Row * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t RowToJson( const Row * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Sheet in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in HoldersTable.cpp; link it to use them. +bool SheetFromJson( SheetBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t SheetToJsonMeasure( const Sheet * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t SheetToJson( const Sheet * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Item in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// HoldersTable.cpp; link it to use them. +bool ItemFromJson( Item & value, const char * text, int64_t bytes, TableReport * report ); +int64_t ItemToJsonMeasure( const Item & value ); +int64_t ItemToJson( const Item & value, char * buffer, int64_t capacity ); + +// Squad in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in HoldersTable.cpp; link it to use them. +bool SquadFromJson( SquadBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t SquadToJsonMeasure( const Squad * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t SquadToJson( const Squad * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Army in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in HoldersTable.cpp; link it to use them. +bool ArmyFromJson( ArmyBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t ArmyToJsonMeasure( const Army * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t ArmyToJson( const Army * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Deck in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in HoldersTable.cpp; link it to use them. +bool DeckFromJson( DeckBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t DeckToJsonMeasure( const Deck * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t DeckToJson( const Deck * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/MigrateTable.cpp b/testdata/golden/tables/lists/MigrateTable.cpp new file mode 100644 index 000000000..bd509a04b --- /dev/null +++ b/testdata/golden/tables/lists/MigrateTable.cpp @@ -0,0 +1,3149 @@ +// Code generated by the schema compiler from Migrate.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — the TABLE wire's text form (docs/SPEC-TABLES.md §16). +// Compile this file to use FromJson / ToJson; a project that +// never reads or writes a text does not compile it and pays nothing. + +#include "MigrateTable.h" + +#include // the text form: number formatting +#include // the text form: exact number conversion +#include // the text form: the runtime's decimal point + +// The guard is not vestigial. Several listdemo Table.cpp files may be +// concatenated into ONE translation unit — a unity build — and without it +// each would redefine the walk. It is also why the walk's functions may be +// weak (vague linkage) across separate objects: ODR requires their +// definitions to be token-identical, and the generic-walk gate is what +// proves that, byte for byte, across every generated .cpp. +#ifndef LISTDEMO_SCHEMA_TABLE_JSON +#define LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +// ---- the pointer adapters (docs/SPEC-TABLES.md §16.7) ---- +// +// The walk below is ONE walk, byte-identical in every generated .cpp, and a +// pointer is the one kind it cannot walk alone: reading one needs the +// builder's arena and writing one needs a region's deref, and neither exists +// in a unit that declares no pointer. So the walk calls these three and does +// not define them. A unit with no pointer defines them as stubs no field ever +// reaches; a pointered unit defines them in the graph half that follows the +// walk. + +struct TableJsonIn; +struct TableJsonOut; + +// a pointer field's object, or the `&node` reference standing in for it, into +// the slot; the cursor is on the opening brace +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// the node a pointer slot names, in place — or as `&node` when it is shared +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// the FIRST key of an object the walk is skipping begins with `&`: the cursor is +// on its value. A dropped definition still takes its label (§16.7); a fixed reader +// skips the value whole, as it skips everything else it does not place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); + +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- +// +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map +inline bool TableJsonIsMap( const TableFieldInfo * f ); +// the map as a plain JSON object keyed by the KEY, in ASCENDING key order +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that object back into the slot, in whatever order the text gives it +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); + +// ---- json walk: begin ---- +// +// The TEXT form (docs/SPEC-TABLES.md §16): one table, one text, one walk over the +// reflection descriptors (§8). Reading fills ONE caller-owned instance and +// allocates nothing beyond it; writing targets a caller buffer with the +// wire's measure/write symmetry. Everything AROUND this — which file goes +// with which instance, what key an instance is filed under, how instances +// link into a root table's collections — is a packer's opinion and stays +// with the tool that holds it. +// +// The dialect: trailing commas are accepted on read (the authoring files +// this exists for carry them) and never written; comments are not JSON and +// are refused; unknown keys are skipped and counted; a duplicate key is +// last-wins and counted; a key present with the wrong JSON type is skipped +// and counted, never coerced. + +static const int32_t kTableJsonMaxDepth = 128; + +// A key longer than this cannot name a field, so it is skipped as unknown. +static const int32_t kTableJsonMaxKey = 256; + +// The longest numeric token the walk will convert. Anything longer is a +// value no field can hold and counts as a kind mismatch. +static const int32_t kTableJsonMaxNumber = 512; + +// The decimal point the C runtime is CURRENTLY using. Number conversion is +// the one locale-sensitive corner of the grammar — JSON's point is always +// '.', the runtime's is whatever the program set — so every number crosses +// this one character on the way out and on the way back in. Nothing else in +// the walk consults the locale. +inline char TableJsonDecimalPoint() +{ + const struct lconv * conv = localeconv(); + if ( conv != NULL && conv->decimal_point != NULL && conv->decimal_point[0] != 0 ) + { + return conv->decimal_point[0]; + } + return '.'; +} + +// ---- storage access: the descriptors give an offset and a width, and the +// ---- storage is the HOST's, so every load and store goes through a width +// ---- switch rather than a memcpy into the low bytes of a wider word + +// finite: not a NaN, not an infinity. Written without — the walk's +// runtime surface stays the handful of functions it already names. +// A vocabulary entry the descriptor could not spell. The generated name +// functions answer "???" for a value outside the declared set, and that is +// not a name — writing it would put a spelling in the text that the reader +// then counts as unknown, turning a refusal into a silent loss. +inline bool TableJsonNamed( const char * name ) +{ + return name != NULL && strcmp( name, "???" ) != 0; +} + +inline bool TableJsonFinite( double v ) +{ + return v == v && v <= 1.7976931348623157e308 && v >= -1.7976931348623157e308; +} + +inline uint64_t TableJsonGetRaw( const void * storage, uint32_t width ) +{ + switch ( width ) + { + case 1: { uint8_t v = 0; memcpy( &v, storage, 1 ); return v; } + case 2: { uint16_t v = 0; memcpy( &v, storage, 2 ); return v; } + case 4: { uint32_t v = 0; memcpy( &v, storage, 4 ); return v; } + case 8: { uint64_t v = 0; memcpy( &v, storage, 8 ); return v; } + } + return 0; +} + +inline void TableJsonSetRaw( void * storage, uint32_t width, uint64_t value ) +{ + switch ( width ) + { + case 1: { uint8_t v = (uint8_t) value; memcpy( storage, &v, 1 ); break; } + case 2: { uint16_t v = (uint16_t) value; memcpy( storage, &v, 2 ); break; } + case 4: { uint32_t v = (uint32_t) value; memcpy( storage, &v, 4 ); break; } + case 8: { uint64_t v = value; memcpy( storage, &v, 8 ); break; } + } +} + +inline int64_t TableJsonGetSigned( const void * storage, uint32_t width ) +{ + uint64_t raw = TableJsonGetRaw( storage, width ); + if ( width < 8 ) + { + uint64_t sign = uint64_t( 1 ) << ( width * 8 - 1 ); + if ( ( raw & sign ) != 0 ) + { + raw |= ~( ( sign << 1 ) - 1 ); + } + } + return (int64_t) raw; +} + +// ---- the WIDE kinds (docs/SPEC-TABLES.md §3, §16.2) ---- +// +// The 128-bit integers and the fixed-point family convert EXACTLY, over two +// 64-bit lanes: a 128-bit integer is a decimal integer, a fixed value a +// decimal in WHOLE UNITS (1.0, -0.25, 3.0000152587890625) and nothing +// on either path passes through a double. Nothing here needs a 128-bit type +// either, which is what keeps this walk one text for every unit. +struct TableJsonWide +{ + uint64_t lo; + uint64_t hi; +}; + +inline bool TableJsonKindWide( uint8_t kind ) { return kind >= 18 && kind <= 29; } +inline bool TableJsonKindWideSigned( uint8_t kind ) { return kind == 18 || ( kind >= 20 && kind <= 24 ); } +inline bool TableJsonKindFixed( uint8_t kind ) { return kind >= 20 && kind <= 29; } + +inline bool TableJsonWideZero( TableJsonWide v ) { return v.lo == 0 && v.hi == 0; } +inline bool TableJsonWideNegative( TableJsonWide v ) { return ( v.hi >> 63 ) != 0; } + +inline int TableJsonWideCompare( TableJsonWide a, TableJsonWide b, bool is_signed ) +{ + if ( is_signed && TableJsonWideNegative( a ) != TableJsonWideNegative( b ) ) { return TableJsonWideNegative( a ) ? -1 : 1; } + if ( a.hi != b.hi ) { return a.hi < b.hi ? -1 : 1; } + if ( a.lo != b.lo ) { return a.lo < b.lo ? -1 : 1; } + return 0; +} + +inline TableJsonWide TableJsonWideShl( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.hi = v.lo << ( n - 64 ); return r; } + r.hi = ( v.hi << n ) | ( v.lo >> ( 64 - n ) ); + r.lo = v.lo << n; + return r; +} + +inline TableJsonWide TableJsonWideShr( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.lo = v.hi >> ( n - 64 ); return r; } + r.lo = ( v.lo >> n ) | ( v.hi << ( 64 - n ) ); + r.hi = v.hi >> n; + return r; +} + +inline TableJsonWide TableJsonWideNeg( TableJsonWide v ) +{ + TableJsonWide r; + r.lo = ~v.lo + 1; + r.hi = ~v.hi + ( r.lo == 0 ? 1 : 0 ); + return r; +} + +// v = v * m + a; the return is the carry out of 128 bits +inline uint32_t TableJsonWideMulAdd( TableJsonWide * v, uint32_t m, uint32_t a ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t carry = a; + for ( int i = 0; i < 4; i++ ) + { + uint64_t p = limb[i] * m + carry; + limb[i] = p & 0xffffffffull; + carry = p >> 32; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) carry; +} + +// v = v / d; the return is the remainder +inline uint32_t TableJsonWideDiv( TableJsonWide * v, uint32_t d ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t rem = 0; + for ( int i = 3; i >= 0; i-- ) + { + uint64_t cur = ( rem << 32 ) | limb[i]; + limb[i] = cur / d; + rem = cur % d; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) rem; +} + +// The storage of a wide kind, as lanes. A sixteen-byte storage is serialize's +// pair — native __int128 in the host's byte order, or the emulated struct with +// its low lane first — so the lanes are read in the host's order; a narrower +// storage is one lane, sign-extended for a signed kind. +inline TableJsonWide TableJsonWideLoad( const void * storage, uint32_t width, bool is_signed ) +{ + TableJsonWide v = { 0, 0 }; + if ( width == 16 ) + { + uint64_t half[2]; + memcpy( half, storage, 16 ); + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + v.lo = little ? half[0] : half[1]; + v.hi = little ? half[1] : half[0]; + return v; + } + v.lo = is_signed ? (uint64_t) TableJsonGetSigned( storage, width ) : TableJsonGetRaw( storage, width ); + v.hi = ( is_signed && ( v.lo >> 63 ) != 0 ) ? ~uint64_t( 0 ) : 0; + return v; +} + +inline void TableJsonWideStore( void * storage, uint32_t width, TableJsonWide v ) +{ + if ( width == 16 ) + { + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + uint64_t half[2]; + half[0] = little ? v.lo : v.hi; + half[1] = little ? v.hi : v.lo; + memcpy( storage, half, 16 ); + return; + } + TableJsonSetRaw( storage, width, v.lo ); +} + +// a counted field's companion: a string's length, a bytes' length, a counted +// array's count. Bounded by the declared extent on the way out, so a storage +// invariant a caller broke cannot walk off the end of the array. +inline int32_t TableJsonCount( const void * base, const TableFieldInfo * f ) +{ + if ( !f->counted ) + { + return f->array_bound; + } + int32_t count = 0; + memcpy( &count, (const uint8_t *) base + f->count_offset, sizeof( count ) ); + if ( count < 0 ) { count = 0; } + if ( count > f->array_bound ) { count = f->array_bound; } + return count; +} + +inline void TableJsonSetCount( void * base, const TableFieldInfo * f, int32_t count ) +{ + if ( f->counted ) + { + memcpy( (uint8_t *) base + f->count_offset, &count, sizeof( count ) ); + } +} + +// ---- what a field's kind expects to see in the text ---- +// +// One classifier, consulted by both directions, so a reader and a writer can +// never disagree about a kind's JSON form. 'o' object, 'a' array, 's' +// string, 'n' number, 'b' boolean. +// +// A vocabulary field is spelled by NAME: an enum is one name, a flags mask +// is the array of the names of its set bits. The two are told apart by the +// id column — an enum variant rides under a wire id, a flags BIT never does +// (docs/SPEC-TABLES.md §4), so a name function with no id function is flags. +// +// bytes(N) is the one kind whose element kind does not decide its form: it +// shares u8 with a plain array of u8, and rides as base64. The schema type +// name settles it, and "bytes" is a keyword no declaration can claim. +inline bool TableJsonIsBytes( const TableFieldInfo * f ) +{ + return f->is_array && f->kind == 6 && strcmp( f->type_name, "bytes" ) == 0; +} + +// An ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): its JSON form is an OBJECT +// keyed by variant name, not a positional array, because that is what the +// storage is — one slot per variant, addressed by the variant. +inline bool TableJsonIsKeyed( const TableFieldInfo * f ) +{ + return f->key_name != NULL; +} + +// THE KEY A STORAGE SLOT HOLDS (§2.4, §8): the storage shifts left, so slot i +// holds the key i + 1 and nothing is stored for None. This is the ONE place +// the walker spells the shift. +inline uint64_t TableJsonKeyedSlotKey( int64_t slot ) +{ + return (uint64_t) ( slot + 1 ); +} + +// A slot whose key names a variant of the keying enum. Every slot in +// [0, array_bound) does, unless the enum carries max-headroom variants outside +// a table closure, where a reserved value names nothing and its key id is 0 — +// the reserved id no declared name can fold to (§5). +inline bool TableJsonKeyedSlotValid( const TableFieldInfo * f, int64_t slot ) +{ + return f->key_id( TableJsonKeyedSlotKey( slot ) ) != 0; +} + +inline bool TableJsonIsFlags( const TableFieldInfo * f ) +{ + return f->enum_name != NULL && f->variant_id == NULL; +} + +inline bool TableJsonIsEnum( const TableFieldInfo * f ) +{ + return f->variant_id != NULL && f->arms == NULL; +} + +inline char TableJsonShape( const TableFieldInfo * f ) +{ + if ( TableJsonIsMap( f ) ) return 'o'; // a MAP: an object keyed by the KEY (§2.8) + if ( f->kind == 12 ) return 's'; // string + if ( TableJsonIsBytes( f ) ) return 's'; // bytes: base64 + if ( TableJsonIsKeyed( f ) ) return 'o'; // an object keyed by variant NAME + if ( f->is_array ) return 'a'; + if ( f->arms != NULL ) return 'o'; // union: an object with ONE key + if ( f->kind == 13 ) return 'o'; // nested table or type + if ( f->kind == 17 ) return f->table != NULL ? 'o' : 's'; // a pointer: the pointee's object in place, or null (§16.7); a byte buffer's string (§2.5) + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// the ELEMENT shape of an array field — the same classifier one level down +inline char TableJsonElementShape( const TableFieldInfo * f ) +{ + if ( f->arms != NULL ) return 'o'; // an element of an array of unions: one key, the arm (§2.6) + if ( f->kind == 13 ) return 'o'; + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// A guarded group rides only when its guard reads true — the wire's own +// elision (§4), carried into the text so a text and a wire written from one +// instance say the same thing. The guard is spelled as its branch condition +// over bool fields of the SAME type ("at_rest", "!at_rest", +// "active && has_target"), so evaluating it is a walk of the same +// descriptor. Nothing is inferred in the other direction: reading places +// every key it can name, and the guard is a plain bool key (§16.2). +inline bool TableJsonGuardHolds( const void * base, const TableTypeInfo * info, const char * guard ) +{ + const char * p = guard; + for ( ;; ) + { + while ( *p == ' ' || *p == '&' ) { p++; } + if ( *p == 0 ) { return true; } + bool want = true; + if ( *p == '!' ) { want = false; p++; } + const char * start = p; + while ( *p != 0 && *p != ' ' && *p != '&' ) { p++; } + size_t length = (size_t) ( p - start ); + bool value = false; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( strlen( f->name ) == length && strncmp( f->name, start, length ) == 0 ) + { + value = TableJsonGetRaw( (const uint8_t *) base + f->offset, f->elem_size ) != 0; + break; + } + } + if ( value != want ) { return false; } + } +} + +// ---- writing ---- + +// The writer sink MEASURES when the buffer is NULL and WRITES when it is +// not, over one code path — so measure and write agree byte for byte, the +// wire's invariant (§9) carried across. +struct TableJsonOut +{ + char * buffer; + int64_t capacity; + int64_t offset; + bool overflow; + void * graph; // the pointered write's identity map (§16.7); NULL for a fixed table + + void raw( const char * data, int64_t count ) + { + if ( buffer != NULL ) + { + if ( offset + count > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) count ); + } + offset += count; + } + void put( char c ) { raw( &c, 1 ); } + void text( const char * s ) { raw( s, (int64_t) strlen( s ) ); } + void line( int32_t depth ) + { + put( '\n' ); + for ( int32_t i = 0; i < depth; i++ ) { raw( " ", 2 ); } + } +}; + +inline const char * TableJsonBase64Alphabet() +{ + return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; +} + +inline void TableJsonWriteBase64( TableJsonOut & out, const uint8_t * data, int32_t length ) +{ + const char * alphabet = TableJsonBase64Alphabet(); + out.put( '"' ); + int32_t i = 0; + for ( ; i + 3 <= length; i += 3 ) + { + uint32_t triple = ( uint32_t( data[i] ) << 16 ) | ( uint32_t( data[i+1] ) << 8 ) | uint32_t( data[i+2] ); + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], + alphabet[ ( triple >> 6 ) & 0x3f ], alphabet[ triple & 0x3f ] }; + out.raw( quad, 4 ); + } + if ( i < length ) + { + int32_t left = length - i; + uint32_t triple = uint32_t( data[i] ) << 16; + if ( left == 2 ) { triple |= uint32_t( data[i+1] ) << 8; } + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], '=', '=' }; + if ( left == 2 ) { quad[2] = alphabet[ ( triple >> 6 ) & 0x3f ]; } + out.raw( quad, 4 ); + } + out.put( '"' ); +} + +// One UTF-8 sequence at s, or -1 when the bytes there are not one. Rejects +// the lot: a stray continuation, an overlong form, a surrogate half, and +// anything past U+10FFFF. +inline int32_t TableJsonUtf8( const char * s, int32_t remaining, int32_t * width ) +{ + unsigned char lead = (unsigned char) s[0]; + int32_t want = 0; + int32_t code = 0; + if ( lead < 0x80 ) { *width = 1; return lead; } + else if ( lead >= 0xc2 && lead <= 0xdf ) { want = 2; code = lead & 0x1f; } + else if ( lead >= 0xe0 && lead <= 0xef ) { want = 3; code = lead & 0x0f; } + else if ( lead >= 0xf0 && lead <= 0xf4 ) { want = 4; code = lead & 0x07; } + else { return -1; } + if ( remaining < want ) { return -1; } + for ( int32_t i = 1; i < want; i++ ) + { + unsigned char next = (unsigned char) s[i]; + if ( ( next & 0xc0 ) != 0x80 ) { return -1; } + code = ( code << 6 ) | ( next & 0x3f ); + } + if ( want == 3 && code < 0x800 ) { return -1; } // overlong + if ( want == 4 && code < 0x10000 ) { return -1; } // overlong + if ( code >= 0xd800 && code <= 0xdfff ) { return -1; } // a surrogate half + if ( code > 0x10ffff ) { return -1; } + *width = want; + return code; +} + +// A JSON text MUST be valid UTF-8 (RFC 8259 §8.1). The read path is +// byte-transparent — the wire imposes no encoding (§3) and a string may hold +// anything — so the WRITER is where that obligation is met: a byte that is +// not part of a well-formed sequence is written as U+FFFD, one per bad byte, +// and never raw. A text this walk writes is therefore readable by any +// conforming parser, which a raw byte would not be. The cost is stated +// plainly: for a string holding invalid UTF-8, the round trip is NOT +// byte-identical, because the alternative is emitting a text that is not +// JSON. +inline void TableJsonWriteString( TableJsonOut & out, const char * s, int32_t length ) +{ + static const char hex[] = "0123456789abcdef"; + out.put( '"' ); + for ( int32_t i = 0; i < length; i++ ) + { + unsigned char c = (unsigned char) s[i]; + switch ( c ) + { + case '"': out.raw( "\\\"", 2 ); break; + case '\\': out.raw( "\\\\", 2 ); break; + case '\b': out.raw( "\\b", 2 ); break; + case '\f': out.raw( "\\f", 2 ); break; + case '\n': out.raw( "\\n", 2 ); break; + case '\r': out.raw( "\\r", 2 ); break; + case '\t': out.raw( "\\t", 2 ); break; + default: + if ( c < 0x20 ) + { + char escape[6] = { '\\', 'u', '0', '0', hex[ c >> 4 ], hex[ c & 0xf ] }; + out.raw( escape, 6 ); + } + else if ( c < 0x80 ) + { + out.put( (char) c ); + } + else + { + int32_t width = 0; + if ( TableJsonUtf8( s + i, length - i, &width ) < 0 ) + { + out.raw( "\xef\xbf\xbd", 3 ); // U+FFFD, one per bad byte + } + else + { + out.raw( s + i, width ); + i += width - 1; + } + } + break; + } + } + out.put( '"' ); +} + +inline void TableJsonWriteUnsigned( TableJsonOut & out, uint64_t value ) +{ + char digits[24]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) ( value % 10 ) ); + value /= 10; + } while ( value != 0 ); + char text[24]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); +} + +inline void TableJsonWriteSigned( TableJsonOut & out, int64_t value ) +{ + if ( value < 0 ) + { + out.put( '-' ); + TableJsonWriteUnsigned( out, uint64_t( 0 ) - (uint64_t) value ); + return; + } + TableJsonWriteUnsigned( out, (uint64_t) value ); +} + +// A wide kind writes its raw storage as §16.2's text: a 128-bit integer as a +// decimal integer; a fixed value in WHOLE UNITS as the shortest exact decimal +// with at least one fractional digit (1.0, -0.25), the spelling the schema text +// gives a fixed default. The fraction terminates because a dyadic fraction has +// a finite decimal expansion — at most F digits. +inline void TableJsonWriteWide( TableJsonOut & out, const void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + TableJsonWide v = TableJsonWideLoad( storage, f->elem_size, is_signed ); + if ( is_signed && TableJsonWideNegative( v ) ) + { + out.put( '-' ); + v = TableJsonWideNeg( v ); + } + int frac = f->frac_bits; + TableJsonWide whole = TableJsonWideShr( v, frac ); + char digits[40]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) TableJsonWideDiv( &whole, 10 ) ); + } while ( !TableJsonWideZero( whole ) ); + char text[40]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); + if ( !TableJsonKindFixed( f->kind ) ) { return; } + out.put( '.' ); + // the fraction bits alone: v with everything at and above bit F cleared + TableJsonWide fraction = v; + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + if ( frac == 0 ) { fraction.lo = 0; } + if ( TableJsonWideZero( fraction ) ) + { + out.put( '0' ); + return; + } + while ( !TableJsonWideZero( fraction ) ) + { + // ×10: the digit is what lands at and above bit F, including the + // carry out of 128 bits when F leaves no room for it below + uint32_t carry = TableJsonWideMulAdd( &fraction, 10, 0 ); + uint64_t digit = TableJsonWideShr( fraction, frac ).lo; + if ( frac > 64 ) { digit |= uint64_t( carry ) << ( 128 - frac ); } + out.put( (char) ( '0' + (int) digit ) ); + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + } +} + +// A float writes at the SHORTEST precision that reads back as the same value +// at the field's own width, so a round trip is exact and a text stays +// readable. Non-finite values have no JSON spelling at all, and the writer +// REFUSES rather than losing one silently — the same rule measure and save +// already apply to an enum value no variant names (§5). +inline bool TableJsonWriteFloat( TableJsonOut & out, double value, bool single ) +{ + if ( !TableJsonFinite( value ) ) { return false; } + char text[64]; + int low = single ? 6 : 15; + int high = single ? 9 : 17; + int length = 0; + for ( int digits = low; ; digits++ ) + { + length = snprintf( text, sizeof( text ), "%.*g", digits, value ); + if ( length <= 0 || length >= (int) sizeof( text ) ) { return false; } + if ( digits >= high ) { break; } + // the round-trip check runs BEFORE the decimal point is normalised: + // the token still carries whatever point snprintf just produced + if ( single ) + { + if ( (double) strtof( text, NULL ) == value ) { break; } + } + else + { + if ( strtod( text, NULL ) == value ) { break; } + } + } + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int i = 0; i < length; i++ ) + { + if ( text[i] == point ) { text[i] = '.'; } + } + } + out.raw( text, length ); + return true; +} + +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration writes through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ); + +// one scalar, at one storage address: a nested object, a union, a +// vocabulary, or a number +inline bool TableJsonWriteScalar( TableJsonOut & out, const void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; None is {} + const TableUnionInfo * arms = f->arms(); + uint64_t tag = TableJsonGetRaw( (const uint8_t *) storage + arms->tag_offset, arms->tag_size ); + if ( tag == 0 ) + { + out.raw( "{}", 2 ); + return true; + } + if ( (int64_t) tag > f->enum_max ) + { + return false; // a tag no arm names, exactly as measure refuses it + } + const char * arm = f->enum_name( tag ); + // and refuse on the NAME, not merely on the bound: §16.2 says a value + // no variant NAMES is refused, so the check is the name. Writing + // whatever came back would emit "???", a spelling the reader counts + // as unknown — a silent round-trip loss in place of a refusal. + if ( !TableJsonNamed( arm ) ) { return false; } + out.put( '{' ); + out.line( depth + 1 ); + TableJsonWriteString( out, arm, (int32_t) strlen( arm ) ); + out.raw( ": ", 2 ); + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2): an arm that names + // no declaration carries the FIELD descriptor a field of its type + // would carry, offsets taken inside the union storage (§2.6), so the + // value walks through the field writer one key down. + if ( arms->arms[tag].field != NULL ) + { + if ( !TableJsonWriteField( out, storage, arms->arms[tag].field, depth + 1 ) ) + { + return false; + } + } + else if ( arms->arms[tag].table == NULL ) + { + out.raw( "null", 4 ); // a payload-free arm: the name selects it (§2.6) + } + else if ( !TableJsonWriteValue( out, (const uint8_t *) storage + arms->arms[tag].offset, arms->arms[tag].table, depth + 1 ) ) + { + return false; + } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->kind == 13 ) + { + return TableJsonWriteValue( out, storage, f->table, depth ); + } + if ( TableJsonIsEnum( f ) ) + { + uint64_t value = TableJsonGetRaw( storage, f->elem_size ); + // a value no variant names has no text spelling, exactly as it has no + // wire identity: the writer REFUSES rather than writing None over it, + // the rule measure and save already apply (docs/SPEC-TABLES.md §5) + if ( (int64_t) value > f->enum_max ) { return false; } + if ( value != 0 && f->variant_id( value ) == 0 ) { return false; } + const char * name = f->enum_name( value ); + if ( !TableJsonNamed( name ) ) { return false; } + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + return true; + } + if ( TableJsonIsFlags( f ) ) + { + uint64_t bits = TableJsonGetRaw( storage, f->elem_size ); + if ( bits == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + bool first = true; + for ( int64_t bit = 0; bit < 64; bit++ ) + { + if ( ( bits & ( uint64_t( 1 ) << bit ) ) == 0 ) { continue; } + if ( bit > f->enum_max ) + { + return false; // a bit no variant names has no text spelling + } + const char * name = f->enum_name( (uint64_t) bit ); + if ( !TableJsonNamed( name ) ) { return false; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + } + out.line( depth ); + out.put( ']' ); + return true; + } + switch ( f->kind ) + { + case 1: + out.text( TableJsonGetRaw( storage, f->elem_size ) != 0 ? "true" : "false" ); + return true; + case 10: + { + float v = 0.0f; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, (double) v, true ); + } + case 11: + { + double v = 0.0; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, v, false ); + } + case 2: case 3: case 4: case 5: + TableJsonWriteSigned( out, TableJsonGetSigned( storage, f->elem_size ) ); + return true; + default: + if ( TableJsonKindWide( f->kind ) ) + { + TableJsonWriteWide( out, storage, f ); + return true; + } + TableJsonWriteUnsigned( out, TableJsonGetRaw( storage, f->elem_size ) ); + return true; + } +} + +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ) +{ + const uint8_t * storage = (const uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonWriteMap( out, (const void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } + if ( f->kind == 17 && !f->is_array ) + { + return TableJsonWritePointer( out, storage, f, depth ); + } + if ( f->kind == 17 ) + { + // an ARRAY OF POINTERS (§2.1): the pointer row per element — the + // pointee's object in place, null, or `&node` for a shared one (§16.7) + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWritePointer( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; + } + if ( f->kind == 12 ) + { + TableJsonWriteString( out, (const char *) storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + TableJsonWriteBase64( out, storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + // one entry per SLOT, keyed by the variant that owns it, so inserting + // a variant next season moves nothing in the text either. Slot i holds + // the key i + 1: nothing is stored for None, so nothing is written for it. + out.put( '{' ); + bool first = true; + for ( int64_t slot = 0; slot < f->array_bound; slot++ ) + { + if ( !TableJsonKeyedSlotValid( f, slot ) ) { continue; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + const char * key = f->key_name( TableJsonKeyedSlotKey( slot ) ); + TableJsonWriteString( out, key, (int32_t) strlen( key ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteScalar( out, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + if ( first ) { out.raw( "}", 1 ); return true; } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->is_array ) + { + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWriteScalar( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + out.line( depth ); + out.put( ']' ); + return true; + } + return TableJsonWriteScalar( out, storage, f, depth ); +} + +// One instance's fields, in DECLARATION ORDER, defaults included — a text is +// for people and tools, and a text that elides is a text a reader has to know +// the schema to complete. `any` says whether the object is already open on +// entry — a shared node's `&node` opens it before the fields (§16.7) — and +// whether it is open on return. +inline bool TableJsonWriteFields( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth, bool & any ) +{ + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( f->guard[0] != 0 && !TableJsonGuardHolds( base, info, f->guard ) ) { continue; } + // an ABSENT optional writes no key: presence of the key IS the + // presence (§16.2), so an absent field is an absent key and nothing + // else would read back as absent + if ( f->optional && + TableJsonGetRaw( (const uint8_t *) base + f->present_offset, 1 ) == 0 ) + { + continue; + } + if ( !any ) { out.put( '{' ); } + else { out.put( ',' ); } + any = true; + out.line( depth + 1 ); + TableJsonWriteString( out, f->json, (int32_t) strlen( f->json ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, base, f, depth + 1 ) ) { return false; } + } + return true; +} + +// One instance as one object. The writer carries the reader's depth cap +// (§16.2): a pointer chain nests as deep as it is long (§16.7), and a text the +// writer produced past the cap would be a text the reader refuses. +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { return false; } + bool any = false; + if ( !TableJsonWriteFields( out, base, info, depth, any ) ) { return false; } + if ( !any ) + { + out.raw( "{}", 2 ); + return true; + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- reading ---- + +struct TableJsonIn +{ + const char * text; + int64_t size; + int64_t pos; + TableReport * report; + bool bad; // the text is not JSON: the walk stops and keeps what it placed + void * graph; // the pointered read's builder and label map (§16.7); NULL for a fixed table +}; + +inline void TableJsonSpace( TableJsonIn & in ) +{ + while ( in.pos < in.size ) + { + char c = in.text[in.pos]; + if ( c == ' ' || c == '\t' || c == '\n' || c == '\r' ) { in.pos++; continue; } + // comments are not JSON, and a walk that guessed at one would be + // reading a dialect nobody wrote down + if ( c == '/' ) { in.bad = true; } + return; + } +} + +inline char TableJsonPeek( TableJsonIn & in ) +{ + TableJsonSpace( in ); + return in.pos < in.size ? in.text[in.pos] : 0; +} + +// the shape of the value sitting at the cursor, without consuming it +inline char TableJsonValueShape( TableJsonIn & in ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return 'o'; + case '[': return 'a'; + case '"': return 's'; + case 't': case 'f': return 'b'; + case 'n': return 'z'; + case 0: return 0; + default: return 'n'; + } +} + +inline bool TableJsonLiteral( TableJsonIn & in, const char * word ) +{ + int64_t length = (int64_t) strlen( word ); + if ( in.pos + length > in.size || memcmp( in.text + in.pos, word, (size_t) length ) != 0 ) + { + in.bad = true; + return false; + } + in.pos += length; + return true; +} + +// one \uXXXX escape body; -1 when the four hex digits are not there +inline int TableJsonHex4( TableJsonIn & in ) +{ + if ( in.pos + 4 > in.size ) { return -1; } + int value = 0; + for ( int i = 0; i < 4; i++ ) + { + char c = in.text[in.pos + i]; + int digit; + if ( c >= '0' && c <= '9' ) { digit = c - '0'; } + else if ( c >= 'a' && c <= 'f' ) { digit = c - 'a' + 10; } + else if ( c >= 'A' && c <= 'F' ) { digit = c - 'A' + 10; } + else { return -1; } + value = ( value << 4 ) | digit; + } + in.pos += 4; + return value; +} + +inline int32_t TableJsonEncodeUtf8( uint32_t code, char * unit ) +{ + if ( code < 0x80 ) { unit[0] = (char) code; return 1; } + if ( code < 0x800 ) + { + unit[0] = (char) ( 0xc0 | ( code >> 6 ) ); + unit[1] = (char) ( 0x80 | ( code & 0x3f ) ); + return 2; + } + if ( code < 0x10000 ) + { + unit[0] = (char) ( 0xe0 | ( code >> 12 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( code & 0x3f ) ); + return 3; + } + unit[0] = (char) ( 0xf0 | ( code >> 18 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 12 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[3] = (char) ( 0x80 | ( code & 0x3f ) ); + return 4; +} + +// Scan one JSON string into a caller buffer. Bytes are appended ONE CODE +// POINT AT A TIME — an escape's encoding, or a UTF-8 sequence read whole — +// so a string longer than the field is clamped AT A CODE POINT BOUNDARY and +// never cut through a multi-byte character. Clamping is counted, never +// fatal, exactly as it is on the wire (§4). A NULL destination scans past a +// string without keeping it. +inline bool TableJsonScanString( TableJsonIn & in, char * out, int32_t capacity, int32_t * length ) +{ + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + int32_t placed = 0; + bool clamped = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos]; + if ( c == '"' ) { in.pos++; break; } + char unit[4]; + int32_t unit_length = 0; + if ( c == '\\' ) + { + in.pos++; + if ( in.pos >= in.size ) { in.bad = true; return false; } + char escape = in.text[in.pos++]; + switch ( escape ) + { + case '"': unit[0] = '"'; unit_length = 1; break; + case '\\': unit[0] = '\\'; unit_length = 1; break; + case '/': unit[0] = '/'; unit_length = 1; break; + case 'b': unit[0] = '\b'; unit_length = 1; break; + case 'f': unit[0] = '\f'; unit_length = 1; break; + case 'n': unit[0] = '\n'; unit_length = 1; break; + case 'r': unit[0] = '\r'; unit_length = 1; break; + case 't': unit[0] = '\t'; unit_length = 1; break; + case 'u': + { + int high = TableJsonHex4( in ); + if ( high < 0 ) { in.bad = true; return false; } + uint32_t code = (uint32_t) high; + if ( high >= 0xd800 && high <= 0xdbff && in.pos + 2 <= in.size && + in.text[in.pos] == '\\' && in.text[in.pos + 1] == 'u' ) + { + int64_t mark = in.pos; + in.pos += 2; + int low = TableJsonHex4( in ); + if ( low >= 0xdc00 && low <= 0xdfff ) + { + code = 0x10000 + ( ( (uint32_t) high - 0xd800 ) << 10 ) + ( (uint32_t) low - 0xdc00 ); + } + else + { + in.pos = mark; // a lone lead surrogate rides as itself + } + } + // a surrogate half that never found its partner has no + // UTF-8 encoding: encoding it anyway would manufacture + // CESU-8 — invalid UTF-8 — out of input that was valid + // JSON, so it reads as the replacement character + if ( code >= 0xd800 && code <= 0xdfff ) { code = 0xfffd; } + unit_length = TableJsonEncodeUtf8( code, unit ); + break; + } + default: in.bad = true; return false; + } + } + else if ( (unsigned char) c < 0x20 ) + { + in.bad = true; // a raw control character is not a JSON string body + return false; + } + else + { + // a UTF-8 sequence read WHOLE, so the clamp below can only land + // between code points. Only bytes that ACTUALLY look like + // continuations are taken: the wire imposes no encoding (§3), so + // a string may legitimately hold a stray lead byte, and one at + // the end of a text must not swallow the closing quote. + unsigned char lead = (unsigned char) c; + int32_t want = 1; + if ( ( lead & 0xe0 ) == 0xc0 ) { want = 2; } + else if ( ( lead & 0xf0 ) == 0xe0 ) { want = 3; } + else if ( ( lead & 0xf8 ) == 0xf0 ) { want = 4; } + unit[0] = c; + in.pos++; + unit_length = 1; + while ( unit_length < want && in.pos < in.size && + ( (unsigned char) in.text[in.pos] & 0xc0 ) == 0x80 ) + { + unit[unit_length++] = in.text[in.pos++]; + } + } + if ( out == NULL ) + { + placed += unit_length; // measured and not kept: a byte buffer's read sizes its node this way (§2.5) + } + else if ( placed + unit_length <= capacity ) + { + memcpy( out + placed, unit, (size_t) unit_length ); + placed += unit_length; + } + else + { + clamped = true; + } + } + if ( clamped ) { in.report->clamped++; } + if ( length != NULL ) { *length = placed; } + return true; +} + +// the numeric token at the cursor, copied out whole; false = not a number +// Scan one number, to JSON's OWN grammar (RFC 8259 §6) and not to a run of +// number-ish characters: +// +// number = [ "-" ] int [ frac ] [ exp ] +// int = "0" / ( digit1-9 *digit ) +// frac = "." 1*digit +// exp = ( "e" / "E" ) [ "-" / "+" ] 1*digit +// +// Scanning the production is what makes a typo in an authoring file a +// DIAGNOSTIC rather than a value: "1-2" scans as 1 and leaves "-2" where the +// object expects a comma, so the text is malformed — which is what §16.2 +// already promises. A permissive scan would hand "1-2" to a digit loop and +// report a clamp, and a config pipeline would never hear about it. Leading +// "+", leading zeros, ".5" and "3." are not JSON either. +inline bool TableJsonWalkNumber( TableJsonIn & in, bool * integral ) +{ + TableJsonSpace( in ); + bool whole = true; + if ( in.pos < in.size && in.text[in.pos] == '-' ) { in.pos++; } + // int: a lone zero, or a non-zero digit and any digits after it + if ( in.pos >= in.size ) { return false; } + if ( in.text[in.pos] == '0' ) + { + in.pos++; + } + else if ( in.text[in.pos] >= '1' && in.text[in.pos] <= '9' ) + { + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + } + else + { + return false; + } + // frac + if ( in.pos < in.size && in.text[in.pos] == '.' ) + { + in.pos++; + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + // exp + if ( in.pos < in.size && ( in.text[in.pos] == 'e' || in.text[in.pos] == 'E' ) ) + { + in.pos++; + if ( in.pos < in.size && ( in.text[in.pos] == '-' || in.text[in.pos] == '+' ) ) { in.pos++; } + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + *integral = whole; + return true; +} + +// the same production, with the token kept for conversion +inline bool TableJsonScanNumber( TableJsonIn & in, char * token, int32_t capacity, int32_t * length, bool * integral ) +{ + TableJsonSpace( in ); + int64_t start = in.pos; + if ( !TableJsonWalkNumber( in, integral ) ) { return false; } + int64_t count = in.pos - start; + if ( count <= 0 || count >= capacity ) { return false; } + memcpy( token, in.text + start, (size_t) count ); + token[count] = 0; + *length = (int32_t) count; + return true; +} + +// the token's exact double, through the runtime's own converter — which +// speaks the LOCALE's decimal point, so the token crosses back over that +// character on its way in +inline double TableJsonTokenDouble( const char * token, int32_t length, bool single ) +{ + char work[kTableJsonMaxNumber]; + memcpy( work, token, (size_t) length ); + work[length] = 0; + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int32_t i = 0; i < length; i++ ) + { + if ( work[i] == '.' ) { work[i] = point; } + } + } + if ( single ) { return (double) strtof( work, NULL ); } + return strtod( work, NULL ); +} + +// the token's exact integer, parsed digit by digit so no width and no +// locale can move it. Saturation is reported as a clamp, the wire's rule for +// a value outside what the reader can hold (§4). +inline int64_t TableJsonTokenInteger( const char * token, int32_t length, bool is_signed, bool * saturated ) +{ + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) + { + negative = token[i] == '-'; + i++; + } + uint64_t magnitude = 0; + bool over = false; + for ( ; i < length; i++ ) + { + uint64_t digit = (uint64_t) ( token[i] - '0' ); + if ( magnitude > ( UINT64_MAX - digit ) / 10 ) { over = true; break; } + magnitude = magnitude * 10 + digit; + } + if ( !is_signed ) + { + // -0 IS zero, and clamping it would report an event that did not + // happen; only a real negative magnitude is out of range here + if ( negative ) { *saturated = magnitude != 0; return 0; } + if ( over ) { *saturated = true; return (int64_t) UINT64_MAX; } + *saturated = false; + return (int64_t) magnitude; + } + if ( negative ) + { + if ( over || magnitude > ( uint64_t( 1 ) << 63 ) ) { *saturated = true; return INT64_MIN; } + *saturated = false; + if ( magnitude == ( uint64_t( 1 ) << 63 ) ) { return INT64_MIN; } + return -(int64_t) magnitude; + } + if ( over || magnitude > (uint64_t) INT64_MAX ) { *saturated = true; return INT64_MAX; } + *saturated = false; + return (int64_t) magnitude; +} + +// A number token into a wide kind's raw storage (docs/SPEC-TABLES.md §16.2). A +// 128-bit integer takes any token whose VALUE is integral; a fixed field any +// token whose value is EXACTLY representable in its Q I.F — a finer fraction +// is the wrong shape for the field, counted as a kind mismatch and never +// rounded, the rule SPEC.md §4.6 gives a fixed default. A magnitude past 128 +// bits saturates and counts as a clamp, as an int64 field saturates at +// INT64_MAX; the declared range clamps after it, on the RAW scale, as it does +// for every bounded scalar. +// +// The token is normalized to its digits with the decimal point after "point" +// of them. An integer part past 40 digits is above 2^128 whatever the digits +// are, and a value below 10^-40 is finer than 2^-127, the finest fraction any +// F can spell — so outside that band the answer is known without the +// arithmetic, and a token spelling 1e999999999 costs nothing to refuse. +inline bool TableJsonReadWide( TableJsonIn & in, const char * token, int32_t length, void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + int frac = f->frac_bits; + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { negative = token[i] == '-'; i++; } + const char * int_digits = token + i; + int32_t int_len = 0; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { int_len++; i++; } + const char * frac_digits = token + i; + int32_t frac_len = 0; + if ( i < length && token[i] == '.' ) + { + i++; + frac_digits = token + i; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { frac_len++; i++; } + } + int64_t exp = 0; + if ( i < length && ( token[i] == 'e' || token[i] == 'E' ) ) + { + i++; + bool exp_negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { exp_negative = token[i] == '-'; i++; } + while ( i < length && token[i] >= '0' && token[i] <= '9' ) + { + if ( exp < 100000 ) { exp = exp * 10 + ( token[i] - '0' ); } + i++; + } + if ( exp_negative ) { exp = -exp; } + } + // the digits, with the point after "point" of them; leading and trailing + // zeros stripped. digit( k ) reads the k-th of the int and frac runs. + int32_t start = 0, end = int_len + frac_len; + int64_t point = int_len + exp; + while ( start < end && ( start < int_len ? int_digits[start] : frac_digits[start - int_len] ) == '0' ) { start++; point--; } + while ( end > start && ( end - 1 < int_len ? int_digits[end - 1] : frac_digits[end - 1 - int_len] ) == '0' ) { end--; } + + TableJsonWide raw = { 0, 0 }; + bool saturated = false; + TableJsonWide signed_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) >> 1 }; + TableJsonWide signed_min = { 0, uint64_t( 1 ) << 63 }; + TableJsonWide unsigned_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) }; + if ( start == end ) + { + // zero, and -0 IS zero + } + else if ( point > 40 ) + { + saturated = true; + if ( !negative ) { raw = is_signed ? signed_max : unsigned_max; } + else if ( is_signed ) { raw = signed_min; } + } + else if ( point < -40 ) + { + in.report->kind_mismatch++; // finer than any F can spell + return true; + } + else + { + // the fraction FIRST, so an inexact value is the wrong shape whatever + // its magnitude: its digits, with the zeros a negative point puts in + // front, doubled F times; each doubling's carry is the next bit, and + // the value is exact iff nothing is left after the last one + char fd[kTableJsonMaxNumber + 48]; + int32_t fn = 0; + for ( int64_t z = point; z < 0; z++ ) { fd[fn++] = 0; } + for ( int32_t k = (int32_t) ( point > 0 ? point : 0 ) + start; k < end; k++ ) + { + fd[fn++] = (char) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ); + } + TableJsonWide fraction = { 0, 0 }; + for ( int b = 0; b < frac; b++ ) + { + int carry = 0; + for ( int32_t k = fn - 1; k >= 0; k-- ) + { + int d = fd[k] * 2 + carry; + fd[k] = (char) ( d % 10 ); + carry = d / 10; + } + fraction = TableJsonWideShl( fraction, 1 ); + fraction.lo |= (uint64_t) carry; + } + for ( int32_t k = 0; k < fn; k++ ) + { + if ( fd[k] != 0 ) + { + in.report->kind_mismatch++; + return true; + } + } + // then the whole part, saturating past 128 bits + TableJsonWide whole = { 0, 0 }; + for ( int64_t k = start; k < start + point && !saturated; k++ ) + { + uint32_t digit = k < end ? (uint32_t) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ) : 0; + if ( TableJsonWideMulAdd( &whole, 10, digit ) != 0 ) { saturated = true; } + } + if ( !saturated && frac > 0 && !TableJsonWideZero( TableJsonWideShr( whole, 128 - frac ) ) ) { saturated = true; } + if ( !saturated ) + { + raw = TableJsonWideShl( whole, frac ); + raw.lo |= fraction.lo; + raw.hi |= fraction.hi; + } + if ( is_signed ) + { + if ( !saturated && !negative && TableJsonWideNegative( raw ) ) { saturated = true; } + if ( !saturated && negative && TableJsonWideCompare( raw, signed_min, false ) > 0 ) { saturated = true; } + if ( saturated ) { raw = negative ? signed_min : signed_max; } + else if ( negative ) { raw = TableJsonWideNeg( raw ); } + } + else + { + if ( saturated ) { raw = unsigned_max; } + if ( negative && !TableJsonWideZero( raw ) ) { raw.lo = 0; raw.hi = 0; saturated = true; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->wide != NULL ) + { + TableJsonWide lo = { f->wide->lo[0], f->wide->lo[1] }; + TableJsonWide hi = { f->wide->hi[0], f->wide->hi[1] }; + if ( TableJsonWideCompare( raw, lo, is_signed ) < 0 ) { raw = lo; in.report->clamped++; } + else if ( TableJsonWideCompare( raw, hi, is_signed ) > 0 ) { raw = hi; in.report->clamped++; } + } + TableJsonWideStore( storage, f->elem_size, raw ); + return true; +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ); + +inline bool TableJsonSkipContainer( TableJsonIn & in, char close, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; // the opening bracket + bool first = true; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == close ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + if ( close == '}' ) + { + // the key is kept, because a skipped OBJECT may still be a + // pointer's: an `&node` opening it names a node the storage could + // not hold, and the numbering has to survive the drop (§16.7). + // Anywhere but first, the prefix is the reserved key out of place + // — in a pointered unit; a fixed unit skips the value whole. + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( key[0] == '&' && in.graph != NULL ) + { + if ( !first ) { in.report->malformed = true; in.bad = true; return false; } + if ( !TableJsonSkippedAmpersand( in, key, depth ) ) { return false; } + first = false; + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } + } + first = false; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return TableJsonSkipContainer( in, '}', depth ); + case '[': return TableJsonSkipContainer( in, ']', depth ); + case '"': return TableJsonScanString( in, NULL, 0, NULL ); + case 't': return TableJsonLiteral( in, "true" ); + case 'f': return TableJsonLiteral( in, "false" ); + case 'n': return TableJsonLiteral( in, "null" ); + case 0: in.bad = true; return false; + default: + { + // consumed, never converted: skipping needs no buffer, and this + // is the one walk a hostile text drives to the depth cap. It is + // the SAME production the value path scans, so an unknown key + // cannot smuggle past a number a named key would refuse. + bool integral = false; + if ( !TableJsonWalkNumber( in, &integral ) ) { in.bad = true; return false; } + return true; + } + } +} + +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration reads through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ); + +// place one scalar at one storage address +inline bool TableJsonReadScalar( TableJsonIn & in, void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; {} is None, and + // two keys is a text this walk will not guess at + const TableUnionInfo * arms = f->arms(); + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, 0 ); + if ( TableJsonPeek( in ) == '}' ) { in.pos++; return true; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t tag = 0; + for ( int64_t t = 1; t <= f->enum_max; t++ ) + { + if ( strcmp( f->enum_name( (uint64_t) t ), key ) == 0 ) { tag = t; break; } + } + if ( tag == 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + void * payload = (uint8_t *) storage + arms->arms[tag].offset; + const TableFieldInfo * arm = arms->arms[tag].field; + bool placed = true; + if ( arm != NULL ) + { + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2). A value of + // the wrong shape for that row is a KIND MISMATCH: the union + // reads None, the event is counted, and the enclosing object + // continues — the rule a FIELD's value lives under, one key + // down. A pointer arm's null is a null pointer, not a shape + // error, exactly as a pointer field's is (§16.7). + char got = TableJsonValueShape( in ); + if ( arm->kind == 17 && !arm->is_array && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + memset( payload, 0, (size_t) arms->arms[tag].size ); + } + else if ( got != TableJsonShape( arm ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( arm->kind == 17 && !arm->is_array ) + { + // A POINTER ARM'S VALUE IS THE POINTEE IN PLACE, or a + // node reference to one (§16.7) — the read a pointer + // FIELD takes, which is not the scalar walk + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadPointer( in, payload, arm, depth + 1 ) ) { return false; } + } + else + { + // SELECTION ZERO-ESTABLISHES THE ARM (SPEC §5): an arm + // takes no specified default, so zero is the establish + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadField( in, storage, arm, depth + 1 ) ) { return false; } + } + } + else if ( arms->arms[tag].table != NULL ) + { + if ( TableJsonValueShape( in ) != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else + { + arms->arms[tag].table->reset( payload ); + if ( !TableJsonReadTable( in, payload, arms->arms[tag].table, depth + 1 ) ) { return false; } + } + } + else + { + // A PAYLOAD-FREE ARM'S VALUE IS null (§2.6): the arm name + // selects it and there is nothing to place + if ( TableJsonValueShape( in ) != 'z' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( !TableJsonLiteral( in, "null" ) ) + { + return false; + } + } + if ( placed ) + { + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, (uint64_t) tag ); + } + } + char c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; // a second key: a one-of with two arms is not a value + return false; + } + if ( f->kind == 13 ) + { + f->table->reset( storage ); + return TableJsonReadTable( in, storage, f->table, depth + 1 ); + } + if ( TableJsonIsEnum( f ) ) + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + for ( int64_t v = 0; v <= f->enum_max; v++ ) + { + if ( strcmp( f->enum_name( (uint64_t) v ), name ) == 0 ) + { + TableJsonSetRaw( storage, f->elem_size, (uint64_t) v ); + return true; + } + } + // a name this build cannot name reads as None and counts as unknown, + // exactly as an unknown variant id does on the wire (§4) + TableJsonSetRaw( storage, f->elem_size, 0 ); + in.report->unknown++; + return true; + } + if ( TableJsonIsFlags( f ) ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + uint64_t bits = 0; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( c != '"' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + bool found = false; + for ( int64_t bit = 0; bit <= f->enum_max; bit++ ) + { + if ( strcmp( f->enum_name( (uint64_t) bit ), name ) == 0 ) + { + bits |= uint64_t( 1 ) << bit; + found = true; + break; + } + } + if ( !found ) { in.report->unknown++; } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + TableJsonSetRaw( storage, f->elem_size, bits ); + return true; + } + if ( f->kind == 1 ) + { + char c = TableJsonPeek( in ); + if ( c == 't' ) { if ( !TableJsonLiteral( in, "true" ) ) { return false; } TableJsonSetRaw( storage, f->elem_size, 1 ); return true; } + if ( !TableJsonLiteral( in, "false" ) ) { return false; } + TableJsonSetRaw( storage, f->elem_size, 0 ); + return true; + } + char token[kTableJsonMaxNumber]; + int32_t length = 0; + bool integral = false; + if ( !TableJsonScanNumber( in, token, kTableJsonMaxNumber, &length, &integral ) ) + { + in.bad = true; + return false; + } + if ( TableJsonKindWide( f->kind ) ) + { + return TableJsonReadWide( in, token, length, storage, f ); + } + if ( f->kind == 10 || f->kind == 11 ) + { + bool single = f->kind == 10; + double value = TableJsonTokenDouble( token, length, single ); + // A magnitude the field's format cannot hold is the WRONG SHAPE for + // the kind, and it never reaches storage: 1e400 is not a float64 and + // 1e300 is not a float32. Storing the infinity the conversion + // produced would leave an instance this walk called CLEAN that + // ToJsonMeasure then refuses forever (a non-finite float has no JSON + // spelling), and §16.1's one invariant is that a text which reads + // clean writes back. + if ( !TableJsonFinite( value ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( f->has_range ) + { + if ( value < f->range_min ) { value = f->range_min; in.report->clamped++; } + else if ( value > f->range_max ) { value = f->range_max; in.report->clamped++; } + } + if ( single ) + { + float narrow = (float) value; + if ( !TableJsonFinite( (double) narrow ) ) + { + in.report->kind_mismatch++; + return true; + } + memcpy( storage, &narrow, sizeof( narrow ) ); + } + else + { + memcpy( storage, &value, sizeof( value ) ); + } + return true; + } + // JSON HAS ONE NUMBER TYPE. 2.0 IS the integer 2 and 1e3 IS 1000, and a + // library that round-trips numbers through a double emits them that way — + // this walker's own float writer emits 1e+21. So an integer field takes + // any number whose VALUE is integral, however it was spelled; only a + // genuinely fractional value is the wrong shape for it. + bool is_signed = f->kind >= 2 && f->kind <= 5; + bool saturated = false; + int64_t value = 0; + if ( integral ) + { + value = TableJsonTokenInteger( token, length, is_signed, &saturated ); + } + else + { + double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( is_signed ) + { + if ( d >= 9223372036854775808.0 ) { value = INT64_MAX; saturated = true; } + else if ( d < -9223372036854775808.0 ) { value = INT64_MIN; saturated = true; } + else if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) d; } + } + else + { + if ( d < 0.0 ) + { + // a negative for an unsigned field clamps to zero, as the + // exact digit path already does + if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + value = 0; + saturated = true; + } + else if ( d >= 18446744073709551616.0 ) { value = (int64_t) UINT64_MAX; saturated = true; } + else if ( d != (double) (uint64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) (uint64_t) d; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->has_range ) + { + if ( (double) value < f->range_min ) { value = (int64_t) f->range_min; in.report->clamped++; } + else if ( (double) value > f->range_max ) { value = (int64_t) f->range_max; in.report->clamped++; } + } + // the field's own storage width is the last bound: a value past it + // clamps rather than wrapping, which is what the wire does too + if ( f->elem_size < 8 ) + { + if ( is_signed ) + { + int64_t high = ( int64_t( 1 ) << ( f->elem_size * 8 - 1 ) ) - 1; + int64_t low = -high - 1; + if ( value > high ) { value = high; in.report->clamped++; } + else if ( value < low ) { value = low; in.report->clamped++; } + } + else + { + uint64_t high = ( uint64_t( 1 ) << ( f->elem_size * 8 ) ) - 1; + if ( value < 0 ) { value = 0; in.report->clamped++; } + else if ( (uint64_t) value > high ) { value = (int64_t) high; in.report->clamped++; } + } + } + // at eight bytes the storage IS the parser's width, and an unsigned value + // past INT64_MAX rides here as a negative int64 by design — the token + // parser already turned a NEGATIVE token for an unsigned field into a + // clamped zero, so there is nothing left to bound. + TableJsonSetRaw( storage, f->elem_size, (uint64_t) value ); + return true; +} + +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ) +{ + uint8_t * storage = (uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonReadMap( in, (void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + + if ( f->kind == 12 ) + { + int32_t length = 0; + if ( !TableJsonScanString( in, (char *) storage, f->array_bound, &length ) ) { return false; } + storage[length] = 0; + TableJsonSetCount( base, f, length ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + // base64 decodes STRAIGHT INTO the field's storage, six bits at a + // time — no window, no temporary, so a bytes(N) of any declared + // extent reads the same way. A base64 body carries no escapes, so a + // backslash in one is simply not an alphabet character. + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + memset( storage, 0, (size_t) f->array_bound ); + TableJsonSetCount( base, f, 0 ); + const char * alphabet = TableJsonBase64Alphabet(); + int32_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + bool clamped = false; + bool malformed = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + const char * at = c != 0 ? strchr( alphabet, c ) : NULL; + if ( at == NULL ) { malformed = true; continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( at - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < f->array_bound ) + { + storage[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); + } + else + { + clamped = true; + } + } + } + if ( malformed ) + { + // a body that is not base64 is the wrong shape for the kind: the + // field keeps its default and the event is counted + in.report->kind_mismatch++; + return true; + } + if ( clamped ) { in.report->clamped++; } + TableJsonSetCount( base, f, placed ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + // every slot back to its declared defaults first, so a key the text + // omits keeps them and a repeated field key cannot leave an earlier + // occurrence's slots standing + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + void * slot = storage + (int64_t) i * f->elem_size; + if ( f->kind == 13 ) { f->table->reset( slot ); } + else { memset( slot, 0, (size_t) f->elem_size ); } + } + char shape = TableJsonElementShape( f ); + // A KEYED OBJECT'S KEYS ARE KEYS: a variant named twice is a duplicate + // key like any other, last-wins and counted (§16.2). Tracked the way + // a table's own field keys are — a bounded, allocation-free bitmask; + // a vocabulary wider than this still reads, its repeats simply stop + // being counted. + uint64_t seen[8] = {}; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t slot = -1; + for ( int64_t v = 0; v < f->array_bound; v++ ) + { + // nothing is stored for None, so "None" finds no slot and is + // an unknown key like any other name this reader cannot place + if ( !TableJsonKeyedSlotValid( f, v ) ) { continue; } + if ( strcmp( f->key_name( TableJsonKeyedSlotKey( v ) ), key ) == 0 ) { slot = v; break; } + } + if ( slot >= 0 && slot < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( slot & 63 ); + if ( ( seen[slot >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[slot >> 6] |= bit; + } + if ( slot < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; + } + if ( f->is_array ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + // LAST WINS has to be true of a repeated ARRAY key too, and it is + // wire-visible: a fixed array writes every slot, so a second, shorter + // occurrence overlaying a prefix would leave the first occurrence's + // tail standing. The field goes back to its declared defaults before + // this occurrence's elements are placed — the re-establishment a nested + // table and a union arm already get. A table element's defaults are + // its own (the reset hook); every other element kind's storage + // default is zero, which is what the generated array declares. + if ( f->kind == 13 ) + { + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + f->table->reset( storage + (int64_t) i * f->elem_size ); + } + } + else + { + memset( storage, 0, (size_t) f->array_bound * (size_t) f->elem_size ); + } + TableJsonSetCount( base, f, 0 ); + int32_t placed = 0; + char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( placed >= f->array_bound ) + { + // more elements than the reader's bound: the bounded prefix + // is kept and the excess counts, the wire's rule (§4) + in.report->clamped++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( f->kind == 17 ) + { + // an element of an ARRAY OF POINTERS (§2.1): null is a null slot, an + // object is the pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( storage + (int64_t) placed * f->elem_size, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed++; + } + else + { + if ( !TableJsonReadScalar( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + // a fixed array's tail keeps the defaults the prefill left there, + // exactly as a short wire count does + TableJsonSetCount( base, f, placed ); + return true; + } + return TableJsonReadScalar( in, storage, f, depth ); +} + +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ); + +// ONE table object: keys are field keys, unknown ones are skipped and +// counted, a repeated key is last-wins and counted. The instance is already +// at its declared defaults when this is entered, so a key the text never +// mentions keeps the default an absent field takes on the wire (§4). +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + return TableJsonReadTableKeys( in, base, info, depth, NULL ); +} + +// The keys of an object whose brace is already consumed. A pointer's object +// opens the same way a table's does, but its FIRST key may be `&node` (§16.7) +// and the adapter that reads it has to scan the key to know — so it hands the +// key it scanned in as `first_key`, with the colon consumed, and this places +// it before scanning the rest. +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ) +{ + // duplicate tracking, bounded and allocation-free: a table with more + // fields than this still reads, its repeats simply stop being counted + uint64_t seen[8] = {}; + for ( ;; ) + { + char key[kTableJsonMaxKey]; + char c = 0; + if ( first_key != NULL ) + { + memcpy( key, first_key, strlen( first_key ) + 1 ); // scanned into a buffer this size by the caller + first_key = NULL; + } + else + { + c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + } + int32_t index = -1; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + if ( strcmp( info->fields[i].json, key ) == 0 ) { index = i; break; } + } + if ( key[0] == '&' ) + { + // THE AMPERSAND PREFIX IS RESERVED TO THE FORM (docs/SPEC-TABLES.md + // §16.7). No declaration may take a key beginning with it, so this + // is never a field this build lacks — it is the sharing construct + // somewhere it cannot stand: `&node` is the FIRST key of a pointer's + // object and nothing else, and the adapter that reads a pointer + // has consumed it before these keys are read. MALFORMED, refused + // and counted; never counted as unknown, never skipped. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( index < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + const TableFieldInfo * f = &info->fields[index]; + if ( index < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( index & 63 ); + if ( ( seen[index >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[index >> 6] |= bit; + } + // PRESENCE OF THE KEY IS THE PRESENCE (§16.2): reaching this line + // is the key being present, so an optional is set present + // whatever its value — with one exception the page names: a JSON + // null, which reads as ABSENT rather than as a value. + char got = TableJsonValueShape( in ); + if ( f->kind == 17 && !f->is_array ) + { + // a pointer: null is a null pointer, an object is the pointee + // in place or an `&node` reference to one (§16.7), a string is + // a BYTE BUFFER's bytes (§2.5), and anything else is the wrong + // shape for the kind + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) base + f->offset, f->elem_size, 0 ); + } + else if ( got != TableJsonShape( f ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) base + f->offset, f, depth ) ) + { + return false; + } + } + else if ( f->optional && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + // absent, and back at its defaults: a repeated key whose last + // occurrence is null must not leave an earlier value standing + if ( f->table != NULL ) { f->table->reset( (uint8_t *) base + f->offset ); } + else { memset( (uint8_t *) base + f->offset, 0, (size_t) f->elem_size ); } + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 0 ); + } + else + { + if ( got != TableJsonShape( f ) ) + { + // the wrong JSON type for the kind: skipped, never coerced + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, base, f, depth ) ) + { + return false; + } + if ( f->optional ) + { + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 1 ); + } + } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +// ---- the two entry points the per-table wrappers name ---- + +inline bool TableJsonRead( void * value, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = NULL; + info->reset( value ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, value, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +inline int64_t TableJsonWrite( const void * value, const TableTypeInfo * info, char * buffer, int64_t capacity ) +{ + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = NULL; + if ( !TableJsonWriteValue( out, value, info, 0 ) ) { return -1; } + // THE CANONICAL TEXT ENDS WITH EXACTLY ONE NEWLINE (docs/SPEC-TABLES.md + // §16.1). Every writer emits it — this walk, the C# walk and + // "schema unpack" — and every reader accepts a text with or without one, + // because the trailing whitespace a read already skips is what makes the + // two the same text. It is a byte of the FORM rather than a file + // convention: a text that is written to a file, pasted into a diff and + // handed back through a pipe has to be one text in all three places, and a + // buffer whose last byte is a closing brace is the one shape that is not. + out.put( '\n' ); + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json walk: end ---- + +// ---- json graph walk: begin ---- +// +// THE VARIABLE CLASS's half of the text form (docs/SPEC-TABLES.md §16.7). The +// walk above places every kind but one; this defines the three adapters it +// calls for that one, and the two entry points a pointered table's wrappers +// name. The text is the fixed class's — a pointee is an object in place — and a +// node named more than once carries `&node`: defined once, with its fields, +// and referenced after by `{ "&node": N }` alone. + +// ---- the identity map ---- +// +// ONE map shape serves both directions. Writing keys it by a node's ADDRESS and +// counts the slots that name the node, so the second pass knows at a node's +// first occurrence whether it will be named again; reading keys it by the +// text's own label and answers the node it defined. Open addressing, a +// multiply-shift hash and quadrupling growth — TablePackMap's shape (§6.2), on +// the same terms: proportional to nodes, never to bytes, on the authoring +// side, and released before the call returns. + +struct TableJsonGraphEntry +{ + uint64_t key; // a node's address (write) or a label (read); 0 is an empty slot + int64_t count; // write: how many slots name this node + int64_t label; // write: the `&node` label assigned at its first write, 0 until then + uint8_t open; // the descent is still open: a reference here is a cycle (write), a self-reference (read) + uint32_t node; // read: the node's arena offset; 0 for a definition the reader dropped + const TableTypeInfo * type; // read: the node's table; NULL for a dropped one +}; + +struct TableJsonGraphMap +{ + TableJsonGraphEntry * entries; + int64_t capacity; // a power of two, or zero while empty + int64_t count; + TableAllocator allocator; // the caller's pair (§6.5): the builder's on read, the one handed to ToJson on write +}; + +inline void TableJsonGraphMapInit( TableJsonGraphMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TableJsonGraphMapShutdown( TableJsonGraphMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TableJsonGraphMapInit( map, map.allocator ); +} + +inline int64_t TableJsonGraphMapSlot( const TableJsonGraphMap & map, uint64_t key ) +{ + uint64_t hash = key * 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != 0 && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TableJsonGraphEntry * TableJsonGraphMapFind( TableJsonGraphMap & map, uint64_t key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +inline bool TableJsonGraphMapGrow( TableJsonGraphMap & map ) +{ + TableJsonGraphMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 64; + grown.count = 0; + grown.entries = (TableJsonGraphEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TableJsonGraphEntry ) ); // zeroed, by the pair's contract + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == 0 ) { continue; } + grown.entries[ TableJsonGraphMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// the entry for a key, made if it was not there; `taken` says which. NULL is the +// allocator refusing, and the walk refuses with it. +inline TableJsonGraphEntry * TableJsonGraphMapReach( TableJsonGraphMap & map, uint64_t key, bool & taken ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TableJsonGraphMapGrow( map ) ) { return NULL; } + } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + taken = entry->key != key; + if ( taken ) + { + entry->key = key; + map.count++; + } + return entry; +} + +// ---- reading: into a builder ---- + +struct TableJsonGraphIn +{ + TableWorker * worker; // where every node comes from + TableJsonGraphMap labels; // a label -> the node it defined +}; + +// `&node`'s value, the LABEL: a positive integer spelled as one — digits, no sign, no +// fraction, no exponent, no leading zero (§16.7). Anything else is malformed. +inline bool TableJsonScanLabel( TableJsonIn & in, uint64_t & label ) +{ + TableJsonSpace( in ); + if ( in.pos >= in.size || in.text[in.pos] < '1' || in.text[in.pos] > '9' ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + uint64_t value = 0; + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) + { + uint64_t digit = (uint64_t) ( in.text[in.pos] - '0' ); + if ( value > ( UINT64_MAX - digit ) / 10 ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + value = value * 10 + digit; + in.pos++; + } + label = value; + return true; +} + +// A BYTE BUFFER's text (docs/SPEC-TABLES.md §2.5, §16.2): a string. For a +// *string the string's bytes become the blob; for a *bytes the string is base64 +// and its decoded bytes do. The blob is allocated at EXACTLY the decoded +// length — the string is scanned once without keeping it to learn the length, +// and once into the node — so a blob of any size reads with no window and no +// bound to clamp against. A *bytes body that is not base64 is the wrong shape +// for the kind: the reference stays null and the event is counted. +inline bool TableJsonReadBlob( TableJsonIn & in, void * slot, const TableFieldInfo * f ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + TableRef * ref = (TableRef *) slot; + ref->value = 0; + if ( strcmp( f->type_name, "string" ) == 0 ) + { + const int64_t mark = in.pos; + int32_t length = 0; + if ( !TableJsonScanString( in, NULL, 0, &length ) ) { return false; } + in.pos = mark; + char * data = TableStringEmplace( *graph->worker, *ref, NULL, (int64_t) length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int32_t placed = 0; + return TableJsonScanString( in, data, length, &placed ); + } + // base64: the alphabet characters decide the length, six bits apiece + const char * alphabet = TableJsonBase64Alphabet(); + const int64_t mark = in.pos + 1; + int64_t symbols = 0; + bool malformed = false; + in.pos++; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + if ( c == 0 || strchr( alphabet, c ) == NULL ) { malformed = true; continue; } + symbols++; + } + if ( malformed ) + { + in.report->kind_mismatch++; + return true; + } + const int64_t length = ( symbols * 6 ) / 8; + uint8_t * data = TableBytesEmplace( *graph->worker, *ref, length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int64_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + for ( int64_t at = mark; ; at++ ) + { + char c = in.text[at]; + if ( c == '"' ) { break; } + const char * symbol = c != '=' ? strchr( alphabet, c ) : NULL; + if ( symbol == NULL ) { continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( symbol - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < length ) { data[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); } + } + } + return true; +} + +// A pointer's object. Its FIRST key decides what it is: `&node` naming a label not +// yet defined, with fields after it, is a DEFINITION; `&node` naming one already +// defined, alone, is a REFERENCE; any other key is a node named once, its +// object in place. The node comes from the +// builder's arena, and the slot holds its arena offset (§6.3). A pointer whose +// target is a BYTE BUFFER — no table — takes a string instead (§2.5). +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( f->table == NULL ) { return TableJsonReadBlob( in, slot, f ); } + // the pointee nests one level down, exactly as a by-value table does, and + // takes the same cap: a chain nests as deep as it is long (§16.7) + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + char c = TableJsonPeek( in ); + if ( c == '}' ) + { + // an empty object: a node at its defaults, named once + in.pos++; + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + return true; + } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( strcmp( key, "&node" ) != 0 ) + { + // a node named once: the pointee's object in place, and this key is + // its first field — unless it is the reserved prefix under a spelling + // this form does not have, which ReadTableKeys refuses + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return TableJsonReadTableKeys( in, node, f->table, depth + 1, key ); + } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->labels, label, taken ); + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + // ONE SPELLING, and what follows the label says which half it is: fields + // after a label the text has not defined DEFINE it, and a label alone that + // the text has defined REFERS to it. The other two are malformed — a label + // alone that the text never defined, which would otherwise read as a default + // node under a silent report, and a field after a label already defined, + // which would be a second definition. That is what keeps a typo loud. + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + bool bare = c == '}'; + if ( bare == taken ) { in.report->malformed = true; in.bad = true; return false; } + if ( bare ) + { + // A REFERENCE. A label is defined when its object CLOSES, so a + // reference met inside its own definition — at any depth of by-value + // nesting — names a node whose descent is still open: the cycle the + // wire refuses (§3.1), refused here where it is written. A definition + // the reader dropped names no node, so the slot stays null with + // nothing more counted — the drop was counted where it happened. A + // node of another table than the slot declares is a kind mismatch, as + // on the wire. + in.pos++; + if ( entry->open != 0 ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + if ( entry->type == NULL ) + { + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + if ( entry->type != f->table ) + { + memcpy( slot, &ref, sizeof( ref ) ); + in.report->kind_mismatch++; + return true; + } + ref.value = (int64_t) entry->node; + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + // A DEFINITION: the node is allocated, the label is its, and the keys after + // `&node` are its fields. The entry is OPEN until the object closes, so a + // reference to the label from inside the node's own fields is refused as + // the cycle it is; the node and its table are filled in at the close. + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + entry->open = 1; + if ( !TableJsonReadTableKeys( in, node, f->table, depth + 1, NULL ) ) { return false; } + entry = TableJsonGraphMapFind( graph->labels, label ); // the map may have grown under the descent + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + memcpy( &ref, slot, sizeof( ref ) ); + entry->node = (uint32_t) ref.value; + entry->type = f->table; + entry->open = 0; + return true; +} + +// An `&`-prefixed key opening an object the walk is SKIPPING — a value past an +// array's bound, an unknown key's value, a value of the wrong shape. A +// definition in there still takes its label, so the numbering survives whatever +// the storage could not hold (§16.7): the label is registered with no node, and a +// reference to it reads null. Any other prefixed key is the reserved prefix +// out of place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL || strcmp( key, "&node" ) != 0 ) { in.report->malformed = true; in.bad = true; return false; } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + if ( TableJsonGraphMapReach( graph->labels, label, taken ) == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return true; // a fresh entry is node 0, type NULL: a definition with no node +} + +// ---- writing: from a region's const root ---- + +struct TableJsonGraphOut +{ + TableJsonGraphMap nodes; // a node's address -> how many slots name it, and its `&node` once assigned + bool counting; // PASS ONE: count the references, refuse a cycle, emit nothing + int64_t next_label; +}; + +// The node a slot names: null as `null`, a node named once as its object in +// place, and a node named more than once under the construct. Which of the +// last two it is was learned in pass one; pass two spells it. +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphOut * graph = (TableJsonGraphOut *) out.graph; + if ( graph == NULL ) { return false; } + const void * node = f->resolve( slot ); + if ( node == NULL ) + { + out.raw( "null", 4 ); + return true; + } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->nodes, (uint64_t) (uintptr_t) node, taken ); + if ( entry == NULL ) { return false; } + if ( f->table == NULL ) + { + // A BYTE BUFFER (§2.5, §16.7): its text is a string, which has no + // first key to carry `&node`, so a blob named from more than one + // slot has no spelling this form can carry and the graph is refused — + // as a shared node with nothing to write is. A blob named once is its + // bytes in place: base64 for a *bytes, the string itself for a *string. + if ( graph->counting ) { entry->count++; return true; } + if ( entry->count > 1 ) { return false; } + const TableBlob * blob = (const TableBlob *) node; + if ( blob->length > (uint32_t) 0x7fffffff ) { return false; } + if ( strcmp( f->type_name, "string" ) == 0 ) { TableJsonWriteString( out, (const char *) ( blob + 1 ), (int32_t) blob->length ); } + else { TableJsonWriteBase64( out, (const uint8_t *) ( blob + 1 ), (int32_t) blob->length ); } + return true; + } + if ( graph->counting ) + { + // PASS ONE: one visit per node, every slot that names it counted, and + // a reference to a node whose descent is still open is a cycle — + // refused here as the wire refuses it (§3.1) + entry->count++; + if ( !taken ) { return entry->open == 0; } + entry->open = 1; + if ( !TableJsonWriteValue( out, node, f->table, depth ) ) { return false; } + entry = TableJsonGraphMapFind( graph->nodes, (uint64_t) (uintptr_t) node ); // the map may have grown under the descent + if ( entry == NULL ) { return false; } + entry->open = 0; + return true; + } + // PASS TWO: a node named once is its object in place; a node named more + // than once is DEFINED at its first occurrence — `&node` first, then its + // fields — and REFERENCED by `&node` alone after that, spelled the same way at + // every site. Labels run from 1 in first-write order and are the text's own, + // so a stray number in a hand-edited text is most often one never defined. + if ( entry->count <= 1 ) + { + return TableJsonWriteValue( out, node, f->table, depth ); + } + if ( depth > kTableJsonMaxDepth ) { return false; } + if ( entry->label != 0 ) + { + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + out.line( depth ); + out.put( '}' ); + return true; + } + entry->label = ++graph->next_label; + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + bool any = true; + int64_t before = out.offset; + if ( !TableJsonWriteFields( out, node, f->table, depth, any ) ) { return false; } + // a definition carries at least one field, because a label alone is a + // reference: a shared node with nothing to write has no definition this + // form can spell, and the writer refuses it as it refuses any value it + // cannot spell (§16.3) + if ( out.offset == before ) { return false; } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- the two entry points a pointered table's wrappers name ---- + +// The text into the builder's root. Every node the text names is allocated in +// the builder's arena through the field's own Emplace; the label map is the +// walk's, released before this returns. The root itself takes no label — nothing +// may name it (§16.7) — so an `&node` at the root is refused like any other key +// of the prefix. +inline bool TableJsonReadGraph( TableWorker & worker, void * root, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + if ( worker.arena == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } + TableJsonGraphIn graph; + graph.worker = &worker; + TableJsonGraphMapInit( graph.labels, worker.arena->allocator ); + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = &graph; + info->reset( root ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, root, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + TableJsonGraphMapShutdown( graph.labels ); + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +// The text of a region's const root: measured when the buffer is NULL, written +// when it is not, over one code path. Two passes over one walk — the first +// counts how many slots name each node and refuses a cycle, the second writes +// — so a node's first occurrence knows whether it will be named again. The +// ROOT's entry is open for the whole first pass, so a reference back at it is +// the cycle it is (§3.1), and it takes no label. +inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * info, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + if ( root == NULL ) { return -1; } + TableJsonGraphOut graph; + TableJsonGraphMapInit( graph.nodes, allocator ); + graph.counting = true; + graph.next_label = 0; + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph.nodes, (uint64_t) (uintptr_t) root, taken ); + if ( entry == NULL ) { TableJsonGraphMapShutdown( graph.nodes ); return -1; } + entry->open = 1; + TableJsonOut count; + count.buffer = NULL; + count.capacity = 0; + count.offset = 0; + count.overflow = false; + count.graph = &graph; + bool ok = TableJsonWriteValue( count, root, info, 0 ); + graph.counting = false; + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = &graph; + if ( ok ) { ok = TableJsonWriteValue( out, root, info, 0 ); } + TableJsonGraphMapShutdown( graph.nodes ); + if ( !ok ) { return -1; } + out.put( '\n' ); // the canonical text ends with exactly one newline (§16.1) + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json graph walk: end ---- + +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + +// ---- json map walk: begin ---- + +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} + +// the entry's two rows: fields[0] IS the key and fields[1] IS the value, which +// is what makes a user's own table of pairs the same bytes (§2.8) +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } + +inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } +inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } + +// AN INTEGER KEY IS THE INTEGER'S DECIMAL SPELLING, QUOTED, because a JSON +// object's keys are strings. Written digit by digit so no locale can move it. +inline void TableJsonWriteMapIntegerKey( TableJsonOut & out, const void * storage, const TableFieldInfo * key ) +{ + uint64_t magnitude = 0; + bool negative = false; + if ( TableJsonMapKeySigned( key ) ) + { + int64_t value = 0; + switch ( key->kind ) + { + case 2: value = (int64_t) *(const int8_t *) storage; break; + case 3: value = (int64_t) *(const int16_t *) storage; break; + case 4: value = (int64_t) *(const int32_t *) storage; break; + default: value = *(const int64_t *) storage; break; + } + negative = value < 0; + magnitude = negative ? ( ~(uint64_t) value ) + 1 : (uint64_t) value; + } + else + { + switch ( key->kind ) + { + case 6: magnitude = (uint64_t) *(const uint8_t *) storage; break; + case 7: magnitude = (uint64_t) *(const uint16_t *) storage; break; + case 8: magnitude = (uint64_t) *(const uint32_t *) storage; break; + default: magnitude = *(const uint64_t *) storage; break; + } + } + char digits[24]; + int32_t at = (int32_t) sizeof( digits ); + do { digits[--at] = (char) ( '0' + ( magnitude % 10 ) ); magnitude /= 10; } while ( magnitude != 0 ); + if ( negative ) { digits[--at] = '-'; } + TableJsonWriteString( out, digits + at, (int32_t) sizeof( digits ) - at ); +} + +inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const TableFieldInfo * key ) +{ + const uint8_t * storage = (const uint8_t *) entry + key->offset; + if ( TableJsonMapKeyIsString( key ) ) + { + // A STRING KEY IS THE STRING (§2.8): every JSON key of a map object is + // a KEY OF THE MAP and none is a field key, so the `&` prefix §16.7 + // reserves for field keys is ordinary data here. + TableJsonWriteString( out, (const char *) storage, *(const int32_t *) ( (const uint8_t *) entry + key->count_offset ) ); + return; + } + TableJsonWriteMapIntegerKey( out, (const void *) storage, key ); +} + +// ToJson WRITES ENTRIES IN ASCENDING KEY ORDER, so unpack then pack is +// byte-stable and a diff of two texts is a diff of two maps (§2.8, §17.2). +// A region holds them in that order already, so this is the array in place. +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "{}", 2 ); return true; } + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); + out.put( '{' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); + TableJsonWriteMapKey( out, entry, key ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// AN INTEGER KEY IS READ BY §16.2's INTEGER RULE AND BY NOTHING ELSE, so +// "2.0" and "1e3" are the integers 2 and 1000 and "-0" is zero. The token is +// walked as a JSON number over its own bytes; a token that rule calls +// malformed makes the KEY malformed, and a genuinely fractional value, or one +// outside the key kind's range, is kind_mismatch for that entry. +inline bool TableJsonMapKeyValue( const char * token, int32_t length, const TableFieldInfo * key, + int64_t & value, bool & fits ) +{ + fits = false; + TableReport scratch; + TableJsonIn probe = { token, (int64_t) length, 0, &scratch, false, NULL }; + bool integral = false; + if ( !TableJsonWalkNumber( probe, &integral ) ) { return false; } + if ( probe.pos != (int64_t) length ) { return false; } // trailing bytes: not a number + if ( !integral ) + { + const double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) { return true; } // a value no key kind holds + const double whole = d < 0 ? -d : d; + if ( whole != (double) (int64_t) whole ) { return true; } // genuinely fractional + } + bool saturated = false; + const bool is_signed = TableJsonMapKeySigned( key ); + value = integral ? TableJsonTokenInteger( token, length, is_signed, &saturated ) + : (int64_t) TableJsonTokenDouble( token, length, false ); + if ( saturated ) { return true; } // outside every width: kind_mismatch, never clamped + switch ( key->kind ) + { + case 2: fits = value >= -128 && value <= 127; break; + case 3: fits = value >= -32768 && value <= 32767; break; + case 4: fits = value >= -2147483647 - 1 && value <= 2147483647; break; + case 5: fits = true; break; + case 6: fits = value >= 0 && value <= 255; break; + case 7: fits = value >= 0 && value <= 65535; break; + case 8: fits = value >= 0 && (uint64_t) value <= 4294967295ull; break; + default: fits = integral; break; // uint64: the token's own magnitude + } + return true; +} + +// FromJson READS KEYS IN WHATEVER ORDER THE TEXT GIVES THEM. A repeated key is +// last-wins and counted duplicate, the object rule (§16.2) applied inside the +// map. An empty object is an empty map, and null is kind_mismatch. +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const char shape = TableJsonShape( value ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char token[kTableJsonMaxKey]; + int32_t token_length = 0; + if ( !TableJsonScanString( in, token, kTableJsonMaxKey - 1, &token_length ) ) { return false; } + token[token_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t key_value = 0; + bool place = true; + if ( !TableJsonMapKeyIsString( key ) ) + { + bool fits = false; + if ( !TableJsonMapKeyValue( token, token_length, key, key_value, fits ) ) + { + // A MALFORMED KEY STOPS THE READ where §16.1's rule stops it, + // with the instance holding what was placed before the stop. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( !fits ) { in.report->kind_mismatch++; place = false; } + } + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; + if ( place && entry == NULL ) + { + // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the + // wire's rule, because a clamped key is a merged entry (§2.8). + in.report->clamped++; + } + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) + { + in.report->duplicate++; // last-wins, the object rule inside the map + } + const char got = TableJsonValueShape( in ); + if ( entry == NULL ) + { + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( value->kind == 17 && !value->is_array ) + { + // A POINTER VALUE IS SHARED EXACTLY AS A POINTER FIELD IS (§2.8): + // null is a null slot, an object is the pointee in place or an + // &node reference to one (§16.7), anything else is the wrong shape — + // the same three the field-key loop gives a pointer field, because + // an entry's value IS a field line. + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) entry + value->offset, value->elem_size, 0 ); + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) entry + value->offset, value, depth + 1 ) ) + { + return false; + } + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, entry, value, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json map walk: end ---- + +// ---- json list walk: begin ---- + +// an unbounded array is the out-of-line array that is not a map (§8.1) +inline bool TableJsonIsList( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && !TableJsonIsMap( f ); +} + +// ToJson WRITES THE ELEMENTS IN INDEX ORDER, which is the only order there is, +// so unpack then pack is byte-stable without a rule of its own (§2.9, §17.2). +// A region holds the array in place, so this steps it at the descriptor's pitch. +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + const uint8_t * elements = TableJsonExtentElements( slot ); + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const uint8_t * element = elements + (int64_t) i * f->elem_size; + if ( f->kind == 17 ) + { + // a []*T's elements take the pointer row (§16.7): the pointee's + // object in place, null, or `&node` for a shared one + if ( !TableJsonWritePointer( out, element, f, depth + 1 ) ) { return false; } + } + else if ( !TableJsonWriteScalar( out, element, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; +} + +// FromJson READS EVERY ELEMENT THE TEXT CARRIES, appending each through the +// descriptor's place resolver: `[]` is an empty list, and null is +// kind_mismatch, the array row's own rule (§16.2). LAST WINS holds for a +// repeated key: the list goes back to EMPTY before this occurrence's elements +// land, the builder's storage being reclaimed at reset (§2.9). +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) slot, 8, 0 ); + TableJsonSetRaw( (uint8_t *) slot + 8, 4, 0 ); + const char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + void * element = f->place( *graph->worker, slot, NULL, 0, 0 ); + if ( element == NULL ) + { + // NOT ADDED: the arena could not carve another segment, or the + // count met the int32 cap. The text cannot be placed whole, and + // the read stops where §16.1's rule stops it. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( f->kind == 17 ) + { + // an element of a []*T (§2.9): null is a null slot, an object is the + // pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) element, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, element, f, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + // the wrong shape for the element kind: the slot keeps its + // defaults and the event counts, the array row's rule (§16.2) + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, element, f, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json list walk: end ---- + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +bool UnitFromJson( Unit & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, UnitTableType(), text, bytes, report ); +} + +int64_t UnitToJsonMeasure( const Unit & value ) +{ + return TableJsonWrite( &value, UnitTableType(), NULL, 0 ); +} + +int64_t UnitToJson( const Unit & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, UnitTableType(), buffer, capacity ); +} + +bool BoundedFromJson( Bounded & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, BoundedTableType(), text, bytes, report ); +} + +int64_t BoundedToJsonMeasure( const Bounded & value ) +{ + return TableJsonWrite( &value, BoundedTableType(), NULL, 0 ); +} + +int64_t BoundedToJson( const Bounded & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, BoundedTableType(), buffer, capacity ); +} + +bool UnboundedFromJson( UnboundedBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Unbounded * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, UnboundedTableType(), text, bytes, report ); +} + +int64_t UnboundedToJsonMeasure( const Unbounded * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, UnboundedTableType(), NULL, 0, allocator ); +} + +int64_t UnboundedToJson( const Unbounded * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, UnboundedTableType(), buffer, capacity, allocator ); +} + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/MigrateTable.h b/testdata/golden/tables/lists/MigrateTable.h new file mode 100644 index 000000000..9bc92d3cd --- /dev/null +++ b/testdata/golden/tables/lists/MigrateTable.h @@ -0,0 +1,5609 @@ +// Code generated by the schema compiler from Migrate.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — protocol id 0xa5fbe602c119cdd9 (packets only: tables version by field id, not by protocol id) +// The TABLE wire (evolution-tolerant, docs/SPEC-TABLES.md): no serialize +// dependency — includable from any TU. + +#pragma once + +#include +#include // the prefill's scalar-array fills +#include // offsetof, for the reflection descriptors + +// ---- the hooks (docs/USAGE.md, "the C++ table runtime's hooks") ---- +// +// schema_assert — the runtime's own assert, and the refusal a debugger reads. +// NDEBUG removes it, exactly as it removes assert. A caller who already routes +// serialize's asserts writes `#define schema_assert serialize_assert` before +// including this header and both halves land in one handler. +#ifndef schema_assert +#include +#define schema_assert assert +#endif // #ifndef schema_assert + +// schema_fatal — what stands after the assert on a path that cannot continue. +// NDEBUG does not remove it. Supply it and is never included. +#ifndef schema_fatal +#include // abort +#define schema_fatal abort +#endif // #ifndef schema_fatal + +// schema_allocate / schema_release — what "no allocator handed in" means for +// this program. schema_allocate hands back ZEROED bytes and NULL on failure: +// an arena segment is copied whole, padding included, so anything left +// uninitialized here would reach a packed region. Supply both and +// is never included; hand a TableAllocator to a builder to route one +// structure's allocations somewhere else again. +#ifndef schema_allocate +#include // calloc, free +#define schema_allocate( bytes ) calloc( (size_t) 1, (size_t) ( bytes ) ) +#define schema_release( pointer ) free( pointer ) +#endif // #ifndef schema_allocate +#include // a node's lifetime starts in arena storage (placement new) +#include // one atomic per slab: the arena is lock-free by ownership + +#include "Migrate.h" + +#ifndef LISTDEMO_SCHEMA_TABLE_PRIMITIVES +#define LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +// THE CODEC DOES NOT DEPEND ON THE COMPILER'S INLINING BUDGET. A table of a +// realistic field count emits one large body per type, and the cursor a body +// writes through lives in the caller's `TableWriter`: across a call boundary +// that cursor round-trips through memory, and a `uint8_t *` store may alias the +// writer itself, so every put reloads it. When a budget runs out mid-body the +// codec silently degrades to that shape. Forcing the primitives and the +// fixed-class bodies inline is what keeps the cursor in registers and lets +// adjacent constant framing bytes merge into one store. +#if defined( _MSC_VER ) +#define LISTDEMO_TABLE_INLINE __forceinline +#elif defined( __GNUC__ ) || defined( __clang__ ) +#define LISTDEMO_TABLE_INLINE inline __attribute__(( always_inline )) +#else +#define LISTDEMO_TABLE_INLINE inline +#endif + +namespace listdemo { + +// WHY A READ WAS REFUSED, by name (docs/SPEC-TABLES.md §3.3, §11). A REFUSAL +// is not one of §4's events: nothing is decoded, no counter moves and no +// damage is reported, so five zero counters and a false flag are what a clean +// read prints too and only the verdict tells them apart. The reason says which +// refusal it was. +// +// This is the MESSAGE PATH's vocabulary and not the cooked form's (§7.4): a +// caller meeting one of these has been refused a MESSAGE on a connection, +// which is a different recovery with a different owner than a file a header +// match turned down. +enum TableMessageReason +{ + newer_form, // a FORM BYTE this reader does not carry (§3) + no_vocabulary, // no table for this connection: the message arrived before the announcement, or after a refused one + second_announcement, // a second announcement on a connection: it sets nothing, amends nothing, and the connection closes + vocabulary_too_large, // an announcement above the receiver's declared bound, refused before an entry is touched + message_form_as_file // a form 2 wire where a FILE was expected: its table is somewhere else +}; + +// The table-wire read report — the permissive contract's ledger. Silence +// (all zero) means the data matched this reader's schema exactly. +struct TableReport +{ + int32_t unknown = 0; // unknown field ids skipped (newer data) + int32_t kind_mismatch = 0; // known id, changed type — skipped, never misdecoded + int32_t clamped = 0; // out-of-range values clamped to declared bounds + // a key the TEXT form saw twice: last wins, and the repeat is counted + // (docs/SPEC-TABLES.md §16.2). The wire never raises it — a body carrying an + // id twice is legal input whose last occurrence wins, silently (§3). + int32_t duplicate = 0; + bool malformed = false; // framing damage; decode stopped, partial result kept + // THE REFUSAL VERDICT, which is not one of §4's events and moves no counter + // (docs/SPEC-TABLES.md §3): a FORM BYTE this reader does not carry. Five + // zero counters and a false flag are what a clean read prints too, so the + // verdict is what tells the two apart. + bool refused = false; + // WHICH refusal, and it is read only when refused is set: a read that + // was not refused has no reason, and this member is the one the caller + // must not look at then (docs/SPEC-TABLES.md §3.3). + TableMessageReason reason = newer_form; +}; + + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; +// ---- reflection (tables only, docs/SPEC-TABLES.md) ---- +// +// Static field descriptors for every type in the table closure: name, wire +// id/kind, storage offset, bounds, ranges, enum names and branch guards — +// enough to walk, print, diff, edit or bind any table value at runtime with +// no RTTI and no schema files. TableType() returns X's descriptor. + +struct TableTypeInfo; + +// One arm of a union field: where its payload sits inside the union's storage +// and what its payload looks like. The arm's NAME and its table-wire id come +// from the field's enum_name/variant_id functions at the same tag, so nothing +// is spelled twice (docs/SPEC-TABLES.md §8). +struct TableFieldInfo; + +struct TableUnionArmInfo +{ + uint32_t offset; // offsetof the arm's payload within the union storage + const TableTypeInfo * table; // the arm payload's descriptor, or NULL + // AN ARM IS A FIELD LINE (docs/SPEC-TABLES.md §2.6): an arm that names no + // declared type or table carries the FIELD descriptor a field of that + // type would carry instead — offsets taken within the union storage — so + // a generic walk meets an arm's kind, width, bounds and companions where + // it meets a field's. Exactly one of the two is non-NULL on a set arm. + const TableFieldInfo * field; + uint32_t size; // the arm's whole storage, which selection zero-establishes +}; + +// A union field's shape: the tag, and the arms indexed by it. Arms run +// [0, enum_max]; index 0 is the EMPTY arm and carries no payload. +struct TableUnionInfo +{ + uint32_t tag_offset; // offsetof the tag within the union storage + uint32_t tag_size; // sizeof the tag + const TableUnionArmInfo * arms; +}; + +// The exact raw range of a wide-kind field (docs/SPEC-TABLES.md §8.2): two 128-bit +// values as 64-bit lanes, low lane first, two's complement for the signed kinds. +struct TableWideRange +{ + uint64_t lo[2]; + uint64_t hi[2]; +}; + +// the arena's allocation front, defined with the variable-length runtime +// below; a descriptor names it only through a pointer parameter. +struct TableWorker; + +struct TableFieldInfo +{ + const char * name; // schema field name, e.g. "health" + const char * json; // the TEXT form's key: the json = "key" attribute, else name (§16.3) + const char * type_name; // schema type name, e.g. "float32", "Grade" + uint64_t id; // table-wire field id: fnv1a64 of the name, of the was alias after a rename (§5) + uint8_t kind; // table-wire kind; for arrays/strings/bytes, the ELEMENT kind + bool is_array; // fixed or counted array (bytes included) + bool is_pointer; // a *T pointer field: storage is an 8-byte TableRef; the target is a table + // THE TWO THE TEXT FORM NEEDS (docs/SPEC-TABLES.md §16.7), and they + // are here for the same reason is_pointer is: the walk is ONE walk + // over descriptors and cannot spell a target's own At or + // Emplace. `resolve` reads a slot in a REGION and answers the + // node it names, or NULL; `emplace` allocates one in a BUILDER's + // arena and points the slot at it. NULL on every field that is not + // a pointer, and emitted only in a unit that declares one. + const void * (*resolve)( const void * slot ); + void * (*emplace)( TableWorker & worker, void * slot ); + bool counted; // a _count/_length int32 companion exists (counted arrays, strings, bytes) + bool optional; // a ?T field: a _present bool companion decides whether it rides + int32_t array_bound; // array capacity / string max length; 0 for plain scalars + uint32_t offset; // offsetof the storage member + uint32_t elem_size; // sizeof the member (element size for arrays) + uint32_t count_offset; // offsetof the _count/_length companion, or 0xffffffff + uint32_t present_offset; // offsetof the _present companion, or 0xffffffff + const TableTypeInfo * table; // nested table's descriptor, or NULL + bool has_range; // a declared [min, max] (int or float) + double range_min; // NOTE: int64 ranges beyond 2^53 lose precision here + double range_max; + // the WIDE kinds (18-29, docs/SPEC-TABLES.md §3, §8.2): frac_bits is a fixed + // field's F — its storage holds units × 2^F — and wide is the declared + // range on that RAW scale, exact, as two 128-bit two's-complement values + // in 64-bit lanes (low lane first). NULL where the declaration bounds + // nothing (a bare uint128) and for every other kind; frac_bits is 0 for + // every kind that is not fixed-point. range_min/range_max still carry + // the declared bounds as doubles — whole units for a fixed field — for + // a walker that only shows them. + uint8_t frac_bits; + const TableWideRange * wide; + int64_t enum_max; // enums: highest valid value (None = 0 always valid); + // unions: the arm count (tag range [0, enum_max]); + // flags: the highest declared BIT INDEX; else -1 + // the vocabulary's names, indexed the same way enum_max bounds: an enum's + // value -> name, a union's tag -> arm name, a FLAGS field's bit index -> + // variant name. NULL for every other kind. + const char * (*enum_name)( uint64_t value ); + // the TABLE-WIRE id of one variant (docs/SPEC-TABLES.md §5): for an enum, the + // hash of the variant's name; for a union, the hash of the arm's name. + // 0 is the reserved id — an enum's None, a union's empty. NULL for every + // other kind — a FLAGS field's variants have no per-variant wire id (§4), + // so a NULL here beside a non-NULL enum_name is what says "flags". + // Walk [0, enum_max] to enumerate a vocabulary and its ids. + uint64_t (*variant_id)( uint64_t value ); + // an ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): the array has one slot per + // variant of key_type_name, indexed by the variant's value, and its slots + // ride under variant ids rather than positions. key_name and key_id are + // the key's vocabulary — walk [0, array_bound) to print slots by name. + // NULL on every other field. + const char * key_type_name; + const char * (*key_name)( uint64_t value ); + uint64_t (*key_id)( uint64_t value ); + // union fields: the tag and its arms, behind a function so the whole + // descriptor stays CONSTANT-INITIALISED (a captureless lambda converts to + // a function pointer at compile time; the arms themselves are a static + // inside it). NULL for every other kind. + const TableUnionInfo * (*arms)(); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded +}; + +struct TableTypeInfo +{ + const char * name; // schema type name + uint32_t size; // sizeof the storage struct + int32_t num_fields; + const TableFieldInfo * fields; + // put one instance back at its declared defaults, in place. A generic + // walker that fills a value has to be able to establish the defaults an + // absent field takes, and it holds no type to spell — this is the one + // thing the descriptors could not express without it. Placement-new + // value-init, exactly what the wire's read path does, and no temporary. + void (*reset)( void * storage ); + // the DERIVED mode (docs/SPEC-TABLES.md): false = fixed-size, a plain + // relocatable struct; true = variable-length, built through a Builder + // and read through a region root. Nobody declares it; the compiler + // works it out. + bool variable; +}; + +struct TableWriter +{ + uint8_t * buffer; + int64_t capacity; + int64_t offset = 0; + bool overflow = false; + + // the parameters do not repeat the member names: a parameter that hides a + // member is a warning the estate's compilers disagree about (gcc's + // -Wshadow and cl's C4458 refuse it, clang's -Wshadow does not), and this + // is a header a consumer compiles under its OWN flags + TableWriter( uint8_t * to_buffer, int64_t to_capacity ) : buffer( to_buffer ), capacity( to_capacity ) {} + + LISTDEMO_TABLE_INLINE void raw( const void * data, int64_t bytes ) + { + if ( offset + bytes > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) bytes ); + offset += bytes; + } + LISTDEMO_TABLE_INLINE void put8( uint8_t v ) { raw( &v, 1 ); } + LISTDEMO_TABLE_INLINE void put16( uint16_t v ) { uint8_t b[2] = { uint8_t( v ), uint8_t( v >> 8 ) }; raw( b, 2 ); } + LISTDEMO_TABLE_INLINE void put32( uint32_t v ) { uint8_t b[4] = { uint8_t( v ), uint8_t( v >> 8 ), uint8_t( v >> 16 ), uint8_t( v >> 24 ) }; raw( b, 4 ); } + LISTDEMO_TABLE_INLINE void put64( uint64_t v ) { put32( uint32_t( v ) ); put32( uint32_t( v >> 32 ) ); } + // a 128-bit value as two lanes, the low half first (docs/SPEC-TABLES.md §3) + LISTDEMO_TABLE_INLINE void put128( uint64_t lo, uint64_t hi ) { put64( lo ); put64( hi ); } + // EVERY LENGTH, COUNT, INDEX AND ID REFERENCE IS ONE CANONICAL UNSIGNED + // LEB128 (docs/SPEC-TABLES.md §3): seven value bits a byte, the lowest + // group first, the high bit set on every byte but the last. One value has + // one spelling, so two conforming writers agree byte for byte. + LISTDEMO_TABLE_INLINE void putleb( uint64_t v ) + { + while ( v >= 0x80 ) { put8( uint8_t( v ) | 0x80 ); v >>= 7; } + put8( uint8_t( v ) ); + } +}; + +// TableLebBytes is one value's spelling length, which a MEASURE needs before +// the bytes exist — the length of a body has to be known before it is written, +// because a length whose own width moves cannot be patched in place. +inline int64_t TableLebBytes( uint64_t v ) +{ + int64_t n = 1; + while ( v >= 0x80 ) { v >>= 7; n++; } + return n; +} + +// THE ID TABLE, WRITER SIDE (docs/SPEC-TABLES.md §3). It holds every id the +// body used, once each, in FIRST-USE order over the whole wire, and the body +// names them by position: reference k is the kth entry, counted from 1, and +// reference 0 names NO ID. +// +// Its capacity is a COMPILE-TIME fact of the unit — the distinct names its +// table closure can spell — so a save allocates nothing: the table is a local +// of Measure and of Save. The bucket chain makes ref constant time and makes +// truncate constant time too, which is what an ELIDED field needs: a field +// that turns out not to ride costs nothing in the id table either, so the walk +// interns its id, builds the payload that decides, and undoes the entry when +// nothing rides. +struct TableIds +{ + static const int32_t kCapacity = 60; + static const int32_t kBuckets = 128; + + uint64_t ids[ kCapacity ]; + int32_t chain[ kCapacity ]; + int32_t head[ kBuckets ]; + int32_t count; + bool overflow; + // THE MESSAGE FORM'S SLOTS (docs/SPEC-TABLES.md §3.3). A form 2 wire + // names ids through the CONNECTION's table, which is the unit's whole + // vocabulary in a compiler-settled order — so every reference is known at + // compile time and rides at the header as a literal beside the id. This + // flag is what selects it: false interns the id in first-use order and + // writes a trailer, true answers the slot and writes none, and the walk + // that decides is one walk. + bool vocabulary; + + TableIds() : count( 0 ), overflow( false ), vocabulary( false ) + { + for ( int32_t i = 0; i < kBuckets; i++ ) { head[i] = -1; } + } + + static LISTDEMO_TABLE_INLINE uint32_t bucket_of( uint64_t id ) + { + return uint32_t( ( id * 0x9E3779B97F4A7C15ull ) >> 57 ) & uint32_t( kBuckets - 1 ); + } + + // the reference an id takes: its message-form SLOT under the connection's + // table, or the file's own first-use entry + LISTDEMO_TABLE_INLINE uint64_t ref( uint64_t id, uint64_t slot ) + { + if ( vocabulary ) { return slot; } + return intern( id ); + } + + // the FILE form's half, appending the id on first use + uint64_t intern( uint64_t id ) + { + const uint32_t b = bucket_of( id ); + for ( int32_t i = head[b]; i >= 0; i = chain[i] ) + { + if ( ids[i] == id ) { return uint64_t( i ) + 1; } + } + if ( count >= kCapacity ) { overflow = true; return 1; } + ids[count] = id; chain[count] = head[b]; head[b] = count; count++; + return uint64_t( count ); + } + + // undo every entry appended since mark. An entry removed is the most + // recent one in its bucket, so it sits at that bucket's head. + void truncate( int32_t mark ) + { + // a SLOT costs no entry, so an elided field has nothing to undo + if ( vocabulary ) { return; } + while ( count > mark ) + { + count--; + head[ bucket_of( ids[count] ) ] = chain[count]; + } + } +}; + +// TableIdsBytes is the trailer's own size: the entries, each a fixed +// little-endian u64, and the ENTRY COUNT, the one fixed-width number on the +// wire (docs/SPEC-TABLES.md §3). +inline int64_t TableIdsBytes( const TableIds & ids ) { return int64_t( ids.count ) * 8 + 8; } + +// TableIdsWrite puts the trailer where the walk ended: a writer never patches, +// because first-use order is known only when the walk ends. +inline void TableIdsWrite( TableWriter & w, const TableIds & ids ) +{ + for ( int32_t i = 0; i < ids.count; i++ ) { w.put64( ids.ids[i] ); } + w.put64( uint64_t( ids.count ) ); +} + +// THE ID TABLE, READER SIDE (docs/SPEC-TABLES.md §3). A reader locates it from +// the END of the wire and resolves it ONCE, at open: the entries are eight +// bytes each and a body names them by position, so every field dispatches +// through an index rather than through a search over hashes. +struct TableIdTable +{ + const uint8_t * entries = NULL; + int64_t count = 0; + + // the id a reference names. ref is 1-based and bounds-checked by the + // caller: a reference ABOVE the entry count is framing damage on the body + // that carries it, and 0 names no id at all. + uint64_t at( uint64_t ref ) const + { + const uint8_t * e = entries + ( ref - 1 ) * 8; + uint64_t lo = uint64_t( e[0] ) | uint64_t( e[1] ) << 8 | uint64_t( e[2] ) << 16 | uint64_t( e[3] ) << 24; + uint64_t hi = uint64_t( e[4] ) | uint64_t( e[5] ) << 8 | uint64_t( e[6] ) << 16 | uint64_t( e[7] ) << 24; + return lo | ( hi << 32 ); + } +}; + +struct TableReader +{ + const uint8_t * buffer; + int64_t size; + int64_t offset = 0; + TableReport * report; + const TableIdTable * ids = NULL; + // ONLY THE ROOT BODY CARRIES THE NODE TABLE (docs/SPEC-TABLES.md §3.1), so + // a body has to know which it is: the reserved id inside a NESTED body is + // malformed, because a second numbering cannot exist. Every reader made + // for a payload is nested; the two the wire surfaces make for a root say so. + bool nested = true; + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report ) + : buffer( from_buffer ), size( from_size ), report( to_report ) {} + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report, const TableIdTable * to_ids ) + : buffer( from_buffer ), size( from_size ), report( to_report ), ids( to_ids ) {} + + LISTDEMO_TABLE_INLINE bool has( int64_t bytes ) const { return offset + bytes <= size; } + // A LENGTH IS A 64-BIT NUMBER AND A BUFFER IS NOT (docs/SPEC-TABLES.md + // §3): every length, count and index on this wire has sixty-four bits of + // capability, so one past what remains must be compared UNSIGNED. Casting + // it to int64 first turns 0xFFFFFFFFFFFFFFFF into -1, and a negative + // length looks like room. + LISTDEMO_TABLE_INLINE bool room( uint64_t bytes ) const { return bytes <= (uint64_t) ( size - offset ); } + LISTDEMO_TABLE_INLINE uint8_t get8() { return buffer[offset++]; } + LISTDEMO_TABLE_INLINE uint16_t get16() { uint16_t v = uint16_t( buffer[offset] ) | uint16_t( buffer[offset+1] ) << 8; offset += 2; return v; } + LISTDEMO_TABLE_INLINE uint32_t get32() { uint32_t v = uint32_t( buffer[offset] ) | uint32_t( buffer[offset+1] ) << 8 | uint32_t( buffer[offset+2] ) << 16 | uint32_t( buffer[offset+3] ) << 24; offset += 4; return v; } + LISTDEMO_TABLE_INLINE uint64_t get64() { uint64_t lo = get32(); uint64_t hi = get32(); return lo | ( hi << 32 ); } + LISTDEMO_TABLE_INLINE void get128( uint64_t & lo, uint64_t & hi ) { lo = get64(); hi = get64(); } + + // ONE CANONICAL UNSIGNED LEB128 (docs/SPEC-TABLES.md §3), and a + // non-minimal spelling is MALFORMED: 0x80 0x00 and 0x00 both spell zero, + // and only the second is legal input. An encoding past ten bytes, or a + // tenth byte with a bit above the 64th value bit, is malformed on the same + // rule. false = framing damage on the body carrying it. + bool getleb( uint64_t & value ) + { + // A NUMBER THIS READER REFUSES LEAVES THE CURSOR WHERE IT WAS. The + // caller's next question is often "did this body end exactly at its + // L", and a rejected number that had moved the cursor would answer + // that question with the damage already stepped over. + const int64_t at = offset; + value = 0; + uint32_t shift = 0; + for ( int32_t i = 0; i < 10; i++ ) + { + if ( !has( 1 ) ) { offset = at; return false; } + const uint8_t b = get8(); + if ( i == 9 && b > 1 ) { offset = at; return false; } + value |= uint64_t( b & 0x7F ) << shift; + if ( ( b & 0x80 ) == 0 ) + { + if ( i > 0 && b == 0 ) { offset = at; return false; } // a redundant continuation + return true; + } + shift += 7; + } + offset = at; + return false; + } + + // resolve one id reference against the file's table. false = a reference + // ABOVE the entry count, or a 0 where an id is required, both of which + // are framing damage on the body that carries it. + bool getid( uint64_t & id ) + { + uint64_t ref = 0; + if ( !getleb( ref ) ) { return false; } + if ( ref == 0 || ids == NULL || ref > (uint64_t) ids->count ) { return false; } + id = ids->at( ref ); + return true; + } + + // skip one payload by kind; false = framing damage. FOUR RULES COVER THE + // SET (docs/SPEC-TABLES.md §3), and a kind outside it is not skippable — + // which is why the set is closed and why kind 31 exists. + bool skip( uint8_t kind ) + { + switch ( kind ) + { + // the fixed-width kinds, each by its width: 18-29 are the 128-bit integers and + // the fixed-point family at every storage width (docs/SPEC-TABLES.md §3) + case 1: case 2: case 6: case 20: case 25: return has( 1 ) ? ( offset += 1, true ) : false; + case 3: case 7: case 21: case 26: return has( 2 ) ? ( offset += 2, true ) : false; + case 4: case 8: case 10: case 22: case 27: return has( 4 ) ? ( offset += 4, true ) : false; + case 5: case 9: case 11: case 23: case 28: return has( 8 ) ? ( offset += 8, true ) : false; + case 18: case 19: case 24: case 29: return has( 16 ) ? ( offset += 16, true ) : false; + case 17: case 30: // a NODE INDEX (§3.1) and an ENUM's variant reference: one LEB128 and stop + { + uint64_t ignored = 0; + return getleb( ignored ); + } + case 12: case 13: case 14: case 16: case 31: case 32: // 31 is the ESCAPE, 32 the payload-free kind + { + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + case 15: // union: the arm id reference, then its kind, its L and its payload (reference 0 = empty) + { + uint64_t arm = 0; + if ( !getleb( arm ) ) return false; + if ( arm == 0 ) return true; + if ( !has( 1 ) ) return false; + offset += 1; // the arm's kind byte + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + } + return false; + } +}; + +// The RESERVED node-table id, the one id the language holds back +// (docs/SPEC-TABLES.md §3.1, §5). It rides in every unit, pointered or not, +// because every body has to know that a NESTED body claiming one is damaged. +static const uint64_t kTableNodeTableFieldId = 0xFFFFFFFFFFFFFFFFull; + +// TableWireForm is the FORM BYTE, and it is the whole header +// (docs/SPEC-TABLES.md §3). A reader that meets a byte it does not know +// refuses the wire by name and never reports damage. +const uint8_t kTableWireForm = 1; + +// TableOpen reads the form byte and the trailer, in that order, and hands back +// the ROOT BODY. It answers one of three verdicts, because five zero counters +// and a false flag are what a clean read prints too: +// +// TableOpenOk the form is known and the table read whole +// TableOpenRefused a FORM BYTE this reader does not carry: nothing is +// decoded, nothing is counted, and no damage is reported +// TableOpenDamaged a table that cannot be read whole — fewer than eight +// bytes, a count whose entries run past the front of the +// file, a count that leaves no room for the form byte, or +// ONE ID IN TWO ENTRIES. The whole wire is malformed, +// nothing is decoded, and one event is counted. +// TableOpenBodyStopped the form and the table were good and the ROOT BODY +// could not be walked to its own terminator. What it +// decoded before that is kept, as everywhere on this wire. +enum TableOpenVerdict { TableOpenOk, TableOpenRefused, TableOpenDamaged, TableOpenBodyStopped }; + +inline TableOpenVerdict TableOpen( const uint8_t * buffer, int64_t bytes, TableIdTable & table, int64_t & body_bytes ) +{ + if ( bytes < 1 ) { return TableOpenDamaged; } + if ( buffer[0] != kTableWireForm ) { return TableOpenRefused; } + if ( bytes < 9 ) { return TableOpenDamaged; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + uint64_t count = lo | ( hi << 32 ); + if ( count > (uint64_t) ( bytes / 8 ) ) { return TableOpenDamaged; } + const int64_t span = (int64_t) count * 8 + 8; + if ( span + 1 > bytes ) { return TableOpenDamaged; } + table.entries = buffer + bytes - span; + table.count = (int64_t) count; + // THE ENTRIES ARE DISTINCT: a table that carries one id twice is malformed + // for the whole wire, because no wire this schema writes carries a repeat + // and it would leave one more shape of table for a hostile writer to aim + // at (docs/SPEC-TABLES.md §3). + for ( int64_t i = 1; i < table.count; i++ ) + { + const uint64_t id = table.at( uint64_t( i ) + 1 ); + for ( int64_t j = 0; j < i; j++ ) + { + if ( table.at( uint64_t( j ) + 1 ) == id ) { return TableOpenDamaged; } + } + } + body_bytes = bytes - span - 1; + return TableOpenOk; +} + +// TableBodyExtent walks a body's framing to the zero reference that ends it, +// so a reader can tell a body that ENDED EARLY — leaving bytes no field claims +// — from one that is merely damaged. ANY BYTE BETWEEN THE ROOT'S TERMINATOR +// AND THE TABLE'S FIRST ENTRY IS MALFORMED, because no field claims it and the +// two ends of the file have met (docs/SPEC-TABLES.md §3). +inline bool TableBodyEndsEarly( const uint8_t * body, int64_t bytes, const TableIdTable & table ) +{ + TableReport ignored; + TableReader r( body, bytes, &ignored, &table ); + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { return false; } + if ( ref == 0 ) { return r.offset != bytes; } + if ( ref > (uint64_t) table.count ) { return false; } + if ( !r.has( 1 ) ) { return false; } + if ( !r.skip( r.get8() ) ) { return false; } + } +} + +// THE MESSAGE FORM (docs/SPEC-TABLES.md §3.3): a FILE carries its own id +// table and a MESSAGE STREAM announces one and then carries none. +// +// A form 2 wire is TWO PARTS, the form byte and the root body: the body ends +// at its own zero reference as it does in a file, there is no trailer, and the +// message's last byte is the body's terminator. Its references resolve against +// the CONNECTION's table, which is the unit's whole vocabulary in the order +// the compiler settled. +const uint8_t kTableWireMessageForm = 2; + +// The RESERVED build-version id, the second id the language holds back (§5, +// §11), beside the node table's. It is the announcement's one required field, +// and a reserved id in any body but the one whose transport it is, is +// malformed (§3.1). +static const uint64_t kTableBuildVersionFieldId = 0xFFFFFFFFFFFFFFFEull; + +// The reserved NODE-TABLE id's own slot in this unit's vocabulary (§3.3). A +// pointered message names the node table through it, exactly as every other +// field header names its id through a slot. +static const uint64_t kTableNodeTableFieldSlot = 38; + +// THE UNIT'S ANNOUNCEMENT, byte for byte: 61 entries and 508 bytes. It is an +// ordinary form 1 FILE — the form byte, a body carrying the BUILD VERSION +// under the reserved id at kind 9, and the trailer that IS the connection's +// table, slot 1 the reserved id and slots 2 and up the vocabulary under one +// numbering. +// +// The vocabulary is the unit's whole closure in the COOK PROJECTION's order +// (§20.2) — each record in the order the projection renders it and each +// record's fields in the order the projection renders them, then each enum's +// variants and each union's arms — followed by the tail the projection does +// not name: the reserved node-table id, the three blob type ids as bytes, +// string and wstring, and every table's own name id in the projection's sorted +// record order. The tail is UNCONDITIONAL, so an ordinary edit only ever grows +// it at its end and never moves a slot a generated field header carries as a +// literal. +static const int64_t kTableAnnounceBytes = 508; +static const uint8_t kTableAnnounce[ kTableAnnounceBytes ] = { + 0x01, 0x01, 0x09, 0xc7, 0x71, 0x45, 0xca, 0xda, 0x0e, 0x7c, 0x8d, 0x00, + 0xfe, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x30, 0xb1, 0x3a, 0xff, + 0x4a, 0xd9, 0xb1, 0x40, 0x20, 0xea, 0x4d, 0x40, 0x8e, 0xa7, 0x19, 0xaa, + 0x26, 0xa9, 0x02, 0x0c, 0x9b, 0x01, 0x48, 0x78, 0xe9, 0xea, 0x71, 0x6f, + 0x0f, 0x01, 0x82, 0xbf, 0x6f, 0x2c, 0x41, 0x4f, 0xbf, 0x84, 0x78, 0x3e, + 0xf3, 0xa4, 0x48, 0x44, 0x19, 0xab, 0xd7, 0x56, 0x05, 0x4a, 0xa3, 0x30, + 0x67, 0x55, 0x5b, 0x85, 0xc9, 0xe2, 0x4e, 0x30, 0x69, 0x6a, 0xb4, 0x81, + 0xfb, 0x67, 0x4d, 0x1a, 0xcf, 0x7b, 0x27, 0x21, 0x74, 0xa2, 0x79, 0x44, + 0x8e, 0xe2, 0xe5, 0xb1, 0x84, 0x76, 0xbc, 0x2e, 0xef, 0x83, 0x76, 0x1e, + 0xc5, 0x99, 0xf7, 0x82, 0x76, 0x4e, 0x0a, 0xd9, 0xa8, 0x2e, 0x86, 0x70, + 0x84, 0xed, 0xf2, 0x4a, 0xbb, 0xf0, 0x0c, 0x9b, 0xcc, 0xfb, 0x2d, 0x73, + 0x68, 0xb7, 0xf0, 0xae, 0x4c, 0x0c, 0xf6, 0x52, 0xbf, 0xe9, 0xd1, 0x2f, + 0x93, 0xcd, 0xda, 0xdb, 0x22, 0x72, 0x34, 0x7d, 0xf6, 0x0b, 0x72, 0x17, + 0x07, 0x17, 0x02, 0x86, 0x4c, 0xf5, 0x63, 0xaf, 0x54, 0x15, 0x02, 0x86, + 0x4c, 0xf4, 0x63, 0xaf, 0x3a, 0x70, 0x6e, 0x3e, 0x93, 0x43, 0xe5, 0x9d, + 0x3d, 0x62, 0xcb, 0x8f, 0xec, 0xfc, 0xf7, 0x39, 0x09, 0x06, 0x02, 0x86, + 0x4c, 0xeb, 0x63, 0xaf, 0x09, 0x4b, 0x4d, 0x57, 0xaa, 0x33, 0x47, 0xd2, + 0x31, 0x54, 0xaf, 0x1d, 0x19, 0x73, 0x50, 0x12, 0xb2, 0x0f, 0x40, 0x27, + 0x0b, 0x6b, 0x98, 0x01, 0x38, 0x81, 0x0a, 0xf1, 0x1f, 0x06, 0xa7, 0xa3, + 0x0f, 0x62, 0xad, 0x07, 0x77, 0x47, 0x82, 0x5f, 0x42, 0x4f, 0x4f, 0x30, + 0x0d, 0x39, 0x84, 0x1c, 0x86, 0x1b, 0x63, 0x8e, 0xba, 0xad, 0xbc, 0xc4, + 0xec, 0x10, 0x5b, 0x36, 0x19, 0x4a, 0xc9, 0x3d, 0xea, 0x0c, 0xe8, 0x30, + 0x94, 0xfd, 0xe4, 0x7c, 0xec, 0x22, 0x02, 0x86, 0x4c, 0xfc, 0x63, 0xaf, + 0x05, 0x28, 0x02, 0x86, 0x4c, 0xff, 0x63, 0xaf, 0x52, 0x26, 0x02, 0x86, + 0x4c, 0xfe, 0x63, 0xaf, 0xb1, 0x45, 0xc3, 0x44, 0x35, 0xab, 0xfe, 0x73, + 0xc0, 0x7f, 0xb3, 0x8a, 0xbe, 0x08, 0x63, 0x7f, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xe4, 0x4f, 0x1c, 0x4f, 0x47, 0xc0, 0x2e, 0x2f, + 0x58, 0xfc, 0xaf, 0xfa, 0xd8, 0xe0, 0x4b, 0x70, 0xc7, 0xd4, 0x7b, 0x26, + 0xb0, 0x9d, 0x29, 0x5f, 0xcc, 0x14, 0x15, 0x7f, 0xcb, 0xc2, 0x58, 0xd8, + 0x84, 0x9e, 0x3a, 0x55, 0x8b, 0x37, 0xe2, 0x06, 0x2a, 0x89, 0xf5, 0x12, + 0x09, 0xc1, 0xca, 0x0a, 0x44, 0xa2, 0x31, 0xc1, 0xad, 0xa7, 0xee, 0xee, + 0xe8, 0xcf, 0xbf, 0x43, 0x73, 0x18, 0x43, 0xd0, 0x42, 0xad, 0xf6, 0xf8, + 0x59, 0x86, 0x63, 0x91, 0xb7, 0xce, 0x00, 0x7c, 0xd1, 0xc5, 0x34, 0x20, + 0x06, 0x68, 0x47, 0x98, 0xd1, 0xa1, 0xcf, 0x52, 0x5f, 0x82, 0x58, 0xac, + 0x36, 0x15, 0x78, 0x5e, 0xb8, 0x8b, 0x59, 0x6f, 0xc9, 0xc6, 0x86, 0xbb, + 0xc3, 0x64, 0x89, 0x50, 0xd2, 0x8d, 0xa7, 0xf1, 0x80, 0xea, 0x3a, 0xb9, + 0xf1, 0x21, 0xf7, 0x41, 0x11, 0xed, 0xd9, 0xce, 0x96, 0x92, 0x43, 0x8a, + 0xfb, 0x06, 0xc9, 0xfe, 0x19, 0xe1, 0x13, 0xa0, 0xa7, 0x0a, 0xc7, 0x54, + 0x12, 0xd6, 0x40, 0xdc, 0x08, 0xf0, 0xf5, 0xc0, 0x24, 0x5f, 0xf8, 0x33, + 0xc8, 0xfb, 0x85, 0x9a, 0xaf, 0xe0, 0xc9, 0x0c, 0x91, 0x0a, 0x55, 0x60, + 0xf7, 0xa2, 0x07, 0xec, 0x8b, 0x6d, 0x02, 0x86, 0x43, 0xf3, 0xc2, 0x2e, + 0x87, 0x27, 0xcc, 0x86, 0xf0, 0xe0, 0x26, 0x8f, 0x3d, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, +}; + +// AnnounceMeasure is the announcement's byte count, which is a constant of the +// unit and not a walk. +inline int64_t AnnounceMeasure() { return kTableAnnounceBytes; } + +// Announce writes the announcement into the caller's buffer and answers the +// bytes written — exactly AnnounceMeasure's answer — or -1 when the buffer is +// too small. It allocates nothing and walks nothing. +inline int64_t Announce( uint8_t * buffer, int64_t capacity ) +{ + if ( buffer == NULL || capacity < kTableAnnounceBytes ) { return -1; } + memcpy( buffer, kTableAnnounce, (size_t) kTableAnnounceBytes ); + return kTableAnnounceBytes; +} + +// TableVocabulary is ONE DIRECTION of ONE CONNECTION's id table (§3.3): the +// entries an announcement carried, whole, under one numbering with slot 1 the +// reserved build-version id. +// +// A peer holds TWO of these for a connection, the one it writes with and the +// one it reads with, and neither is the other's. A restart opens a fresh +// connection with empty tables and nothing is cached across connections, so +// its whole life is one connection's. It BORROWS the announcement's bytes rather than +// copying them, so a receiver holds one table a direction and its memory is +// the bound below and nothing else. +struct TableVocabulary +{ + // THE CONFORMING DEFAULT BOUND (§3.3): 32 KiB a direction, eight times the + // 500-id unit that is already a large one. A connection's table is bounded + // by nothing the wire carries, so the receiver declares the maximum and an + // announcement above it is refused by name before an entry is touched. + static const int64_t kDefaultMaxEntries = 4096; + + TableIdTable table; + uint64_t build_version = 0; + bool announced = false; + int64_t max_entries = kDefaultMaxEntries; +}; + +// AnnounceRead reads an announcement into one direction's table (§3.3). +// +// THE BOUND IS CHECKED BEFORE ANYTHING IS ALLOCATED: the entry count is a +// fixed little-endian u64 at the end, so a receiver reads it, compares it and +// refuses without touching an entry. After that it is §3's ordinary FILE read, +// because the announcement IS a file, with EXACTLY ONE STRICT CHECK over its +// body: the reserved build-version field present, exactly once, under kind 9, +// eight bytes wide. Everything else is an ordinary field under §4's tolerance, +// so an unknown one is skipped and counted and the announcement can GAIN a +// field in a later minor without a lockstep redeploy. +// +// The FIRST announcement sets the table and it is the only one that can. A +// SECOND is refused by name: it does not replace the table, it does not amend +// it and it changes nothing. A refused announcement sets NO TABLE. +inline bool AnnounceRead( TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + if ( vocabulary.announced ) + { + to->refused = true; + to->reason = second_announcement; + return false; + } + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireForm ) + { + to->refused = true; + to->reason = buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + return false; + } + if ( bytes < 9 ) { to->malformed = true; return false; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + if ( ( lo | ( hi << 32 ) ) > (uint64_t) vocabulary.max_entries ) + { + to->refused = true; + to->reason = vocabulary_too_large; + return false; + } + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else { to->refused = true; to->reason = newer_form; } + return false; + } + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) { to->malformed = true; return false; } + // the body, under §4's tolerance and this form's one strict check + TableReader r( buffer + 1, body_bytes, to, &table ); + uint64_t version = 0; + int32_t seen = 0; + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { to->malformed = true; return false; } + if ( ref == 0 ) { break; } + if ( ref > (uint64_t) table.count || !r.has( 1 ) ) { to->malformed = true; return false; } + const uint64_t id = table.at( ref ); + const uint8_t kind = r.get8(); + if ( id != kTableBuildVersionFieldId ) + { + to->unknown++; + if ( !r.skip( kind ) ) { to->malformed = true; return false; } + continue; + } + if ( kind != 9 || !r.has( 8 ) ) { to->refused = true; to->reason = no_vocabulary; return false; } + version = r.get64(); + seen++; + } + if ( seen != 1 ) { to->refused = true; to->reason = no_vocabulary; return false; } + vocabulary.table = table; + vocabulary.build_version = version; + vocabulary.announced = true; + return true; +} + +inline float table_bits_to_float( uint32_t bits ) { float f; memcpy( &f, &bits, 4 ); return f; } +inline uint32_t table_float_to_bits( float f ) { uint32_t b; memcpy( &b, &f, 4 ); return b; } +inline double table_bits_to_double( uint64_t bits ) { double d; memcpy( &d, &bits, 8 ); return d; } +inline uint64_t table_double_to_bits( double d ) { uint64_t b; memcpy( &b, &d, 8 ); return b; } + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +#ifndef LISTDEMO_SCHEMA_TABLE_ARENA +#define LISTDEMO_SCHEMA_TABLE_ARENA + +namespace listdemo { + +// ---- variable-length tables: tuning constants (docs/SPEC-TABLES.md) ---- +// +// The segment size and the count multiply to exactly 2^32: the u32 reference +// is the arena's hard ceiling, and these constants saturate it rather than +// leaving address space unreachable. Slab handout costs one atomic per slab, +// so per-node allocation costs no synchronization at all. + +static const uint32_t kTableSegmentBits = 22; // 4 MiB segments +static const uint32_t kTableSegmentSize = 1u << kTableSegmentBits; +static const uint32_t kTableSegmentMask = kTableSegmentSize - 1u; +static const uint32_t kTableMaxSegments = 1u << ( 32 - kTableSegmentBits ); // 1024 -> 4 GiB +static const uint32_t kTableSlabBytes = 64u * 1024u; // one atomic per slab +static const uint32_t kTableAlign = 8; // every node starts 8-aligned +static const uint32_t kTableAllocFailed = 0xFFFFFFFFu; + +// ---- THE CALLER'S ALLOCATOR (docs/SPEC-TABLES.md §6.5) ---- +// +// Every allocation the variable-length runtime makes goes through one of +// these — the arena's segments, the pack walk's identity map, the numbering's +// entry array, the packed region, and the tool path's node directory. There is +// no other call to the C library on this path, so a counting allocator sees +// every byte and a game's own heap can own all of it. +// +// It is the shape TableBlockAllocator already has (§19.1): two function +// pointers and a context the caller carries. What it adds is a CONTRACT ON +// alloc — the bytes come back ZEROED. Lock copies whole nodes, PADDING +// INCLUDED, so anything left uninitialized reaches a packed region; the default +// pair reaches that through calloc, which costs nothing measurable because a +// fresh segment is untouched pages either way. +struct TableAllocator +{ + void * ( *alloc )( void * context, int64_t bytes ); // ZEROED bytes, NULL on failure + void ( *free )( void * context, void * pointer ); + void * context; +}; + +// The default pair, and it is the one every entry point takes when the caller +// names none. It calls schema_allocate / schema_release, so a program with its +// own C-library replacement can move the floor without writing a struct at all. +inline void * table_default_alloc( void * context, int64_t bytes ) { (void) context; return schema_allocate( bytes ); } +inline void table_default_free( void * context, void * pointer ) { (void) context; schema_release( pointer ); } + +inline TableAllocator TableDefaultAllocator() +{ + TableAllocator allocator; + allocator.alloc = table_default_alloc; + allocator.free = table_default_free; + allocator.context = NULL; + return allocator; +} + +// ---- TableRef: a relocatable reference (never a machine pointer) ---- +// +// Two encodings, one slot, and the FORM says which is in force: +// +// in the arena — the node's arena offset (segment index in the high bits) +// in a region — the SELF-RELATIVE byte delta from this slot's own address, +// so a deref is one add, needs no base pointer, and a whole +// region relocates by memcpy with zero fix-up +// +// 0 is null in both, and a slot can never name the node that contains it, so +// zero names nothing real in either form. +// +// A REGION DELTA HAS NO REQUIRED SIGN (§6.3). A region is packed depth-first, +// so a node's FIRST reference points forward; every LATER reference to that +// same node points BACK at the one body it already has, which is exactly what +// makes one node one node in a region. Sharing and a back-reference are the +// same fact, and nothing validates a reference by its sign. +// +// IT IS EIGHT BYTES, SIGNED, so ONE REGION REACHES EVERYTHING (§6.3, §7): a +// four-byte slot bounded a region at 2 GiB, and the scale a cook exists for is +// *"100mbs or many gigabytes of data in Assets.bin"*. +struct TableRef +{ + int64_t value = 0; + bool null() const { return value == 0; } +}; + +// TableSlot is what Alloc hands back: usable as the node pointer (write +// fields through it) AND as the reference to store in a pointer field. +template struct TableSlot +{ + T * ptr = NULL; + TableRef ref; + T * operator->() const { return ptr; } + T & operator*() const { return *ptr; } + operator T *() const { return ptr; } + operator TableRef() const { return ref; } + bool null() const { return ptr == NULL; } +}; + +inline uint32_t TableAlignUp( uint32_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( kTableAlign - 1 ); } +inline int64_t TableAlignUp64( int64_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( int64_t( kTableAlign ) - 1 ); } + +// ---- a BYTE BUFFER's node (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// A *bytes or *string slot is a TableRef like every pointer slot, and it names +// a BLOB NODE: this eight-byte header and then the bytes, at offset eight so +// the data is eight-aligned. A *string blob carries one more zero byte after +// its data, so a region hands back a C string with no copy. The node's extent +// is the header plus its bytes, rounded to the arena's alignment like every +// node's; on the wire it is a record whose body is the bytes (§3.1). +struct TableBlob +{ + uint32_t length; + uint32_t zero; +}; + +static const int64_t kTableBlobHeader = 8; // length (u32), then four zero bytes +static const int64_t kTableBlobMaxLength = 0xFFFFFFFF; // a record's length is a u32 (§3.1) + +// the node's storage: the header, the bytes, a string's terminator, rounded +// to the arena's alignment like every node +inline int64_t TableBlobStorage( int64_t length, bool terminated ) +{ + return TableAlignUp64( kTableBlobHeader + length + ( terminated ? 1 : 0 ) ); +} + +// What a read answers: a pointer INTO the region and the length, NULL and +// zero for a null slot. Off a locked region, a loaded one or an opened cook +// the pointer is one add from the slot, and nothing is copied. +struct TableBytesView +{ + const uint8_t * data; + int64_t length; +}; + +struct TableStringView +{ + const char * data; // zero-terminated + int64_t length; +}; + +// What AllocBytes and AllocString hand back: the bytes to write through, the +// length asked for, and the reference to store in the slot — the three +// answers TableSlot gives for a table node. +struct TableBytesSlot +{ + uint8_t * data = NULL; + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +struct TableStringSlot +{ + char * data = NULL; // room for length bytes and the terminator, already zero + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +// ---- the arena: segmented, slab-handed, lock-free by ownership ---- +// +// Allocation is thread-local inside a worker's slab — no atomics on the node +// path. A worker takes its next slab with ONE compare-exchange, and a new +// segment is published with one more. Nothing ever moves: a segment, once +// allocated, lives untouched until the arena is torn down, so a T* obtained +// from Alloc stays valid while other workers allocate, and an offset stays +// correct while the arena grows. +// +// The model this DELIBERATELY refuses: one buffer under a lock, grown by +// realloc. A realloc moves the buffer under workers mid-write; offsets fix +// identity but not the raw references already resolved from them, and the +// resulting corruption is invisible until much later. Segments never move, so +// that bug class cannot be written here. +// +// Slack: at most one slab tail per worker plus one slab per segment (a slab +// that will not fit is skipped rather than split), i.e. under 2% of a segment +// plus threads x 64 KiB. That is the price of never synchronizing per node. +struct TableArena +{ + std::atomic segments[ kTableMaxSegments ]; + std::atomic cursor; // (segment << kTableSegmentBits) | bytes handed out + bool locked = false; // MONOTONIC: Lock() is one-way, there is no unlock + // THE ARENA CARRIES ITS OWN, so everything downstream of a builder — + // segments, pack map, numbering, region, node directory — allocates through + // the one pair the caller named, with nothing to thread by hand. + TableAllocator allocator; +}; + +inline void TableArenaInit( TableArena & arena, TableAllocator allocator ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + arena.segments[i].store( NULL, std::memory_order_relaxed ); + } + arena.cursor.store( 0, std::memory_order_relaxed ); + arena.locked = false; + arena.allocator = allocator; +} + +inline void TableArenaShutdown( TableArena & arena ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + uint8_t * segment = arena.segments[i].exchange( NULL, std::memory_order_acq_rel ); + if ( segment != NULL ) { arena.allocator.free( arena.allocator.context, segment ); } + } + arena.cursor.store( 0, std::memory_order_relaxed ); +} + +// one L1 load plus an add: the segment table is 8 KiB and stays hot +inline uint8_t * TableArenaAt( const TableArena & arena, uint32_t offset ) +{ + return arena.segments[ offset >> kTableSegmentBits ].load( std::memory_order_relaxed ) + ( offset & kTableSegmentMask ); +} + +// TableArenaGrabSlab hands one worker its next private slab. Returns +// kTableAllocFailed when the arena's address space or the allocator is +// exhausted — a loud refusal, never a silent smaller slab. +inline uint32_t TableArenaGrabSlab( TableArena & arena ) +{ + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t segment = cursor >> kTableSegmentBits; + uint32_t used = cursor & kTableSegmentMask; + // strictly less: a slab is never split across segments, and the tail + // is the documented slack + if ( used + kTableSlabBytes < kTableSegmentSize ) + { + if ( arena.segments[segment].load( std::memory_order_acquire ) == NULL ) + { + // THE SEGMENT COMES BACK ZEROED, which is the allocator's + // contract and not an extra pass here: Lock copies whole nodes, + // PADDING INCLUDED, so anything uninitialized reaches a packed + // region. Value-initializing a node with placement new zeroes + // its MEMBERS and not its padding, so the zeroing has to happen + // at the segment or not at all. It costs nothing measurable: a + // fresh segment is untouched pages either way, and the default + // pair's calloc has the kernel hand them over zeroed. + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, (int64_t) kTableSegmentSize ); + if ( memory == NULL ) { return kTableAllocFailed; } + uint8_t * expected = NULL; + if ( !arena.segments[segment].compare_exchange_strong( expected, memory, std::memory_order_acq_rel ) ) + { + // another worker published this segment first + arena.allocator.free( arena.allocator.context, memory ); + } + } + if ( arena.cursor.compare_exchange_weak( cursor, cursor + kTableSlabBytes, std::memory_order_acq_rel ) ) + { + return ( segment << kTableSegmentBits ) | used; + } + continue; + } + uint32_t next_segment = segment + 1; + if ( next_segment >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + arena.cursor.compare_exchange_weak( cursor, next_segment << kTableSegmentBits, std::memory_order_acq_rel ); + } +} + +// TableArenaGrabSpan reserves a SPAN of the arena's address space for one node +// larger than a slab — a BYTE BUFFER of any size (docs/SPEC-TABLES.md §2.5) — +// and allocates it as one contiguous block. It takes whole segment indices +// from the cursor, starting at the index after the cursor's so nothing else +// is ever handed out inside the span, and publishes the block under the first +// of them; the indices the span covers past that one stay NULL, which is +// enough, because only a node's START is ever resolved through the segment +// table and a blob's bytes follow its header inside the one allocation. The +// unused tail of the segment the cursor was in is slack, like a slab tail. +// Returns kTableAllocFailed when the address space or the allocator is +// exhausted — a loud refusal, never a smaller blob. +inline uint32_t TableArenaGrabSpan( TableArena & arena, int64_t bytes ) +{ + if ( bytes <= 0 || bytes > ( (int64_t) kTableMaxSegments - 2 ) * (int64_t) kTableSegmentSize ) { return kTableAllocFailed; } + const uint32_t spanned = (uint32_t) ( ( bytes + kTableSegmentSize - 1 ) >> kTableSegmentBits ); + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t start = ( cursor >> kTableSegmentBits ) + 1; + if ( start + spanned >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + uint32_t next = ( start + spanned ) << kTableSegmentBits; + if ( !arena.cursor.compare_exchange_weak( cursor, next, std::memory_order_acq_rel ) ) { continue; } + // the span is this worker's now: nothing else can publish under its + // first index, so a plain store suffices, and the block comes back + // ZEROED like every segment — the blob's bytes and its tail are zeros + // until written + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, bytes ); + if ( memory == NULL ) { return kTableAllocFailed; } + arena.segments[start].store( memory, std::memory_order_release ); + return start << kTableSegmentBits; + } +} + +// ---- TableWorker: one thread's allocation front ---- +// +// The threading contract, stated plainly: +// * Alloc on YOUR OWN worker is safe concurrently with any other worker's. +// No locks, no atomics per node. +// * Writing fields of a node ANOTHER worker allocated is your own +// synchronization problem — this runtime does not arbitrate it. +// * Lock and Save are single-threaded: call them after the workers have +// joined. +struct TableWorker +{ + TableArena * arena = NULL; + uint32_t next = 0; + uint32_t end = 0; + + template TableSlot Alloc() + { + static_assert( alignof( T ) <= kTableAlign, "a table node's alignment must fit the arena's" ); + TableSlot slot; + if ( arena == NULL || arena->locked ) { return slot; } + uint32_t bytes = TableAlignUp( (uint32_t) sizeof( T ) ); + if ( bytes > kTableSlabBytes ) { return slot; } // a node larger than a slab: refused, never split + if ( end == 0 || next + bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return slot; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + uint32_t at = next; + next += bytes; + // A NODE IS BORN IN TWO HALVES: start its lifetime in the raw + // storage, then write the declared defaults ONE MEMBER AT A TIME. + // + // It is "T", not "T{}". Value-initialising the whole aggregate says + // the same thing and costs cl O(BYTES) TO COMPILE — it expands element + // by element in its front end — while both halves here cost + // O(declarations). The slab cap below refuses a large node at RUN + // TIME and bounds nothing at compile time: the cost is paid by + // whatever T a caller instantiates this with. + // Padding is not the difference: value-initialisation zeroes MEMBERS + // and not padding either way, which is why the segment is calloc'd. + // + // TableReset is an OVERLOAD SET, one per closure member, reached from + // this template by argument-dependent lookup on T's own namespace — + // Alloc is a template and cannot spell Reset. + // + // The reset is here because ONE DEFINITION SAYS WHAT THE DECLARED + // DEFAULTS ARE, and it is Reset. Default-initialisation lands on + // the same values today, because a member with a non-zero default + // carries a member initializer that says so — but that is the class + // definition agreeing with Reset, not the arena reading it, and #320's + // fix was itself a pass that MOVED initialisation between the two. + // The arena reads the definition. + slot.ptr = new ( TableArenaAt( *arena, at ) ) T; + TableReset( *slot.ptr ); + slot.ref.value = at; + return slot; + } + + // Alloc a BYTE BUFFER's node of exactly length bytes (docs/SPEC-TABLES.md + // §2.5): the blob header and its bytes, zeroed, in this thread's slab when + // it fits and in a span of the arena's own when it does not. NULL is the + // arena locked, a length below zero or past a record's u32, or the + // allocator refusing. The offset comes back for the reference. + TableBlob * AllocBlob( int64_t length, bool terminated, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( length < 0 || length > kTableBlobMaxLength ) { return NULL; } + const int64_t bytes = TableBlobStorage( length, terminated ); + if ( bytes > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, bytes ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + } + else + { + if ( end == 0 || next + (uint32_t) bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) bytes; + } + TableBlob * blob = (TableBlob *) TableArenaAt( *arena, at ); + blob->length = (uint32_t) length; // the bytes after it are the segment's zeros + blob->zero = 0; + return blob; + } + + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries + // no type id, takes no index and has no Reset, so it goes through the same + // slab and span the blob path uses rather than through Alloc. + uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( bytes <= 0 || align > (int64_t) kTableAlign ) { return NULL; } + const int64_t rounded = TableAlignUp64( bytes ); + if ( rounded > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, rounded ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + return TableArenaAt( *arena, at ); + } + if ( end == 0 || next + (uint32_t) rounded > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) rounded; + return TableArenaAt( *arena, at ); // the segment came back zeroed + } + // a *bytes node: the bytes to write through, and the reference to store + TableBytesSlot AllocBytes( int64_t length ) + { + TableBytesSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, false, at ); + if ( blob == NULL ) { return slot; } + slot.data = (uint8_t *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } + + // a *string node: room for length bytes and the zero byte after them + TableStringSlot AllocString( int64_t length ) + { + TableStringSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, true, at ); + if ( blob == NULL ) { return slot; } + slot.data = (char *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } +}; + +// ---- TablePackMap: the pack walk's identity map (docs/SPEC-TABLES.md §3.1, §6.2) ---- +// +// ONE ENTRY PER REACHABLE NODE, and that map IS identity: a node must know +// where it landed to be named a second time, so Lock packs a shared node ONCE +// and every later reference resolves to the one body it already has. That is +// the same first-visit numbering the wire uses, so the pack order and the node +// order are one order. +// +// COLOURING AN ENTRY WHILE ITS DESCENT IS OPEN COSTS ONE BIT, and it is what +// makes a data cycle free to refuse: a reference to an entry still open is a +// cycle, and Lock returns failure rather than recursing away. The ROOT's entry +// is open for the whole walk. +// +// The map is proportional to NODES, never to bytes, and it lives on the +// AUTHORING side, where §6.5 licenses allocation. Nothing on the reading path +// ever builds one. +struct TablePackEntry +{ + const void * key; // the node's address in the graph being packed + int64_t offset; // where that node landed in the region + uint8_t open; // its descent is still open: a reference here is a cycle +}; + +struct TablePackMap +{ + TablePackEntry * entries = NULL; + int64_t capacity = 0; // a power of two, or zero while empty + int64_t count = 0; + TableAllocator allocator; // the caller's, carried from the walk that built it +}; + +inline void TablePackMapInit( TablePackMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TablePackMapShutdown( TablePackMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TablePackMapInit( map, map.allocator ); +} + +// The two walks behind Lock re-derive the SAME map from the same graph — the +// numbering is never carried between them (§3.1) — so the second starts from +// an empty map and keeps the capacity the first paid for. +inline void TablePackMapReset( TablePackMap & map ) +{ + if ( map.entries != NULL ) { memset( map.entries, 0, (size_t) map.capacity * sizeof( TablePackEntry ) ); } + map.count = 0; +} + +// open addressing, linear probing, a multiply-shift hash over the address: a +// node key is a pointer and its low bits are alignment, so the low bits alone +// would collide on every node of one type +inline int64_t TablePackMapSlot( const TablePackMap & map, const void * key ) +{ + uint64_t hash = (uint64_t) (uintptr_t) key; + hash *= 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != NULL && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TablePackEntry * TablePackMapFind( TablePackMap & map, const void * key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TablePackEntry * entry = &map.entries[ TablePackMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +// QUADRUPLING, not doubling, and the reason is measured: growth rehashes every +// entry, and on a graph of 131,071 nodes the doubling schedule spent 45% of +// Lock in rehashing alone. Quadrupling from 1024 buys 1.35x on that graph and +// keeps the map NODE-proportional (§6.2) — under 128 bytes a node at its +// worst, right after a grow, and about 64 on average. +inline bool TablePackMapGrow( TablePackMap & map ) +{ + TablePackMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 1024; + grown.entries = (TablePackEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TablePackEntry ) ); + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == NULL ) { continue; } + grown.entries[ TablePackMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// REACH a node: one probe answers both questions the walk has. A true "taken" +// says this is a FIRST visit, and the entry is now the node's, coloured open +// at "offset"; otherwise the entry is the one the node already has, and its +// open bit says cycle or sharing. NULL is an allocation failure, and it is a +// refusal like any other: Lock fails rather than packing a graph it cannot +// track. +// +// It is one call and not a find followed by an insert because the walk asks +// this question twice per node — once to measure, once to pack — and every +// probe is a miss into a table larger than L2. +inline TablePackEntry * TablePackMapReach( TablePackMap & map, const void * key, int64_t offset, bool & taken, int64_t & slot ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TablePackMapGrow( map ) ) { return NULL; } + } + slot = TablePackMapSlot( map, key ); + TablePackEntry * entry = &map.entries[slot]; + taken = entry->key != key; // an empty slot is a first visit; the key is never NULL + if ( taken ) + { + entry->key = key; + entry->offset = offset; + entry->open = 1; + map.count++; + } + return entry; +} + +// The descent finished: the node keeps its entry — identity outlives the +// descent — and stops being a cycle. The "hint" is the slot Reach returned, and it +// is checked against the key rather than trusted, so a rehash between the two +// costs a second probe instead of correctness. +inline void TablePackMapClose( TablePackMap & map, const void * key, int64_t hint ) +{ + if ( hint >= 0 && hint < map.capacity && map.entries[hint].key == key ) + { + map.entries[hint].open = 0; + return; + } + TablePackEntry * entry = TablePackMapFind( map, key ); + if ( entry != NULL ) { entry->open = 0; } +} + +// ---- resolution contexts: which encoding a walk is reading ---- + +struct TableArenaCtx { const TableArena * arena; }; +struct TableRegionCtx {}; + +// ---- a BYTE BUFFER's resolution (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// The same two encodings a table pointer has, resolved the same way: a +// self-relative delta in a region — one add, no base — and an arena offset +// while the builder is mutable. The blob is reached through its header, and a +// view is the header plus eight and the header's first word. Nothing here +// allocates and nothing copies: off a locked region, a loaded one or an +// opened cook the view points INTO the region. +inline const TableBlob * TableBlobAt( const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableRegionCtx &, const TableRef & ref ) { return TableBlobAt( ref ); } +inline const TableBlob * TableBlobAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} + +inline TableBytesView TableBytesViewOf( const TableBlob * blob ) +{ + TableBytesView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const uint8_t *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} +inline TableStringView TableStringViewOf( const TableBlob * blob ) +{ + TableStringView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const char *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} + +// the const form's hot path: one add, no base +inline TableBytesView TableBytesAt( const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ref ) ); } +inline TableStringView TableStringAt( const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ref ) ); } +// and the context forms a walk uses: a region context, an arena context, or +// the arena itself while the builder is mutable +template inline TableBytesView TableBytesAt( const Ctx & ctx, const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ctx, ref ) ); } +template inline TableStringView TableStringAt( const Ctx & ctx, const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ctx, ref ) ); } + +// allocate a blob in the arena and point the slot at it; the slot holds the +// arena offset, as every slot does while the builder is mutable +inline uint8_t * TableBytesEmplace( TableWorker & worker, TableRef & slot, int64_t length ) +{ + TableBytesSlot allocated = worker.AllocBytes( length ); + slot = allocated.ref; + return allocated.data; +} +// the text is copied in when one is given; a NULL text leaves the zeros for +// the caller to fill +inline char * TableStringEmplace( TableWorker & worker, TableRef & slot, const char * text, int64_t length ) +{ + TableStringSlot allocated = worker.AllocString( length ); + slot = allocated.ref; + if ( allocated.data != NULL && text != NULL && length > 0 ) { memcpy( allocated.data, text, (size_t) length ); } + return allocated.data; +} + +// ---- the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table, and a +// pointer field rides as an INDEX into it under kind 17. The encoding is +// flat: no pointer edge is a nesting level, so a chain's length is not a depth, +// and two references to one node are one node. +// +// THE FIELD RIDES ONCE: an L with sixty-four bits of capability frames a +// numbering of any size, so the whole numbering is one contiguous payload and a +// save's node bodies have no aggregate ceiling. + +static const uint64_t kTableNodeIndexNull = 0; // absence and null are one value +static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts the table + +// The not-materialized sentinel (§6.3): a record whose type id this build could +// not name. Distinct from every real offset including the root's 0, so an index +// resolving through it yields NULL and can never fabricate the root. +static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; + +// ---- the numbering, on the SAVE side ---- +// +// One entry per reachable node in FIRST-VISIT order, so entry k is node index +// k + 2. The two thunks are what let one loop write a table of mixed types: the +// numbering walk knows each target's type STATICALLY at the site it numbers it, +// so it stores the instantiation there and the loop never asks what a node is. +struct TableNumbering; + +struct TableNodeEntry +{ + const void * node; + uint64_t type_id; + // the type id's MESSAGE-FORM SLOT (docs/SPEC-TABLES.md §3.3), stored where + // the numbering walk stores the id itself and for the same reason: the + // target's type is known STATICALLY at the site that numbers it, so a + // form 2 save reads the slot out of the entry instead of looking an id up. + // Every pointer target's type id is an entry of the announcement, which is + // what makes the slot a compile-time fact of a POINTERED message too. + uint64_t type_slot; + int64_t ( * measure )( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ); + bool ( * save )( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ); +}; + +struct TableNumbering +{ + TablePackMap seen; // node -> index; the ROOT is index 1, open for the whole walk + TableNodeEntry * entries = NULL; + int64_t count = 0; + int64_t capacity = 0; +}; + +// The numbering allocates through the map's pair rather than carrying a second +// copy of it: one numbering is one walk, and a walk has one allocator. +inline void TableNumberingInit( TableNumbering & n, TableAllocator allocator ) +{ + TablePackMapInit( n.seen, allocator ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +inline void TableNumberingShutdown( TableNumbering & n ) +{ + TableAllocator allocator = n.seen.allocator; + TablePackMapShutdown( n.seen ); + allocator.free( allocator.context, n.entries ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +// The index a numbered node was given, for the save that writes it into a +// pointer slot. False means the two walks disagree about the graph, which is a +// refusal and never a guess. +inline bool TableNumberingIndex( const TableNumbering & n, const void * node, uint64_t & index ) +{ + if ( n.seen.capacity == 0 ) { return false; } + const TablePackEntry & entry = n.seen.entries[ TablePackMapSlot( n.seen, node ) ]; + if ( entry.key != node ) { return false; } + index = (uint64_t) entry.offset; + return true; +} + +inline bool TableNumberingAppend( TableNumbering & n, const TableNodeEntry & entry ) +{ + if ( n.count == n.capacity ) + { + // GROW BY COPY, never by realloc: the allocator hook is a PAIR, and a + // game's heap is not required to have a resize primitive at all. The + // schedule quadruples, so the copying is amortized to a constant per + // entry and the growth is the same growth it always was. + int64_t capacity = n.capacity != 0 ? n.capacity * 4 : 256; + TableAllocator allocator = n.seen.allocator; + TableNodeEntry * grown = (TableNodeEntry *) allocator.alloc( allocator.context, capacity * (int64_t) sizeof( TableNodeEntry ) ); + if ( grown == NULL ) { return false; } + if ( n.entries != NULL ) + { + memcpy( grown, n.entries, (size_t) n.count * sizeof( TableNodeEntry ) ); + allocator.free( allocator.context, n.entries ); + } + n.entries = grown; + n.capacity = capacity; + } + n.entries[n.count++] = entry; + return true; +} + +// The thunks the numbering stores. Each resolves to the closure member's own +// MeasureBody / SaveBodyFields through an overload set in the member's DECLARING +// file, reached by argument-dependent lookup at instantiation — the same bridge +// the arena's TableReset uses, and the reason a numbering may span the files of +// one unit without any file naming another's members. +template +inline int64_t TableNodeMeasureThunk( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ) +{ + return TableNodeMeasure( *(const Ctx *) ctx, numbering, ids, *(const T *) node ); +} + +template +inline bool TableNodeSaveThunk( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ) +{ + return TableNodeSave( *(const Ctx *) ctx, numbering, w, ids, *(const T *) node ); +} + +// ---- a BYTE BUFFER's record (docs/SPEC-TABLES.md §2.5, §3.1) ---- +// +// A blob rides as a node record under one of two RESERVED type ids — the fold +// a table's name takes, over the keywords "bytes" and "string", which no table +// can be named — with the bytes as its body and nothing framed inside. These +// two thunks are what the numbering stores for a blob, as it stores a +// member's codec for a table: the length, and the bytes verbatim. +static const uint64_t kTableBytesTypeId = 0x2f2ec0474f1c4fe4ull; // fnv1a64( "bytes" ) +static const uint64_t kTableStringTypeId = 0x704be0d8faaffc58ull; // fnv1a64( "string" ) + +template +inline int64_t TableBlobMeasureThunk( const void *, const TableNumbering &, TableIds &, const void * node ) +{ + return (int64_t) ( (const TableBlob *) node )->length; +} + +template +inline bool TableBlobSaveThunk( const void *, const TableNumbering &, TableWriter & w, TableIds &, const void * node ) +{ + const TableBlob * blob = (const TableBlob *) node; + w.raw( (const void *) ( blob + 1 ), (int64_t) blob->length ); + return true; +} + +// TableNodeTableMeasure and TableNodeTableSave are the framing, and they are +// ONE fill rule written twice — measure derives it from the graph and save +// derives the same one, which is what makes measure == save hold across a +// pointer graph (§3.1). +// +// The field rides ONCE, under the reserved id, kind 12: the payload opens with +// the count and then carries the records back to back, each a type id +// REFERENCE, a length and a body. The reserved id is interned BEFORE the +// records, and a record's type id before its body, which is the first-use order +// the trailer is written in (§3). +template +inline int64_t TableNodeTablePayload( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + int64_t payload = TableLebBytes( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + payload += TableLebBytes( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return -1; } + payload += TableLebBytes( (uint64_t) body ) + body; + } + return payload; +} + +template +inline int64_t TableNodeTableMeasure( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return 0; } // a root that reaches no nodes writes none of them + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return -1; } + return TableLebBytes( ref ) + 1 + TableLebBytes( (uint64_t) payload ) + payload; +} + +template +inline bool TableNodeTableSave( const Ctx & ctx, TableWriter & w, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return true; } + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return false; } + w.putleb( ref ); + w.put8( 12 ); // kind 12 is the opaque byte payload: a reader that cannot name the id skips by L + w.putleb( (uint64_t) payload ); + w.putleb( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + w.putleb( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return false; } + w.putleb( (uint64_t) body ); + if ( !n.entries[k].save( (const void *) &ctx, n, w, ids, n.entries[k].node ) ) { return false; } + } + return true; +} + +// ---- the numbering, on the LOAD side: a region's NODE DIRECTORY (§6.3) ---- +// +// The wire's numbering made resident: one entry per numbered node, in index +// order, position i describing node index i + 1 — so position 0 is the ROOT at +// offset 0. It is ATTRIBUTION, and attribution is separable: nothing that reads +// a structure touches it, a deref is one add on a self-relative offset, and a +// caller may release it once Load returns. +struct TableNodeDirEntry +{ + uint64_t offset; + uint64_t type_id; +}; + +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; + +// TableNodeMap is what a pointer slot resolves through while a body decodes. +struct TableNodeMap +{ + uint8_t * base = NULL; + const TableNodeDirEntry * entries = NULL; + int64_t count = 0; // the ROOT's entry included, so it is records + 1 + bool good = false; // the node table read whole; a numbering that failed resolves nothing + // WHERE THE NODES LIVE, and therefore what a resolved slot holds: a region + // takes the SELF-RELATIVE delta so a deref is one add, and the tool's + // builder path takes the node's ARENA OFFSET (§6.3). + bool arena = false; + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. + TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; +}; + +// TableNodeResolve places one node index in a pointer slot, and every failure +// is one of §4's events with the pointer left null. The declared TARGET type id +// is checked at every index, the root's included: the root carries no record +// and therefore no wire type id, so the READER'S OWN root type is what the +// claim is checked against. +inline void TableNodeResolve( const TableNodeMap & map, TableRef & slot, uint64_t index, uint64_t target, TableReport * report ) +{ + slot.value = 0; + if ( index == kTableNodeIndexNull || !map.good ) { return; } + if ( index - 1 >= (uint64_t) map.count ) + { + report->malformed = true; // an index above node_count + 1 + return; + } + const TableNodeDirEntry & entry = map.entries[index - 1]; + if ( entry.offset == kTableNodeAbsent ) + { + // a node whose type id this build could not name KEEPS ITS INDEX, and + // every pointer naming it reads null. The unknown was counted once, at + // the node, not once per pointer. + return; + } + if ( entry.type_id != target ) + { + report->kind_mismatch++; + return; + } + slot.value = map.arena ? (int64_t) entry.offset + : (int64_t) ( ( map.base + entry.offset ) - (const uint8_t *) &slot ); +} + +// ---- the record SCAN, and it is the whole of load's bound (§3.1) ---- +// +// Reading follows no reference. The scan walks the root body's top-level fields, +// finds the ONE under the reserved id, and reads records out of its payload in +// order — the field rides once, so nothing is copied to make a body contiguous +// and the generated body decoder never learns the transport exists. +struct TableNodeScan +{ + TableReader fields; // over the ROOT body, skipping past everything else + const uint8_t * payload; // the node-table field's payload + int64_t payload_size; + int64_t payload_offset; + bool opened; // the root body has been walked for the field + uint64_t declared; + int64_t records; + bool present; // the root body carries a node table at all + bool malformed; + const TableIdTable * ids; +}; + +inline TableNodeScan TableNodeScanBegin( const uint8_t * body, int64_t size, TableReport * report, const TableIdTable * ids ) +{ + TableNodeScan s = { TableReader( body, size, report, ids ), NULL, 0, 0, false, 0, 0, false, false, ids }; + return s; +} + +// find the node-table field, or answer false when the root body has none. A +// body carrying an id more than once is legal input and THE LAST OCCURRENCE +// WINS (docs/SPEC-TABLES.md §3), so the walk runs to the terminator and keeps +// the last rather than stopping at the first. +inline bool TableNodeScanOpen( TableNodeScan & s ) +{ + if ( s.opened ) { return false; } + s.opened = true; + for ( ;; ) + { + uint64_t ref = 0; + if ( !s.fields.getleb( ref ) ) { break; } + if ( ref == 0 ) { break; } // the terminator + if ( s.ids == NULL || ref > (uint64_t) s.ids->count ) { break; } + const uint64_t id = s.ids->at( ref ); + if ( !s.fields.has( 1 ) ) { break; } + const uint8_t kind = s.fields.get8(); + if ( id == kTableNodeTableFieldId ) + { + s.present = true; + if ( kind != 12 ) { s.malformed = true; return false; } + uint64_t length = 0; + if ( !s.fields.getleb( length ) || !s.fields.room( length ) ) { s.malformed = true; return false; } + s.payload = s.fields.buffer + s.fields.offset; + s.payload_size = (int64_t) length; + s.fields.offset += (int64_t) length; + continue; + } + if ( !s.fields.skip( kind ) ) { break; } + } + if ( s.payload == NULL ) { return false; } + TableReader head( s.payload, s.payload_size, s.fields.report, s.ids ); + if ( !head.getleb( s.declared ) ) { s.malformed = true; return false; } + s.payload_offset = head.offset; + return true; +} + +// the next record, or false at the end of the table — s.malformed says whether +// the end was the end or the framing giving out +inline bool TableNodeScanNext( TableNodeScan & s, uint64_t & type_id, const uint8_t * & body, int64_t & length ) +{ + if ( !s.opened && !TableNodeScanOpen( s ) ) { return false; } + if ( s.payload == NULL || s.payload_offset >= s.payload_size ) { return false; } + TableReader rec( s.payload, s.payload_size, s.fields.report, s.ids ); + rec.offset = s.payload_offset; + uint64_t ref = 0; + if ( !rec.getleb( ref ) || ref == 0 || s.ids == NULL || ref > (uint64_t) s.ids->count ) + { + s.malformed = true; // a type id reference of 0, or one past the table + return false; + } + type_id = s.ids->at( ref ); + uint64_t declared_length = 0; + if ( !rec.getleb( declared_length ) ) + { + s.malformed = true; // a record whose length is damaged + return false; + } + if ( declared_length > (uint64_t) ( s.payload_size - rec.offset ) ) + { + s.malformed = true; // a record whose length runs past its field + return false; + } + body = s.payload + rec.offset; + length = (int64_t) declared_length; + s.payload_offset = rec.offset + length; + s.records++; + return true; +} + +// The record scan is AUTHORITATIVE: node_count is data from the wire, and a +// count that disagrees with the scan is malformed. Nothing is sized from it +// before the scan has confirmed it. +inline bool TableNodeScanWhole( TableNodeScan & s ) +{ + if ( s.malformed ) { return false; } + if ( !s.present ) { return true; } // no node table at all is not a broken one + return s.declared == (uint64_t) s.records; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_ARENA + +#ifndef LISTDEMO_SCHEMA_TABLE_EXTENT +#define LISTDEMO_SCHEMA_TABLE_EXTENT + +namespace listdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_EXTENT + +#ifndef LISTDEMO_SCHEMA_TABLE_MAP +#define LISTDEMO_SCHEMA_TABLE_MAP + +namespace listdemo { + +// ---- a MAP: a sorted entry array, and the lookup over it (§2.8) ---- +// +// On the wire, in a region and in a cook a map is an array of one generated +// ENTRY table held in ascending key order. What this adds is Find — a binary +// search over that array where it lies — and a builder that inserts, replaces +// and erases by key. Nothing here is stored: a region and a cook carry the +// array and the count, and not one byte about a hash or a probe. + +// entries carved from ONE call to the allocator pair; a new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableMapSegmentEntries = 32; + +// TableDeclRef names a type in an unevaluated context and is never defined — +// what 's declval is for, without the include the generated corpus +// refuses to pay for (the iterator_traits note, §13.9). +template T & TableDeclRef(); + +// THE ORDER IS TOTAL, AND IT IS THE SAME IN NINE LANGUAGES (§2.8). Integers +// compare by VALUE, signed for the signed kinds and unsigned for the unsigned. +// Strings compare by BYTES, unsigned, a shorter string that is a prefix of a +// longer one first: memcmp over the common length, then the lengths. Never a +// locale, never a code point, never a case fold. +inline int TableKeyOrder( uint64_t a, uint64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( int64_t a, int64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( const char * a, int32_t a_length, const char * b, int32_t b_length ) +{ + const int32_t common = a_length < b_length ? a_length : b_length; + if ( common > 0 ) + { + const int order = memcmp( (const void *) a, (const void *) b, (size_t) common ); + if ( order != 0 ) { return order < 0 ? -1 : 1; } + } + return a_length < b_length ? -1 : ( a_length > b_length ? 1 : 0 ); +} + +// the length of a NUL-terminated key at a call site, bounded by the storage it +// has to fit: a key one byte longer than the bound is refused, never truncated +inline int32_t TableKeyLength( const char * key, int32_t bound ) +{ + if ( key == NULL ) { return 0; } + for ( int32_t i = 0; i <= bound; i++ ) { if ( key[i] == 0 ) { return i; } } + return bound + 1; // longer than the bound: the caller refuses it +} + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.8, §7.2) ---- +// +// An int64 self-relative reference to the entry array and an int32 count, then +// padding to eight. The reference is a TableRef like a pointer's: in the arena +// it names the builder's HEAD, in a region it is the delta from the slot to +// the first entry, and 0 is the empty map in both. +template struct TableMap +{ + TableRef entries; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Entry * Entries() const + { + return entries.value != 0 ? (const Entry *) ( (const uint8_t *) &entries + entries.value ) : NULL; + } + int32_t size() const { return count; } + + // FIND: floor( log2 n ) + 1 key compares, in place, no allocation. NULL + // when absent, and on a map[K]*T the RESOLVED pointer, which is what a + // pointer field's accessor answers. + template const Entry * FindEntry( Key key ) const + { + const Entry * base = Entries(); + int32_t low = 0, high = count; + while ( low < high ) + { + const int32_t mid = low + ( high - low ) / 2; + const int order = TableEntryOrder( base[mid], key ); + if ( order == 0 ) { return base + mid; } + if ( order < 0 ) { low = mid + 1; } else { high = mid; } + } + return NULL; + } + // the return type is DEDUCED, so it is worked out when a call site + // instantiates Find and not when the holder's record declares the slot — + // which is what lets the entry's own overloads be declared after it + template auto Find( Key key ) const + { + return TableEntryFound( FindEntry( key ) ); + } + + // ---- iteration: ASCENDING key order, the key beside the value ---- + // + // A proxy BY VALUE, the keyed array's shape (§2.4): for ( auto [ key, + // value ] : map ). It carries no iterator_traits, for the reason + // TableKeyed's does not (§13.9). + struct ConstEntry + { + decltype( TableEntryKey( TableDeclRef() ) ) key; + decltype( TableEntryFound( (const Entry *) NULL ) ) value; + }; + + struct ConstIterator + { + const Entry * at; + ConstEntry operator*() const { return ConstEntry{ TableEntryKey( *at ), TableEntryFound( at ) }; } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Entries() }; } + ConstIterator end() const { return ConstIterator{ Entries() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.8, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first entry is inserted. Each +// segment is a fixed number of entries carved from one call to the allocator +// pair. An entry's address is stable for the arena's life, so a value handed +// back by an insert stays valid while other entries arrive. +struct TableMapHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an insert appends into + int32_t live; + int32_t dead; +}; + +template struct TableMapSegment +{ + TableRef next; + int32_t used; // entries carved from this segment + int32_t padding; + uint32_t dead[ ( kTableMapSegmentEntries + 31 ) / 32 ]; // Erase marks one bit, never the entry + Entry entries[ kTableMapSegmentEntries ]; +}; + +inline bool TableMapSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// ---- the ORDERED CURSOR the four writing walks read (§2.8) ---- +// +// Measure, Save, Lock and Cook each write a map's entries in ascending key +// order with no key twice, deriving the order from the builder's entries as +// each walk derives the numbering (§3.1). Nothing passes between them, so +// measure == save over a map is a real check on two sorts agreeing. +// +// A REGION is already sorted, so its cursor is the array in place and +// allocates nothing. The BUILDER's is the sort: an array of entry pointers +// allocated through the pair and released before the walk returns, because +// sorting the segments themselves would move entries whose addresses a caller +// holds. +template struct TableMapCursor +{ + const Entry * const * order = NULL; // the builder's form: sorted pointers + const Entry * entries = NULL; // the region's form: the array in place + int32_t count = 0; + TableAllocator allocator; + bool ok = false; + const Entry * operator[]( int32_t index ) const + { + return order != NULL ? order[index] : entries + index; + } +}; + +// heapsort: O( n log n ) once per map, no recursion, no allocation past the +// pointer array the caller already paid for +template inline void TableMapSort( const Entry ** order, int32_t count ) +{ + for ( int32_t start = count / 2 - 1; start >= 0; start-- ) + { + int32_t root = start; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= count ) { break; } + if ( child + 1 < count && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * swap = order[root]; order[root] = order[child]; order[child] = swap; + root = child; + } + } + for ( int32_t end = count - 1; end > 0; end-- ) + { + const Entry * swap = order[0]; order[0] = order[end]; order[end] = swap; + int32_t root = 0; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= end ) { break; } + if ( child + 1 < end && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * hold = order[root]; order[root] = order[child]; order[child] = hold; + root = child; + } + } +} + +// the REGION form: the array is already sorted, so the cursor is the array +template +inline TableMapCursor TableMapOrder( const TableRegionCtx &, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.entries = map.Entries(); + cursor.count = map.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: gather the LIVE entries out of the segment chain in +// insertion order, then sort. A dead entry costs nothing on any wire (§2.8). +template +inline TableMapCursor TableMapOrder( const TableArena & arena, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.allocator = arena.allocator; + cursor.count = map.count; + if ( map.entries.value == 0 || map.count <= 0 ) { cursor.ok = map.count == 0; cursor.count = 0; return cursor; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + if ( head->live != map.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + const Entry ** order = (const Entry **) arena.allocator.alloc( arena.allocator.context, (int64_t) map.count * (int64_t) sizeof( const Entry * ) ); + if ( order == NULL ) { return cursor; } + int32_t at = 0; + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 && at < map.count ) + { + const TableMapSegment * segment = (const TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used && at < map.count; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + order[at++] = segment->entries + i; + } + segment_ref = segment->next; + } + if ( at != map.count ) + { + arena.allocator.free( arena.allocator.context, order ); + return cursor; + } + TableMapSort( order, map.count ); + cursor.order = order; + cursor.ok = true; + return cursor; +} + +template +inline TableMapCursor TableMapOrder( const TableArenaCtx & ctx, const TableMap & map ) +{ + return TableMapOrder( *ctx.arena, map ); +} + +template inline void TableMapRelease( TableMapCursor & cursor ) +{ + if ( cursor.order != NULL ) { cursor.allocator.free( cursor.allocator.context, (void *) cursor.order ); } + cursor.order = NULL; +} + +// ---- the builder's five (§2.8) ---- +// +// Insert APPENDS after one LINEAR SCAN of the live entries for the key it may +// replace, Find is that same scan, and Erase is the scan and one bit. The +// builder builds NO INDEX, and that is a rule: the sort happens once, at Lock, +// Save or Cook, and every lookup that matters runs over the sorted region. + +// the head, allocated when the first entry is inserted +template +inline TableMapHead * TableMapReach( TableWorker & worker, TableMap & map ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( map.entries.value != 0 ) { return (TableMapHead *) TableArenaAt( *worker.arena, (uint32_t) map.entries.value ); } + uint32_t at = 0; + TableMapHead * head = (TableMapHead *) worker.AllocRaw( (int64_t) sizeof( TableMapHead ), (int64_t) alignof( TableMapHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + map.entries.value = (int64_t) at; + return head; +} + +// one entry's storage, appended: the current segment when it has room, a new +// one carved from one call to the pair when it does not +template +inline Entry * TableMapAppend( TableWorker & worker, TableMapHead * head, TableMap & map ) +{ + TableMapSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableMapSegmentEntries ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableMapSegment *) worker.AllocRaw( (int64_t) sizeof( TableMapSegment ), (int64_t) alignof( TableMapSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableMapSegment * previous = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Entry * entry = segment->entries + segment->used; + segment->used++; + head->live++; + map.count++; + return entry; +} + +// the LINEAR SCAN: the live entries in insertion order, O( n ) key compares +template +inline Entry * TableMapScan( const TableArena & arena, const TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) == 0 ) { return segment->entries + i; } + } + segment_ref = segment->next; + } + return NULL; +} + +// ERASE marks the entry DEAD, one bit in the segment's slot and not in the +// entry table, and decrements the live count. Its storage is reclaimed at +// RESET and never reused mid-build, because reusing a slot would make "an +// entry's address is stable" false for exactly one case. +template +inline bool TableMapErase( TableArena & arena, TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return false; } + TableMapHead * head = (TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) != 0 ) { continue; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + map.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INSERTION order, live entries only (§2.8) ---- +template struct TableMapEach +{ + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableMapSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableMapSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + auto operator*() const { return TableEntryEach( segment->entries + index ); } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableMapEach TableMapEachOf( const TableArena & arena, const TableMap & map ) +{ + TableMapEach each = { &arena, TableRef() }; + if ( map.entries.value != 0 ) + { + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + each.first = head->first; + } + return each; +} + +// ---- the LOAD side: where a decoded entry lands (§2.8) ---- +// +// THE READER TRUSTS NOTHING and spends one compare per entry. Every load path +// applies the same rules and produces one report (§4), so the region load of +// §6.5 and LoadBuilder never disagree about a wire. These two shapes are what +// makes that true with one generated decoder: a REGION carves the entry array +// out of the holder node's own extent, and the TOOL's path appends into the +// builder's arena, and the decoder above them cannot tell which it has. + +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. + +// TableMapFill is one map field being decoded: where the next entry lands, and +// the entry that last LANDED, which is what the ascending check compares +// against. +template struct TableMapFill +{ + TableMap * map = NULL; + Entry * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; +}; + +template +inline TableMapFill TableMapFillBegin( const TableNodeMap & nodes, TableMap & map, uint32_t n ) +{ + TableMapFill fill; + fill.map = ↦ + map.entries.value = 0; + map.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Entry ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Entry ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Entry *) base; + fill.capacity = (int32_t) n; + map.entries.value = (int64_t) ( base - (const uint8_t *) &map.entries ); + fill.ok = true; + return fill; +} + +// the entry that last LANDED — NULL before the first +template inline Entry * TableMapFillLast( TableMapFill & fill ) +{ + if ( fill.map->count <= 0 ) { return NULL; } + if ( fill.array != NULL ) { return fill.array + ( fill.map->count - 1 ); } + return TableMapLive( *fill.worker->arena, *fill.map, fill.map->count - 1 ); +} + +// the next slot, at the entry type's declared defaults +template inline Entry * TableMapFillNext( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + if ( fill.map->count >= fill.capacity ) { return NULL; } + Entry * entry = fill.array + fill.map->count; + TableReset( *entry ); + fill.map->count++; + return entry; + } + TableMapHead * head = TableMapReach( *fill.worker, *fill.map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( *fill.worker, head, *fill.map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// A MAP WITH HALF ITS KEYS IS NOT A MAP (§2.8): at the first entry whose key +// kind disagrees with the reader's declaration the map resets to EMPTY, one +// kind_mismatch is counted for the map, and its remaining bytes are skipped. +template inline void TableMapFillReset( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + fill.map->entries.value = 0; + fill.map->count = 0; + return; + } + if ( fill.map->entries.value != 0 ) + { + TableMapHead * head = (TableMapHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.map->entries.value ); + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + } + fill.map->count = 0; +} + +// an EMPTY map's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableMapFillEnd( TableMapFill & fill ) +{ + if ( fill.array != NULL && fill.map->count == 0 ) { fill.map->entries.value = 0; } +} + +// the k-th LIVE entry of a builder map, in insertion order — what the tool +// path's ascending check compares against +template +inline Entry * TableMapLive( const TableArena & arena, const TableMap & map, int32_t index ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + int32_t at = 0; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( at == index ) { return segment->entries + i; } + at++; + } + segment_ref = segment->next; + } + return NULL; +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.8, §6.5) ---- +// +// LoadMeasure's term for a map is N x sizeof( Entry ) rounded to +// alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this +// reads no field: it walks the map's own header and, where an entry's value +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. +// A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its +// own L and the body's terminator, and under this form's variable lengths that +// footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a +// map's L can carry, and therefore what a LoadMeasure may be asked for. +static const int64_t kTableMapEntryFloor = 2; + +inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry + at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); + at += (int64_t) n * entry_size; + if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// ---- the TEXT form's placement (docs/SPEC-TABLES.md §2.8, §16) ---- +// +// The text is a plain JSON object keyed by the KEY, and the generic walk fills +// it through the ENTRY'S OWN descriptor — so all it needs from here is one +// entry at one key, handed back at its defaults. It is the builder's Insert +// with the ENTRY returned rather than its value, because the walk writes the +// value through a field row and not through a typed pointer. +template +inline Entry * TableMapPlace( TableWorker & worker, TableMap & map, Key key ) +{ + if ( worker.arena == NULL ) { return NULL; } + Entry * found = TableMapScan( *worker.arena, map, key ); + if ( found != NULL ) + { + TableResetMapValue( *found ); // a repeated key is LAST-WINS, whole + return found; + } + TableMapHead * head = TableMapReach( worker, map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( worker, head, map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// ---- the OPTIONAL RUNTIME INDEX (§2.8) ---- +// +// Open addressing with LINEAR PROBING over the sorted array, built AT LOAD for +// a map large enough that log n compares over a cold array cost more than one +// hash and a probe. IT IS NEVER STORED: the caller measures it, owns its +// storage, builds it in one pass and releases it whenever. +// +// ITS HASH AND ITS LOAD FACTOR ARE NOT A CROSS-PORT CONTRACT, and that is a +// rule. What a port is held to is the CONTRACT of the lookup: the same value +// the sorted array's Find returns for the same key, and no allocation past the +// storage the caller handed in. +struct TableMapIndex +{ + int32_t * slots = NULL; // entry indices, +1; 0 is an empty slot + int32_t capacity = 0; + bool good = false; +}; + +// this runtime's own, and no port reproduces it: fnv1a64 over the key's bytes +inline uint64_t TableMapHash( const void * bytes, int32_t length ) +{ + uint64_t hash = 0xCBF29CE484222325ull; + const uint8_t * at = (const uint8_t *) bytes; + for ( int32_t i = 0; i < length; i++ ) { hash ^= (uint64_t) at[i]; hash *= 0x100000001B3ull; } + return hash; +} +inline uint64_t TableMapHash( uint64_t key ) { return TableMapHash( (const void *) &key, (int32_t) sizeof( key ) ); } + +// this runtime's own load factor, and no port reproduces it either: the next +// power of two at or above twice the count, so a probe run stays short +inline int32_t TableMapIndexSlots( int32_t count ) +{ + int32_t slots = 8; + while ( slots < count * 2 ) { slots *= 2; } + return slots; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_MAP + +#ifndef LISTDEMO_SCHEMA_TABLE_LIST +#define LISTDEMO_SCHEMA_TABLE_LIST + +namespace listdemo { + +// ---- an UNBOUNDED ARRAY: a counted array whose count the data decides (§2.9) ---- +// +// On the wire, in a region and in a cook a list is the kind 14 body a [..N]T +// writes, its elements by-value records inside the holder's node extent. What +// this adds is the slot, a builder that appends into segments that never +// move, and a const surface that indexes and iterates in place. There is no +// sort, no key and no lookup: the order is INSERTION order, and it is +// identity the way position is identity in a fixed array. + +// elements carved from ONE call to the allocator pair. A new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableListSegmentElements = 32; + +// THE ELEMENT STORAGE: T itself, and a TableRef slot for a []*T, whose +// elements are references exactly as a pointer field's slot is (§2.1) +template struct TableListStorage { typedef T Element; }; +template struct TableListStorage { typedef TableRef Element; }; + +// WHAT THE CONST FORM ANSWERS: the element by reference, and on a []*T the +// RESOLVED pointer, one add on the self-relative delta, NULL for a null slot, +// exactly as At answers it (§6.2, §6.3) +template struct TableListConst +{ + typedef const T & Result; + static Result At( const T * element ) { return *element; } +}; +template struct TableListConst +{ + typedef const T * Result; + static Result At( const TableRef * element ) + { + return element->value != 0 ? (const T *) ( (const uint8_t *) element + element->value ) : NULL; + } +}; + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.9, §7.2) ---- +// +// An int64 self-relative reference to the element array and an int32 count, +// then padding to eight. The reference is a TableRef like a pointer's: in the +// arena it names the builder's HEAD, in a region it is the delta from the slot +// to the first element, and 0 is the empty list in both. It is the map's slot +// exactly, because it is the same two facts. +template struct TableList +{ + typedef typename TableListStorage::Element Element; + + TableRef elements; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Element * Elements() const + { + return elements.value != 0 ? (const Element *) ( (const uint8_t *) &elements + elements.value ) : NULL; + } + int32_t size() const { return count; } + + // INDEXING IS BOUNDS-CHECKED IN EVERY BUILD (§2.4, §2.9): the extent is a + // number that CAME FROM A FILE, so an index past it is not a mistake a + // release build gets to make cheaply. There is no undefined-behavior path + // here in any configuration. The assert carries the message where a + // debugger can read it and NDEBUG removes that. The fatal is what stands + // after it. Both go through the hooks: define schema_assert and + // schema_fatal and this refusal lands in your own handler. + void RefuseIndex( int32_t index ) const + { + if ( (uint32_t) index >= (uint32_t) count ) + { + schema_assert( false && "an unbounded array is indexed inside its count, which came from a file" ); + schema_fatal(); + } + } + typename TableListConst::Result operator[]( int32_t index ) const + { + RefuseIndex( index ); + return TableListConst::At( Elements() + index ); + } + + // ---- iteration: INDEX order, the element and no key ---- + // + // It carries no iterator_traits, for the reason TableKeyed's does not + // (§13.9). + struct ConstIterator + { + const Element * at; + typename TableListConst::Result operator*() const { return TableListConst::At( at ); } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Elements() }; } + ConstIterator end() const { return ConstIterator{ Elements() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.9, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first element is added. Each +// segment is a fixed number of elements carved from one call to the allocator +// pair. An element's address is stable for the arena's life, so a T * handed +// back by Add stays valid while other elements arrive. +struct TableListHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an Add appends into + int32_t live; + int32_t dead; +}; + +template struct TableListSegment +{ + TableRef next; + int32_t used; // elements carved from this segment + int32_t padding; + uint32_t dead[ ( kTableListSegmentElements + 31 ) / 32 ]; // Erase marks one bit, never the element + Element elements[ kTableListSegmentElements ]; +}; + +inline bool TableListSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// the head, allocated when the first element is added +template +inline TableListHead * TableListReach( TableWorker & worker, TableList & list ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( list.elements.value != 0 ) { return (TableListHead *) TableArenaAt( *worker.arena, (uint32_t) list.elements.value ); } + uint32_t at = 0; + TableListHead * head = (TableListHead *) worker.AllocRaw( (int64_t) sizeof( TableListHead ), (int64_t) alignof( TableListHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + list.elements.value = (int64_t) at; + return head; +} + +// one element's storage, appended: the current segment when it has room, a +// new one carved from one call to the pair when it does not. NULL means NOT +// ADDED: an arena that cannot carve another segment, or a count at the int32 +// cap (§2.2, §2.9). +template +inline typename TableList::Element * TableListAppend( TableWorker & worker, TableListHead * head, TableList & list ) +{ + typedef typename TableList::Element Element; + if ( list.count >= INT32_MAX ) { return NULL; } // the int32 storage cap + TableListSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableListSegmentElements ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableListSegment *) worker.AllocRaw( (int64_t) sizeof( TableListSegment ), (int64_t) alignof( TableListSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableListSegment * previous = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Element * element = segment->elements + segment->used; + segment->used++; + head->live++; + list.count++; + return element; +} + +// ADD, whole: the head, the append, and the element at its declared defaults +// (§2.9). The text form's placement is this same call, because a list has no +// key to place under (§16). +template +inline typename TableList::Element * TableListPlace( TableWorker & worker, TableList & list ) +{ + typedef typename TableList::Element Element; + TableListHead * head = TableListReach( worker, list ); + if ( head == NULL ) { return NULL; } + Element * element = TableListAppend( worker, head, list ); + if ( element == NULL ) { return NULL; } + new ( element ) Element(); // value-init: the declared defaults, and null for a slot + return element; +} + +// ERASE, ADDRESSED BY THE POINTER (§2.9): the element Add handed back is the +// handle, because a list has no key and the address is the one thing the +// builder promises never moves (§6.4). It marks the element DEAD, one bit in +// the segment's slot and not in the element storage, and decrements the live +// count. False when the pointer is not this list's. Its storage is reclaimed +// at RESET and never reused mid-build, the map's rule for the map's reason. +template +inline bool TableListErase( TableArena & arena, TableList & list, const typename TableList::Element * element ) +{ + typedef typename TableList::Element Element; + if ( list.elements.value == 0 || element == NULL ) { return false; } + TableListHead * head = (TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableListSegment * segment = (TableListSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + if ( element >= segment->elements && element < segment->elements + segment->used ) + { + const int32_t i = (int32_t) ( element - segment->elements ); + if ( TableListSegmentDead( segment->dead, i ) ) { return false; } // already erased + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + list.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INDEX order, live elements only (§2.9) ---- +template struct TableListEach +{ + typedef typename TableList::Element Element; + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableListSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableListSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + Element * operator*() const { return segment->elements + index; } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableListEach TableListEachOf( const TableArena & arena, const TableList & list ) +{ + TableListEach each = { &arena, TableRef() }; + if ( list.elements.value != 0 ) + { + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + each.first = head->first; + } + return each; +} + +// ---- the INDEX-ORDER CURSOR the four writing walks read (§2.9) ---- +// +// Measure, Save, Lock and Cook each visit a list's live elements in the order +// they were added, and they allocate nothing to do it: a region's cursor is +// the array in place, and the builder's walks the segment chain. Indexing the +// builder's form is SEQUENTIAL by construction, every walk steps i, i + 1, +// i + 2, so the cursor remembers where the last access landed and moves one +// live slot per step. An access behind the memo restarts from the first +// segment, which no walk here does. +template struct TableListCursor +{ + const Element * elements = NULL; // the region's form: the array in place + const TableArena * arena = NULL; // the builder's form: the segments + TableRef first; + int32_t count = 0; + bool ok = false; + // the memo: the segment and slot the last access landed on, and the live + // index that slot holds + mutable const TableListSegment * segment = NULL; + mutable int32_t within = -1; + mutable int32_t logical = -1; + + const Element * At( int32_t index ) const + { + if ( elements != NULL ) { return elements + index; } + if ( segment == NULL || index < logical ) + { + segment = first.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL; + within = -1; + logical = -1; + } + while ( logical < index ) + { + for ( ;; ) + { + within++; + while ( segment != NULL && within >= segment->used ) + { + segment = segment->next.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + within = 0; + } + if ( segment == NULL ) { return NULL; } // the slot and the head disagree + if ( !TableListSegmentDead( segment->dead, within ) ) { break; } + } + logical++; + } + return segment->elements + within; + } + const Element & operator[]( int32_t index ) const { return *At( index ); } +}; + +// the REGION form: the array is the cursor +template +inline TableListCursor::Element> TableListElements( const TableRegionCtx &, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.elements = list.Elements(); + cursor.count = list.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: the live elements out of the segment chain, in the +// order they were added. A dead element costs nothing on any wire (§2.9). +template +inline TableListCursor::Element> TableListElements( const TableArena & arena, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.arena = &arena; + cursor.count = list.count; + if ( list.elements.value == 0 || list.count <= 0 ) { cursor.ok = list.count == 0; cursor.count = 0; return cursor; } + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + if ( head->live != list.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + cursor.first = head->first; + cursor.ok = true; + return cursor; +} + +template +inline TableListCursor::Element> TableListElements( const TableArenaCtx & ctx, const TableList & list ) +{ + return TableListElements( *ctx.arena, list ); +} + +// ---- the LOAD side: where a decoded element lands (§2.9) ---- +// +// The same two shapes the map's fill takes, because the decoder above them +// cannot tell which it has: a REGION carves the element array out of the +// holder node's own extent, PRE-ORDER, and the TOOL's path appends into the +// builder's arena. +template struct TableListFill +{ + typedef typename TableList::Element Element; + TableList * list = NULL; + Element * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; + bool refused = false; // a count above the int32 cap on the tool's path: LoadBuilder answers NULL +}; + +template +inline TableListFill TableListFillBegin( const TableNodeMap & nodes, TableList & list, uint64_t n ) +{ + typedef typename TableList::Element Element; + TableListFill fill; + fill.list = &list; + list.elements.value = 0; + list.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( n > (uint64_t) INT32_MAX ) + { + // A COUNT ABOVE THE int32 STORAGE CAP (§2.2, §2.9): into a region it was + // refused by LoadMeasure before this ran, and into a builder it is the + // refusal LoadBuilder answers NULL for, moving no counter + fill.refused = nodes.carve->worker != NULL; + return fill; + } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Element ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Element ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Element *) base; + fill.capacity = (int32_t) n; + list.elements.value = (int64_t) ( base - (const uint8_t *) &list.elements ); + fill.ok = true; + return fill; +} + +// the next slot, at the element's declared defaults. NULL when the arena +// could not carve, which the decoder reports as framing damage +template inline typename TableList::Element * TableListFillNext( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count >= fill.capacity ) { return NULL; } + Element * element = fill.array + fill.list->count; + new ( element ) Element(); + fill.list->count++; + return element; + } + return TableListPlace( *fill.worker, *fill.list ); +} + +// A SLOT WHOSE ELEMENT NEVER LANDED is given back (§2.9, §4): the array keeps +// what it decoded, and an element whose own framing gave out before one byte +// of it decoded was not decoded. The region's form uncounts it, and the builder's +// marks it dead, which is what the storage rule allows mid-build. +template inline void TableListFillDrop( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count > 0 ) { fill.list->count--; } + return; + } + if ( fill.list->elements.value == 0 ) { return; } + TableListHead * head = (TableListHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.list->elements.value ); + if ( head->last.value == 0 ) { return; } + TableListSegment * segment = (TableListSegment *) TableArenaAt( *fill.worker->arena, (uint32_t) head->last.value ); + if ( segment->used <= 0 ) { return; } + const int32_t i = segment->used - 1; + if ( TableListSegmentDead( segment->dead, i ) ) { return; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + fill.list->count--; +} + +// an EMPTY list's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableListFillEnd( TableListFill & fill ) +{ + if ( fill.array != NULL && fill.list->count == 0 ) { fill.list->elements.value = 0; } +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.9, §6.5) ---- +// +// N x sizeof( T ) rounded to alignof( T ), AT EVERY DEPTH. N is framing and +// not a value, so this reads no field: it walks the list's own header and, +// where a table element holds a list or a map of its own, the elements' +// headers under it. Every -1 carries its REASON (§6.5): the int32 cap first, +// because a count past it cannot fit any body, and then the body's own L. +inline bool TableListWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t elem_size, int64_t elem_align, uint8_t elem_kind, int64_t elem_floor, + TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != elem_kind ) { return true; } // another element kind: §4's ordinary kind mismatch, the field reads empty + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; } // an N the list's L cannot carry + at = ( at + elem_align - 1 ) & ~( elem_align - 1 ); + at += (int64_t) n * elem_size; + if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_LIST + +#ifndef LISTDEMO_SCHEMA_BUILD_VERSION +#define LISTDEMO_SCHEMA_BUILD_VERSION + +namespace listdemo { + +// THE BUILD VERSION (docs/SPEC-TABLES.md §20): one digest over every fact the bytes +// this build produces depend on — the type wire's protocol id, every record's +// layout as the compiler's own C ABI model computes it, and the facts that +// decide what a load PUTS in those slots. It is the number a cook's header +// carries and the number Open compares, and the number a block's prologue +// carries and BlockOpen compares: a build version answers "which build?" and +// not "which form?", and what separates the two forms is their MAGIC. +// +// There are TWO ids in the design and they are not interchangeable: the +// PROTOCOL ID is the type wire's and nothing else, and the BUILD VERSION is +// what everything cooked or blocked is keyed by. A table edit moves this and +// never the protocol id; a type edit moves both. +static const uint64_t BuildVersion = 0x8d7c0edaca4571c7ull; + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_BUILD_VERSION + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK +#define LISTDEMO_SCHEMA_TABLE_COOK + +namespace listdemo { + +// ---- the cooked form (docs/SPEC-TABLES.md §7) ---- +// +// A cooked file is a HEADER, a DATA part and an ATTRIBUTION part, in that +// order. Every word of the header is a u64 written in the byte order the cook +// was produced in, and the header is 64 bytes: +// +// 0 magic 0x4b4f4f434d484353, read BYTEWISE before anything else +// 8 build_version the unit's id (docs/SPEC-TABLES.md §20) +// 16 byte_order 1 little, 2 big — the order that WROTE the file +// 24 data_length the region's bytes, rounded up to alignment +// 32 attribution_length the directory's bytes, or 0 +// 40 alignment the region's alignment, never below eight +// 48 reserved zero +// 56 reserved zero +// +// The DATA part is Lock's region written verbatim (§7.2) — the root at its +// base — and it is what a runtime points at. The ATTRIBUTION part is the node +// directory (§6.3), and NOTHING THAT READS THE STRUCTURE TOUCHES IT: it is +// written beside the data for schema cook-check, so a build that ships no +// tooling need not carry it at all. +static const int64_t kTableCookHeaderBytes = 64; + +// THE MAGIC'S VALUE, and a consumer written from the page needs the constant +// rather than a description of one. It is "SCHMCOOK" read as ASCII in the byte +// order a little-endian store produces — the same shape the block form's +// SCHMABLK takes, so a hex dump of a little-endian cook is legible and the two +// accelerators sit in one vocabulary. +// +// IT IS STORED IN THE PRODUCER'S ORDER, which is what makes it the byte-order +// check as well as the form check: a consumer reads back this build's +// constant, or that constant byte-reversed — which identifies a cook of the +// OTHER order — or something that is not a cook. All three answers but the +// first refuse, and a cook and a BLOCK are separated here too, because a +// form's identity belongs in its magic rather than in a second digest. +static const uint64_t TableCookMagic = 0x4b4f4f434d484353ull; + +// THIS BUILD's byte order, as the header's own word carries it. The magic is +// what REFUSES a foreign order; this word is what RECORDS which order wrote +// the file, so a refusal names the order rather than inferring it and a tool +// dumping a cook reads the fact. A file whose magic matched and whose order +// word did not is corrupt, and there is no reading that recovers it. +// +// The BUILD VERSION cannot do either job: §20.1 digests byteorder as a +// GENERATION input, little for every target schema generates for today, so +// two builds of one schema for two orders emit the same id. +#if defined( __BYTE_ORDER__ ) && defined( __ORDER_BIG_ENDIAN__ ) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__ +static const uint64_t TableCookByteOrder = 2; // big +#else +static const uint64_t TableCookByteOrder = 1; // little +#endif + +// The greatest region alignment a cooked file may name. The DATA part begins +// at align_up( 64, alignment ), which is 64 for every unit this language can +// declare — the largest alignment it has is sixteen — so a word past this cap +// describes a file no build of this schema wrote (docs/SPEC-TABLES.md §7.1). +static const uint64_t TableCookMaxAlign = 64; + +// The header read, BYTEWISE. memcpy is the portable spelling of "these eight +// bytes, in this machine's order"; every compiler this repo builds under folds +// it to one load, and it is the only read in the whole of Open that is not a +// comparison. +inline uint64_t table_cook_read64( const uint8_t * p ) +{ + uint64_t v; + memcpy( &v, p, sizeof( v ) ); + return v; +} + +// TableCookOpen: THE WHOLE CHECK, in one place, because §7 states the +// enumeration once and every generated Open is that one enumeration plus +// its own root's two layout facts. +// +// THE CHECK, in order: the magic read bytewise, the byte order it establishes, +// the build version against this build's own, both RESERVED words zero, the +// region alignment the header names, the two part lengths against the length +// the caller passed — a truncated file and a file with trailing bytes are the +// same refusal — the root's own storage inside the data part, and the +// alignment of the base. +// +// AND THAT IS ALL OF IT. On a match the bytes ARE what this build wrote, in +// this build's layout and this build's byte order, so there is nothing to +// validate and nothing to fix up: the caller gets the root. Nothing per node +// happens here, which is what makes open O(1) in the file's size; a walk of +// any shape would forfeit that, and validating an untrusted file is schema +// cook-check's job and a person's decision (§7.4). +// +// EVERY NUMBER BELOW COMES OUT OF THE FILE, so the arithmetic is unsigned and +// each term is BOUNDED BEFORE IT IS ADDED: a forged length near 2^64 must +// refuse, and an addition that wrapped would be the defect the comparison +// after it was supposed to catch. Nothing past length is read on any path, +// including every refusing one. +inline const uint8_t * TableCookOpen( const void * bytes, uint64_t length, uint64_t root_size, uint64_t root_align ) +{ + if ( bytes == NULL ) { return NULL; } + if ( length < (uint64_t) kTableCookHeaderBytes ) { return NULL; } + const uint8_t * raw = (const uint8_t *) bytes; + // the MAGIC, bytewise and first: it is what establishes the byte order + // every other header word is read in, so nothing else may be read before + // it. A byte-reversed constant is a cook of the other order and refuses + // here, which is why the order never reaches a fix-up pass. + if ( table_cook_read64( raw ) != TableCookMagic ) { return NULL; } + if ( table_cook_read64( raw + 16 ) != TableCookByteOrder ) { return NULL; } + if ( table_cook_read64( raw + 8 ) != BuildVersion ) { return NULL; } + // the RESERVED words: a non-zero one means a writer used a form this build + // does not understand, and Open refuses rather than ignoring it. + if ( table_cook_read64( raw + 48 ) != 0 ) { return NULL; } + if ( table_cook_read64( raw + 56 ) != 0 ) { return NULL; } + const uint64_t data_length = table_cook_read64( raw + 24 ); + const uint64_t attribution_length = table_cook_read64( raw + 32 ); + const uint64_t alignment = table_cook_read64( raw + 40 ); + // THE ALIGNMENT WORD IS DATA, and it is the one header field the rest of + // the check does arithmetic WITH rather than only comparison against. A + // region's alignment is a power of two, never below eight (the floor that + // puts the attribution part on an eight-byte boundary without a second + // padding rule) and never past the cap above; a word that is none of those + // rounds nothing and aligns nothing, so it is refused before it is used. + if ( alignment < 8 || alignment > TableCookMaxAlign ) { return NULL; } + if ( ( alignment & ( alignment - 1 ) ) != 0 ) { return NULL; } + // and it must be an alignment THE ROOT CAN SIT AT, since the root is at + // the region's base: both are powers of two, so "at least the root's" + // is one division. + if ( ( alignment % root_align ) != 0 ) { return NULL; } + // The DATA part begins at align_up( 64, alignment ). It is DERIVED and not + // a header field, because a fact a reader computes is a fact two writers + // cannot disagree about. + const uint64_t data_offset = ( (uint64_t) kTableCookHeaderBytes + alignment - 1 ) & ~( alignment - 1 ); + if ( length < data_offset ) { return NULL; } + // the two part lengths against the length the caller passed. The whole + // file is data_offset + data_length + attribution_length, and a length + // that is not EXACTLY that refuses — truncation and trailing bytes are one + // refusal, and both terms are subtracted rather than added so no sum can + // carry. + if ( data_length > length - data_offset ) { return NULL; } + if ( attribution_length != length - data_offset - data_length ) { return NULL; } + // the ROOT sits at the region's base, so the region has to hold it: a + // shorter data part describes a root partly outside the file, which is the + // one way a match-and-point reader could hand back storage it never + // received. + if ( data_length < root_size ) { return NULL; } + const uint8_t * base = raw + data_offset; + // the alignment of the BASE. The header pads the data part to the region's + // alignment, so a base an allocator or mmap gave you is already aligned — + // mmap gives page alignment for free — and a base that is not is a caller's + // buffer this form cannot be read out of. + if ( ( (uintptr_t) base % (uintptr_t) alignment ) != 0 ) { return NULL; } + return base; +} + +// ---- the cooked form, the WRITE side (docs/SPEC-TABLES.md §7.6) ---- +// +// THE BYTE ORDER IS THE TARGET'S, NOT THE HOST'S. A cook is produced in the +// byte order of the build that will read it (§7), so the fixing happens here — +// offline, once, on the writing side — and never at Open. Passing +// TableByteOrder::Big on a little-endian machine produces a big-endian build's +// file, and nothing about the writing host reaches the bytes. +enum class TableByteOrder +{ + Little = 1, // the header's byte_order word, and the order every scalar is written in + Big = 2, +}; + +// One store, width as an argument. Every call site passes a literal width, so +// the loop folds to a store (and a byte swap on the foreign order); a name per +// width would claim four §11 names to save nothing. +inline void table_cook_put( uint8_t * at, uint64_t value, int32_t width, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * i ) ); } + } + else + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * ( width - 1 - i ) ) ); } + } +} + +// A 128-bit store as two lanes: sixteen bytes, the low lane first in the +// little order and the high lane first — each lane big-endian — in the big +// order, exactly as a u64 is one lane of eight (docs/SPEC-TABLES.md §7.2). +inline void table_cook_put128( uint8_t * at, uint64_t lo, uint64_t hi, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) { table_cook_put( at, lo, 8, order ); table_cook_put( at + 8, hi, 8, order ); } + else { table_cook_put( at, hi, 8, order ); table_cook_put( at + 8, lo, 8, order ); } +} + +// A buffer piece: the USED bytes and nothing else. The tail is already zero — +// the whole extent was zeroed before any field was written — so this copies the +// used prefix and leaves the rest, which is what makes a string's unused tail a +// consequence of one memset rather than a rule per buffer. A used length past +// the buffer, or below zero, is a value no reader could have produced and it is +// clamped rather than trusted: this writes inside the caller's buffer on every +// input. +inline void table_cook_bytes( uint8_t * at, const void * source, int64_t used, int64_t capacity ) +{ + if ( used <= 0 ) { return; } + const int64_t n = used < capacity ? used : capacity; + memcpy( at, source, (size_t) n ); +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE +#define LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// ---- the cooked form's WRITE side for a POINTERED root (docs/SPEC-TABLES.md §7.6) ---- +// +// A pointered root's cook is the region of §7.2: every node the numbering +// reached (§3.1), once, at its own type's alignment, in index order, the root +// at offset zero. This is that region while it is being laid out and written — +// the tool's own Layout and Write, in one struct. +// +// The OFFSETS are one per node, the root's zero at position 0 and node index k +// at position k - 1, which is the directory's own order (§6.3); they are the +// one allocation the write makes beyond the numbering, and they go through the +// same pair. A measure needs no offsets and leaves the pointer NULL. +struct TableCookRegion +{ + const TableNumbering * numbering = NULL; // node -> index, from the walk that placed it + int64_t * offsets = NULL; // index - 1 -> the node's region offset; NULL while measuring + int64_t count = 0; // nodes, the root included + int64_t bytes = 0; // the data part's length, rounded to align + int64_t align = 0; // the region's alignment: the nodes' greatest, never below eight + uint8_t * base = NULL; // where the data part is being written; NULL while measuring +}; + +// A reference slot: the SELF-RELATIVE delta from the slot's own address to the +// node's start (§6.3), and zero for null. The node is found by the address the +// numbering keyed it under, which is the same address the walk resolved through +// the same context — so a reference the numbering does not carry is a slot the +// walk never reached (a counted array's slot past its count, an absent +// optional's value) holding a node the region will not hold, and it is refused +// rather than written as a delta to nowhere. +inline bool table_cook_ref( const TableCookRegion & region, uint8_t * at, const void * pointee, TableByteOrder order ) +{ + if ( pointee == NULL ) { table_cook_put( at, 0, 8, order ); return true; } + uint64_t index = 0; + if ( !TableNumberingIndex( *region.numbering, pointee, index ) ) { return false; } + if ( index == 0 || index > (uint64_t) region.count ) { return false; } + const int64_t delta = region.offsets[index - 1] - (int64_t) ( at - region.base ); + table_cook_put( at, (uint64_t) delta, 8, order ); + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// table Unit — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Unit { + int32_t v = 0; +}; + +// table Bounded — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Bounded { + Unit items[8]; // used count beside it; count in [0, 8] + int32_t items_count = 0; + int32_t tag = 0; +}; + +// table Unbounded — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Unbounded { + TableList items; // Unit: the element array, empty until an Add + int32_t tag = 0; +}; + +// ---- prefill: the declared defaults, in place (docs/SPEC-TABLES.md) ---- + +inline void UnitReset( Unit & value ); +inline void BoundedReset( Bounded & value ); +inline void UnboundedReset( Unbounded & value ); + +inline void UnitReset( Unit & value ) +{ + value.v = 0; +} + +inline void BoundedReset( Bounded & value ) +{ + UnitReset( value.items[0] ); + for ( int32_t i = 1; i < 8; i++ ) { value.items[i] = value.items[0]; } + value.items_count = 0; + value.tag = 0; +} + +inline void UnboundedReset( Unbounded & value ) +{ + value.items.elements.value = 0; // Unit: empty + value.items.count = 0; + value.items.padding = 0; + value.tag = 0; +} + +// ---- the arena's reset hook (docs/SPEC-TABLES.md §6) ---- +// +// TableWorker::Alloc is a template and cannot name a member's Reset, so +// the arena reaches it through this overload set by argument-dependent +// lookup. It is how a node born in raw arena storage comes to hold the +// declared defaults without value-initialising the whole aggregate. + +inline void TableReset( Unit & value ) { UnitReset( value ); } +inline void TableReset( Bounded & value ) { BoundedReset( value ); } +inline void TableReset( Unbounded & value ) { UnboundedReset( value ); } + +// ---- pointer targets: allocation and resolution (docs/SPEC-TABLES.md §2) ---- +// +// A reference resolves differently in the two forms, and the CONTEXT says +// which: in the arena it is an offset; in a region it is a self-relative +// delta, so the const deref below is one add and needs no base pointer. + +// ---- codecs: measure/save/load per closure member ---- + +inline int64_t UnitMeasureBody( TableIds & ids, const Unit & value ); +LISTDEMO_TABLE_INLINE bool UnitSaveBody( TableWriter & w, TableIds & ids, const Unit & value ); +LISTDEMO_TABLE_INLINE bool UnitLoadBody( TableReader & r, Unit & value ); +inline int64_t BoundedMeasureBody( TableIds & ids, const Bounded & value ); +LISTDEMO_TABLE_INLINE bool BoundedSaveBody( TableWriter & w, TableIds & ids, const Bounded & value ); +LISTDEMO_TABLE_INLINE bool BoundedLoadBody( TableReader & r, Bounded & value ); +template inline int64_t UnboundedMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Unbounded & value ); +template inline bool UnboundedSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Unbounded & value ); +template inline bool UnboundedSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Unbounded & value ); +inline bool UnboundedLoadBody( TableReader & r, const TableNodeMap & nodes, Unbounded & value ); + +// ---- pointer-graph walkers: number (measure/save), pack (Lock) ---- + +template inline bool UnboundedNumber( const Ctx & ctx, TableNumbering & numbering, const Unbounded & value ); +template inline int64_t UnboundedPackMeasure( const Ctx & ctx, TablePackMap & seen, const Unbounded & value ); +template inline bool UnboundedPack( const Ctx & ctx, TablePackMap & seen, const Unbounded & src, Unbounded & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +// ---- the numbering's bridge to each member's codec (docs/SPEC-TABLES.md §3.1) ---- + +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Unbounded & value ) { return UnboundedMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Unbounded & value ) { return UnboundedSaveBody( ctx, numbering, w, ids, value ); } + +inline int64_t UnitMeasureBody( TableIds & ids, const Unit & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.v != 0 ) { bytes += TableLebBytes( ids.ref( 0xaf63eb4c86020609ull, 23 ) ) + 1 + 4; } // v + return bytes; +} + +inline int64_t UnitMeasure( const Unit & value ) +{ + TableIds ids; + const int64_t body = UnitMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool UnitSaveBody( TableWriter & w, TableIds & ids, const Unit & value ) +{ + if ( value.v != 0 ) + { + w.putleb( ids.ref( 0xaf63eb4c86020609ull, 23 ) ); w.put8( 4 ); // v + w.put32( uint32_t( value.v ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t UnitSave( const Unit & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !UnitSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == UnitMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool UnitLoadBody( TableReader & r, Unit & value ) +{ + UnitReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xaf63eb4c86020609ull: // v + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.v = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict UnitLoadVerdict( Unit & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + UnitReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + UnitReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !UnitLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool UnitLoad( Unit & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return UnitLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t UnitMeasureMessage( const Unit & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = UnitMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t UnitSaveMessage( const Unit & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !UnitSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == UnitMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool UnitLoadMessage( Unit & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + UnitReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return UnitLoadBody( r, value ); +} + +inline int64_t BoundedMeasureBody( TableIds & ids, const Bounded & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.items_count < 0 || value.items_count > 8 ) { return -1; } // storage invariant + if ( value.items_count > 0 ) + { + const uint64_t ref_items = ids.ref( 0x3e7884bf4f412c6full, 6 ); + int64_t body_items = 0; + body_items += 1 + TableLebBytes( (uint64_t) ( value.items_count ) ); // the element kind byte and the count + for ( int32_t elem_i = 0; elem_i < value.items_count; elem_i++ ) + { + const int64_t elem_bytes = UnitMeasureBody( ids, value.items[elem_i] ); + if ( elem_bytes < 0 ) { return -1; } + body_items += TableLebBytes( (uint64_t) ( elem_bytes ) ) + ( elem_bytes ); + } + bytes += TableLebBytes( ref_items ) + 1 + TableLebBytes( (uint64_t) ( body_items ) ) + ( body_items ); // items + } + if ( value.tag != 0 ) { bytes += TableLebBytes( ids.ref( 0x56d7ab194448a4f3ull, 7 ) ) + 1 + 4; } // tag + return bytes; +} + +inline int64_t BoundedMeasure( const Bounded & value ) +{ + TableIds ids; + const int64_t body = BoundedMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool BoundedSaveBody( TableWriter & w, TableIds & ids, const Bounded & value ) +{ + if ( value.items_count < 0 || value.items_count > 8 ) { return false; } // storage invariant + if ( value.items_count > 0 ) + { + const uint64_t ref_items = ids.ref( 0x3e7884bf4f412c6full, 6 ); + int64_t body_items = 0; + body_items += 1 + TableLebBytes( (uint64_t) ( value.items_count ) ); // the element kind byte and the count + for ( int32_t elem_i = 0; elem_i < value.items_count; elem_i++ ) + { + const int64_t elem_bytes = UnitMeasureBody( ids, value.items[elem_i] ); + if ( elem_bytes < 0 ) { return false; } + body_items += TableLebBytes( (uint64_t) ( elem_bytes ) ) + ( elem_bytes ); + } + w.putleb( ref_items ); w.put8( 14 ); w.putleb( (uint64_t) body_items ); // items + w.put8( 13 ); w.putleb( (uint64_t) ( value.items_count ) ); + for ( int32_t elem_i = 0; elem_i < value.items_count; elem_i++ ) + { + { + const int64_t elem_len = UnitMeasureBody( ids, value.items[elem_i] ); + if ( elem_len < 0 ) return false; + w.putleb( (uint64_t) elem_len ); + if ( !UnitSaveBody( w, ids, value.items[elem_i] ) ) return false; + } + } + } + if ( value.tag != 0 ) + { + w.putleb( ids.ref( 0x56d7ab194448a4f3ull, 7 ) ); w.put8( 4 ); // tag + w.put32( uint32_t( value.tag ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t BoundedSave( const Bounded & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !BoundedSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == BoundedMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool BoundedLoadBody( TableReader & r, Bounded & value ) +{ + BoundedReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x3e7884bf4f412c6full: // items + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER — the element kind byte and the + // count, so fewer than two bytes — is INERT (§4): the field keeps the + // value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + // A DAMAGED COUNT stops the elements and nothing else: the field + // RODE, so an optional is still PRESENT (§2.3) — only a foreign + // ELEMENT KIND says the payload is not this array's at all. + if ( !counted_ok ) { r.report->malformed = true; } + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + uint64_t keep = count; + if ( keep > 8 ) { keep = 8; r.report->clamped++; } + // elements are BOUNDED by the field body: a count the length + // cannot cover keeps the decoded prefix, flags malformed, and + // the parent continues at the next field — following fields' + // bytes are never fabricated into elements + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + uint64_t decoded = 0; + for ( uint64_t i = 0; i < keep; i++ ) + { + uint64_t elem_len = 0; + if ( !sub.getleb( elem_len ) || !sub.room( elem_len ) ) { r.report->malformed = true; break; } + { + TableReader elem( sub.buffer + sub.offset, (int64_t) elem_len, r.report, r.ids ); + UnitLoadBody( elem, value.items[(int32_t) i] ); + } + sub.offset += (int64_t) elem_len; + decoded = i + 1; + } + value.items_count = (int32_t) decoded; + } + } + r.offset = body_end; // excess elements and slack skip via the length + break; + } + case 0x56d7ab194448a4f3ull: // tag + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.tag = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict BoundedLoadVerdict( Bounded & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + BoundedReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + BoundedReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !BoundedLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool BoundedLoad( Bounded & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return BoundedLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t BoundedMeasureMessage( const Bounded & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = BoundedMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t BoundedSaveMessage( const Bounded & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !BoundedSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == BoundedMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool BoundedLoadMessage( Bounded & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + BoundedReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return BoundedLoadBody( r, value ); +} + +template +inline int64_t UnboundedMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Unbounded & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // items: a kind 14 array of kind 13 elements, INDEX order (§2.9) + TableListCursor cursor_items = TableListElements( ctx, value.items ); + if ( !cursor_items.ok ) { return -1; } // the slot and the head disagree + if ( cursor_items.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_items = ids.ref( 0x3e7884bf4f412c6full, 6 ); + int64_t body_items = 0; + body_items += 1 + TableLebBytes( (uint64_t) ( cursor_items.count ) ); // the element kind byte and the count + for ( int32_t elem_i_items = 0; elem_i_items < cursor_items.count; elem_i_items++ ) + { + const int64_t elem_bytes_items = UnitMeasureBody( ids, cursor_items[elem_i_items] ); + if ( elem_bytes_items < 0 ) { return -1; } + body_items += TableLebBytes( (uint64_t) ( elem_bytes_items ) ) + ( elem_bytes_items ); + } + bytes += TableLebBytes( ref_items ) + 1 + TableLebBytes( (uint64_t) ( body_items ) ) + ( body_items ); + } + } + if ( value.tag != 0 ) { bytes += TableLebBytes( ids.ref( 0x56d7ab194448a4f3ull, 7 ) ) + 1 + 4; } // tag + return bytes; +} + +template +inline bool UnboundedSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Unbounded & value ) +{ + (void) ctx; (void) numbering; + { + TableListCursor cursor_items = TableListElements( ctx, value.items ); // items + if ( !cursor_items.ok ) { return false; } + if ( cursor_items.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_items = ids.ref( 0x3e7884bf4f412c6full, 6 ); + int64_t body_items = 0; + body_items += 1 + TableLebBytes( (uint64_t) ( cursor_items.count ) ); // the element kind byte and the count + for ( int32_t elem_i_items = 0; elem_i_items < cursor_items.count; elem_i_items++ ) + { + const int64_t elem_bytes_items = UnitMeasureBody( ids, cursor_items[elem_i_items] ); + if ( elem_bytes_items < 0 ) { return false; } + body_items += TableLebBytes( (uint64_t) ( elem_bytes_items ) ) + ( elem_bytes_items ); + } + w.putleb( ref_items ); w.put8( 14 ); w.putleb( (uint64_t) body_items ); // items + w.put8( 13 ); w.putleb( (uint64_t) ( cursor_items.count ) ); + for ( int32_t elem_i_items = 0; elem_i_items < cursor_items.count; elem_i_items++ ) + { + { + const int64_t elem_len_items = UnitMeasureBody( ids, cursor_items[elem_i_items] ); + if ( elem_len_items < 0 ) return false; + w.putleb( (uint64_t) elem_len_items ); + if ( !UnitSaveBody( w, ids, cursor_items[elem_i_items] ) ) return false; + } + } + } + } + if ( value.tag != 0 ) + { + w.putleb( ids.ref( 0x56d7ab194448a4f3ull, 7 ) ); w.put8( 4 ); // tag + w.put32( uint32_t( value.tag ) ); + } + return !w.overflow; +} + +template +inline bool UnboundedSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Unbounded & value ) +{ + if ( !UnboundedSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool UnboundedLoadBody( TableReader & r, const TableNodeMap & nodes, Unbounded & value ) +{ + (void) nodes; + UnboundedReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x3e7884bf4f412c6full: // items + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.items, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Unit * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + uint64_t elem_len_items = 0; + if ( !sub.getleb( elem_len_items ) || !sub.room( elem_len_items ) ) { r.report->malformed = true; break; } + { + TableReader elem_items( sub.buffer + sub.offset, (int64_t) elem_len_items, r.report, r.ids ); + UnitLoadBody( elem_items, ( *slot ) ); + } + sub.offset += (int64_t) elem_len_items; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x56d7ab194448a4f3ull: // tag + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.tag = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// UnboundedWireExtent: the extent Unbounded's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool UnboundedWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x3e7884bf4f412c6full && field_kind == 14 ) // items: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Unit ), (int64_t) alignof( Unit ), 13, 2, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// UnboundedExtentAt: the node extent Unbounded's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as UnboundedExtentPack advances it (§2.8, §2.9). +template +inline bool UnboundedExtentAt( const Ctx & ctx, const Unbounded & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.items ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Unit ) - 1 ) & ~( (int64_t) alignof( Unit ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Unit ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t UnboundedExtent( const Ctx & ctx, const Unbounded & value ) +{ + int64_t at = 0; + if ( !UnboundedExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// UnboundedExtentPack: carve Unbounded's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset UnboundedExtentAt advances (§2.8, §2.9). +template +inline bool UnboundedExtentPack( const Ctx & ctx, const Unbounded & src, Unbounded & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.items ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Unit ) - 1 ) & ~( (int64_t) alignof( Unit ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Unit ); + if ( at + bytes > capacity ) { return false; } + Unit * placed = (Unit *) ( extent + at ); + at += bytes; + dst.items.count = cursor.count; + dst.items.padding = 0; + dst.items.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.items.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Unit ) ); // trivially copyable, by construction + } + } + return true; +} + +// ---- Unbounded.items: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Unit * UnboundedItemsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool UnboundedItemsErase( TableArena & arena, TableList & list, const Unit * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach UnboundedItemsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// UnboundedNumber: number everything Unbounded POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool UnboundedNumber( const Ctx & ctx, TableNumbering & numbering, const Unbounded & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// UnboundedPackMeasure: the packed region bytes of everything Unbounded POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t UnboundedPackMeasure( const Ctx & ctx, TablePackMap & seen, const Unbounded & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// UnboundedPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool UnboundedPackEdges( const Ctx & ctx, TablePackMap & seen, const Unbounded & src, Unbounded & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool UnboundedPack( const Ctx & ctx, TablePackMap & seen, const Unbounded & src, Unbounded & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Unbounded ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Unbounded ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !UnboundedExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return UnboundedPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool UnboundedPackEdges( const Ctx & ctx, TablePackMap & seen, const Unbounded & src, Unbounded & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// ---- Unbounded: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: UnboundedBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Unbounded is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct UnboundedBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + UnboundedBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~UnboundedBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + UnboundedBuilder( const UnboundedBuilder & ) = delete; + UnboundedBuilder & operator=( const UnboundedBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Unbounded * GetRoot() { return arena.locked ? NULL : (Unbounded *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Unbounded * AsConst() const { return (const Unbounded *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool UnboundedBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Unbounded & root = *(const Unbounded *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = UnboundedPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = UnboundedExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + Unbounded * destination = new ( packed ) Unbounded; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !UnboundedPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Unbounded on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// UnboundedNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t UnboundedNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// UnboundedNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void UnboundedNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// UnboundedNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t UnboundedNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// UnboundedNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t UnboundedNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// UnboundedNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void UnboundedNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = UnboundedNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? UnboundedNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool UnboundedNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Unbounded & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return UnboundedNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t UnboundedMeasureWire( const Ctx & ctx, const Unbounded & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( UnboundedNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = UnboundedMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t UnboundedSaveWire( const Ctx & ctx, const Unbounded & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !UnboundedNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = UnboundedSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == UnboundedMeasure( root ) +} + +inline int64_t UnboundedMeasure( const Unbounded * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return UnboundedMeasureWire( ctx, *root, allocator ); +} + +inline int64_t UnboundedSave( const Unbounded * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return UnboundedSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t UnboundedMeasure( const UnboundedBuilder & builder ) +{ + if ( builder.region != NULL ) { return UnboundedMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return UnboundedMeasureWire( ctx, *(const Unbounded *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t UnboundedSave( const UnboundedBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return UnboundedSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return UnboundedSaveWire( ctx, *(const Unbounded *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t UnboundedMeasureMessage( const Unbounded * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return UnboundedMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t UnboundedSaveMessage( const Unbounded * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return UnboundedSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t UnboundedMeasureMessage( const UnboundedBuilder & builder ) +{ + if ( builder.region != NULL ) { return UnboundedMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return UnboundedMeasureWire( ctx, *(const Unbounded *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t UnboundedSaveMessage( const UnboundedBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return UnboundedSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return UnboundedSaveWire( ctx, *(const Unbounded *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// UnboundedLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t UnboundedLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !UnboundedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = UnboundedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// UnboundedLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Unbounded * UnboundedLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Unbounded ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !UnboundedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = UnboundedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x2ec2f34386026d8bull; + Unbounded * root = new ( region ) Unbounded; // lifetime only: LoadBody's first act is UnboundedReset + UnboundedReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = UnboundedNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + UnboundedNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + UnboundedNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Unbounded ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + UnboundedLoadBody( r, nodes, *root ); + return root; +} + +// UnboundedLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t UnboundedLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !UnboundedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = UnboundedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// UnboundedLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Unbounded * UnboundedLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Unbounded ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !UnboundedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = UnboundedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x2ec2f34386026d8bull; + Unbounded * root = new ( region ) Unbounded; // lifetime only: LoadBody's first act is UnboundedReset + UnboundedReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Unbounded ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = UnboundedNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + UnboundedNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + UnboundedNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Unbounded ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + UnboundedLoadBody( r, nodes, *root ); + return root; +} + +// UnboundedLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool UnboundedLoadBuilder( UnboundedBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Unbounded * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0x2ec2f34386026d8bull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = UnboundedNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + UnboundedNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = UnboundedLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- the cooked form: point at a cook (docs/SPEC-TABLES.md §7) ---- + +// UnitOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Unit IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Unit * UnitOpen( const void * bytes, uint64_t length ) +{ + return (const Unit *) TableCookOpen( bytes, length, (uint64_t) sizeof( Unit ), (uint64_t) alignof( Unit ) ); +} + +// BoundedOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Bounded IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Bounded * BoundedOpen( const void * bytes, uint64_t length ) +{ + return (const Bounded *) TableCookOpen( bytes, length, (uint64_t) sizeof( Bounded ), (uint64_t) alignof( Bounded ) ); +} + +// UnboundedOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH UnboundedAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Unbounded * UnboundedOpen( const void * bytes, uint64_t length ) +{ + return (const Unbounded *) TableCookOpen( bytes, length, (uint64_t) sizeof( Unbounded ), (uint64_t) alignof( Unbounded ) ); +} + +// ---- the cooked form: WRITE a cook (docs/SPEC-TABLES.md §7.6) ---- +// +// The bytes are `schema cook`'s, and the tool stays the reference: the two +// writers are held to one file, byte for byte, in both byte orders. A cook is +// content-addressed by (asset hash, build version), so two writers of one +// instance produce ONE artifact or the pair means nothing. + +inline void UnitCookBody( uint8_t * at, const Unit & value, TableByteOrder order ); +inline void BoundedCookBody( uint8_t * at, const Bounded & value, TableByteOrder order ); +template inline bool UnboundedCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Unbounded & value, TableByteOrder order ); + +inline void UnitCookBody( uint8_t * at, const Unit & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.v, 4, order ); +} + +inline void BoundedCookBody( uint8_t * at, const Bounded & value, TableByteOrder order ) +{ + // all 8 slots: the storage is allocate-max, and a slot past the count rides as it lies (§7.2) + for ( int32_t i = 0; i < 8; i++ ) + { + UnitCookBody( at + 0 + i * 4, value.items[ i ], order ); + } + table_cook_put( at + 32, (uint64_t) (uint32_t) value.items_count, 4, order ); + table_cook_put( at + 36, (uint64_t) value.tag, 4, order ); +} + +template inline bool UnboundedCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Unbounded & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // items: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.tag, 4, order ); + return true; +} + +template inline bool UnitCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Unit & value, TableByteOrder order ); +template inline bool BoundedCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Bounded & value, TableByteOrder order ); +template inline bool UnboundedCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Unbounded & value, TableByteOrder order ); + +// UnitCookExtent: Unit's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool UnitCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Unit & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// BoundedCookExtent: Bounded's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool BoundedCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Bounded & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// UnboundedCookExtent: Unbounded's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool UnboundedCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Unbounded & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // items: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.items ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( Unit ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 4; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + UnitCookBody( array + i * 4, cursor[i], order ); + } + } + return true; +} + +// UnitCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool UnitCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Unit & value, TableByteOrder order ) +{ + UnitCookBody( at, value, order ); + int64_t extent_at = 0; + return UnitCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// BoundedCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool BoundedCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Bounded & value, TableByteOrder order ) +{ + BoundedCookBody( at, value, order ); + int64_t extent_at = 0; + return BoundedCookExtent( ctx, region, at + 40, extent_at, at, value, order ); +} + +// UnboundedCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool UnboundedCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Unbounded & value, TableByteOrder order ) +{ + if ( !UnboundedCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return UnboundedCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// UnitCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Unit IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t UnitCookMeasure( const Unit & value ) +{ + (void) value; + return 88; // 64 header + 8 data + 16 attribution +} + +// UnitCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract UnitMeasure/UnitSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool UnitCook( const Unit & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) UnitCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 8, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + UnitCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 72, 0, 8, order ); + table_cook_put( raw + 80, 0x8f26e0f086cc2787ull, 8, order ); + return true; +} + +// BoundedCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Bounded IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t BoundedCookMeasure( const Bounded & value ) +{ + (void) value; + return 120; // 64 header + 40 data + 16 attribution +} + +// BoundedCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract BoundedMeasure/BoundedSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool BoundedCook( const Bounded & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) BoundedCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 40, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + BoundedCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 104, 0, 8, order ); + table_cook_put( raw + 112, 0x0acac10912f5892aull, 8, order ); + return true; +} + +// UnboundedCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool UnboundedCookLayout( const Ctx & ctx, const Unbounded & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = UnboundedExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// UnboundedCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t UnboundedCookMeasureFrom( const Ctx & ctx, const Unbounded & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( UnboundedNumberFrom( ctx, numbering, root ) && UnboundedCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// UnboundedCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool UnboundedCookFrom( const Ctx & ctx, const Unbounded & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = UnboundedNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && UnboundedCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = UnboundedCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0x2ec2f34386026d8bull, 8, order ); // the root: fnv1a64( "Unbounded" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// UnboundedCookMeasure / UnboundedCook over a REGION root — a locked builder's AsConst, a +// region UnboundedLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t UnboundedCookMeasure( const Unbounded * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return UnboundedCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool UnboundedCook( const Unbounded * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return UnboundedCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t UnboundedCookMeasure( const UnboundedBuilder & builder ) +{ + if ( builder.region != NULL ) { return UnboundedCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return UnboundedCookMeasureFrom( ctx, *(const Unbounded *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool UnboundedCook( const UnboundedBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return UnboundedCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return UnboundedCookFrom( ctx, *(const Unbounded *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ---- relocatability, enforced: the wire is a pure length-prefixed +// stream AND the decoded storage is pointer-free — every closure type +// must stay trivially copyable and standard-layout, so instances can be +// memcpy'd, mmap'd, shared across processes, and walked through +// descriptor offsets. A failure here means a pointer, virtual or +// non-trivial member crept into generated storage. +// +// They ask the COMPILER ITSELF, which is what every C++ standard library +// answers the same two questions with — and it costs this header no +// include at all. +// A pointer FIELD is a TableRef — eight bytes and no address — so the +// property holds in BOTH forms: a fixed-size table is one relocatable +// struct, and a packed region is one relocatable block whose references +// are self-relative and therefore survive a plain memcpy. +static_assert( __is_trivially_copyable( Unit ), "Unit must stay relocatable" ); +static_assert( __is_standard_layout( Unit ), "Unit must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Bounded ), "Bounded must stay relocatable" ); +static_assert( __is_standard_layout( Bounded ), "Bounded must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Unbounded ), "Unbounded must stay relocatable" ); +static_assert( __is_standard_layout( Unbounded ), "Unbounded must stay standard-layout for offsetof" ); + +// ---- the cook's layout contract (docs/SPEC-TABLES.md §20.3) ---- +// +// The compiler derived every number below from the declaration and folded it +// into the BUILD VERSION; these asserts are this compiler saying whether it +// agrees. The model is not self-evidently right — on 32-bit System V +// alignof(uint64_t) is 4, not 8 — which is precisely why it is asserted +// rather than assumed. +static_assert( sizeof( Unit ) == 4, "Unit's sizeof moved: the build version was taken over 4, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Unit ) == 4, "Unit's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Unit, v ) == 0, "Unit's field v moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Bounded ) == 40, "Bounded's sizeof moved: the build version was taken over 40, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Bounded ) == 4, "Bounded's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Bounded, items ) == 0, "Bounded's field items moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Bounded, tag ) == 36, "Bounded's field tag moved: the build version was taken over offset 36 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Unbounded ) == 24, "Unbounded's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Unbounded ) == 8, "Unbounded's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Unbounded, items ) == 0, "Unbounded's field items moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Unbounded, tag ) == 16, "Unbounded's field tag moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); + +static_assert( alignof( Unit ) <= kTableAlign, "Unbounded.items: an unbounded array's element alignment must fit the arena's" ); + +// ---- reflection descriptors (tables only, docs/SPEC-TABLES.md) ---- + +inline const TableTypeInfo * UnitTableType(); +inline const TableTypeInfo * BoundedTableType(); +inline const TableTypeInfo * UnboundedTableType(); +// The descriptors are CONSTANT-INITIALISED data, and a field's target is +// the ADDRESS of another descriptor. These declarations are what let a +// self- or mutually-referential graph — Node naming itself through *Node — +// be expressed as constant data instead of a lazy link, which could not +// have been written race-free OR recursion-safe. The whole reflection +// surface is therefore immutable: read it from any thread, any time. +extern const TableTypeInfo UnitTableInfo; +extern const TableTypeInfo BoundedTableInfo; +extern const TableTypeInfo UnboundedTableInfo; + +inline const TableFieldInfo UnitTableFields[] = { + { "v", "v", "int32", 0xaf63eb4c86020609ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Unit, v ), (uint32_t) sizeof( Unit::v ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo UnitTableInfo = { "Unit", (uint32_t) sizeof( Unit ), 1, UnitTableFields, +[]( void * p ) { UnitReset( *(Unit *) p ); }, false }; +inline const TableTypeInfo * UnitTableType() { return &UnitTableInfo; } + +inline const TableFieldInfo BoundedTableFields[] = { + { "items", "items", "Unit", 0x3e7884bf4f412c6full, 13, true, false, NULL, NULL, true, false, 8, (uint32_t) offsetof( Bounded, items ), (uint32_t) sizeof( Bounded::items[0] ), (uint32_t) offsetof( Bounded, items_count ), 0xffffffffu, &UnitTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "tag", "tag", "int32", 0x56d7ab194448a4f3ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Bounded, tag ), (uint32_t) sizeof( Bounded::tag ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo BoundedTableInfo = { "Bounded", (uint32_t) sizeof( Bounded ), 2, BoundedTableFields, +[]( void * p ) { BoundedReset( *(Bounded *) p ); }, false }; +inline const TableTypeInfo * BoundedTableType() { return &BoundedTableInfo; } + +inline const TableFieldInfo UnboundedTableFields[] = { + { "items", "items", "Unit", 0x3e7884bf4f412c6full, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Unbounded, items ), (uint32_t) sizeof( Unit ), (uint32_t) offsetof( Unbounded, items.count ), 0xffffffffu, &UnitTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "tag", "tag", "int32", 0x56d7ab194448a4f3ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Unbounded, tag ), (uint32_t) sizeof( Unbounded::tag ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo UnboundedTableInfo = { "Unbounded", (uint32_t) sizeof( Unbounded ), 2, UnboundedTableFields, +[]( void * p ) { UnboundedReset( *(Unbounded *) p ); }, true }; +inline const TableTypeInfo * UnboundedTableType() { return &UnboundedTableInfo; } + +// ---- the text form (docs/SPEC-TABLES.md §16) ---- + +// Unit in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// MigrateTable.cpp; link it to use them. +bool UnitFromJson( Unit & value, const char * text, int64_t bytes, TableReport * report ); +int64_t UnitToJsonMeasure( const Unit & value ); +int64_t UnitToJson( const Unit & value, char * buffer, int64_t capacity ); + +// Bounded in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// MigrateTable.cpp; link it to use them. +bool BoundedFromJson( Bounded & value, const char * text, int64_t bytes, TableReport * report ); +int64_t BoundedToJsonMeasure( const Bounded & value ); +int64_t BoundedToJson( const Bounded & value, char * buffer, int64_t capacity ); + +// Unbounded in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in MigrateTable.cpp; link it to use them. +bool UnboundedFromJson( UnboundedBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t UnboundedToJsonMeasure( const Unbounded * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t UnboundedToJson( const Unbounded * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/ReportTable.cpp b/testdata/golden/tables/lists/ReportTable.cpp new file mode 100644 index 000000000..2e5d5ad3f --- /dev/null +++ b/testdata/golden/tables/lists/ReportTable.cpp @@ -0,0 +1,3153 @@ +// Code generated by the schema compiler from Report.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — the TABLE wire's text form (docs/SPEC-TABLES.md §16). +// Compile this file to use FromJson / ToJson; a project that +// never reads or writes a text does not compile it and pays nothing. + +#include "ReportTable.h" + +#include // the text form: number formatting +#include // the text form: exact number conversion +#include // the text form: the runtime's decimal point + +// The guard is not vestigial. Several listdemo Table.cpp files may be +// concatenated into ONE translation unit — a unity build — and without it +// each would redefine the walk. It is also why the walk's functions may be +// weak (vague linkage) across separate objects: ODR requires their +// definitions to be token-identical, and the generic-walk gate is what +// proves that, byte for byte, across every generated .cpp. +#ifndef LISTDEMO_SCHEMA_TABLE_JSON +#define LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +// ---- the pointer adapters (docs/SPEC-TABLES.md §16.7) ---- +// +// The walk below is ONE walk, byte-identical in every generated .cpp, and a +// pointer is the one kind it cannot walk alone: reading one needs the +// builder's arena and writing one needs a region's deref, and neither exists +// in a unit that declares no pointer. So the walk calls these three and does +// not define them. A unit with no pointer defines them as stubs no field ever +// reaches; a pointered unit defines them in the graph half that follows the +// walk. + +struct TableJsonIn; +struct TableJsonOut; + +// a pointer field's object, or the `&node` reference standing in for it, into +// the slot; the cursor is on the opening brace +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// the node a pointer slot names, in place — or as `&node` when it is shared +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// the FIRST key of an object the walk is skipping begins with `&`: the cursor is +// on its value. A dropped definition still takes its label (§16.7); a fixed reader +// skips the value whole, as it skips everything else it does not place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); + +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- +// +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map +inline bool TableJsonIsMap( const TableFieldInfo * f ); +// the map as a plain JSON object keyed by the KEY, in ASCENDING key order +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that object back into the slot, in whatever order the text gives it +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); + +// ---- json walk: begin ---- +// +// The TEXT form (docs/SPEC-TABLES.md §16): one table, one text, one walk over the +// reflection descriptors (§8). Reading fills ONE caller-owned instance and +// allocates nothing beyond it; writing targets a caller buffer with the +// wire's measure/write symmetry. Everything AROUND this — which file goes +// with which instance, what key an instance is filed under, how instances +// link into a root table's collections — is a packer's opinion and stays +// with the tool that holds it. +// +// The dialect: trailing commas are accepted on read (the authoring files +// this exists for carry them) and never written; comments are not JSON and +// are refused; unknown keys are skipped and counted; a duplicate key is +// last-wins and counted; a key present with the wrong JSON type is skipped +// and counted, never coerced. + +static const int32_t kTableJsonMaxDepth = 128; + +// A key longer than this cannot name a field, so it is skipped as unknown. +static const int32_t kTableJsonMaxKey = 256; + +// The longest numeric token the walk will convert. Anything longer is a +// value no field can hold and counts as a kind mismatch. +static const int32_t kTableJsonMaxNumber = 512; + +// The decimal point the C runtime is CURRENTLY using. Number conversion is +// the one locale-sensitive corner of the grammar — JSON's point is always +// '.', the runtime's is whatever the program set — so every number crosses +// this one character on the way out and on the way back in. Nothing else in +// the walk consults the locale. +inline char TableJsonDecimalPoint() +{ + const struct lconv * conv = localeconv(); + if ( conv != NULL && conv->decimal_point != NULL && conv->decimal_point[0] != 0 ) + { + return conv->decimal_point[0]; + } + return '.'; +} + +// ---- storage access: the descriptors give an offset and a width, and the +// ---- storage is the HOST's, so every load and store goes through a width +// ---- switch rather than a memcpy into the low bytes of a wider word + +// finite: not a NaN, not an infinity. Written without — the walk's +// runtime surface stays the handful of functions it already names. +// A vocabulary entry the descriptor could not spell. The generated name +// functions answer "???" for a value outside the declared set, and that is +// not a name — writing it would put a spelling in the text that the reader +// then counts as unknown, turning a refusal into a silent loss. +inline bool TableJsonNamed( const char * name ) +{ + return name != NULL && strcmp( name, "???" ) != 0; +} + +inline bool TableJsonFinite( double v ) +{ + return v == v && v <= 1.7976931348623157e308 && v >= -1.7976931348623157e308; +} + +inline uint64_t TableJsonGetRaw( const void * storage, uint32_t width ) +{ + switch ( width ) + { + case 1: { uint8_t v = 0; memcpy( &v, storage, 1 ); return v; } + case 2: { uint16_t v = 0; memcpy( &v, storage, 2 ); return v; } + case 4: { uint32_t v = 0; memcpy( &v, storage, 4 ); return v; } + case 8: { uint64_t v = 0; memcpy( &v, storage, 8 ); return v; } + } + return 0; +} + +inline void TableJsonSetRaw( void * storage, uint32_t width, uint64_t value ) +{ + switch ( width ) + { + case 1: { uint8_t v = (uint8_t) value; memcpy( storage, &v, 1 ); break; } + case 2: { uint16_t v = (uint16_t) value; memcpy( storage, &v, 2 ); break; } + case 4: { uint32_t v = (uint32_t) value; memcpy( storage, &v, 4 ); break; } + case 8: { uint64_t v = value; memcpy( storage, &v, 8 ); break; } + } +} + +inline int64_t TableJsonGetSigned( const void * storage, uint32_t width ) +{ + uint64_t raw = TableJsonGetRaw( storage, width ); + if ( width < 8 ) + { + uint64_t sign = uint64_t( 1 ) << ( width * 8 - 1 ); + if ( ( raw & sign ) != 0 ) + { + raw |= ~( ( sign << 1 ) - 1 ); + } + } + return (int64_t) raw; +} + +// ---- the WIDE kinds (docs/SPEC-TABLES.md §3, §16.2) ---- +// +// The 128-bit integers and the fixed-point family convert EXACTLY, over two +// 64-bit lanes: a 128-bit integer is a decimal integer, a fixed value a +// decimal in WHOLE UNITS (1.0, -0.25, 3.0000152587890625) and nothing +// on either path passes through a double. Nothing here needs a 128-bit type +// either, which is what keeps this walk one text for every unit. +struct TableJsonWide +{ + uint64_t lo; + uint64_t hi; +}; + +inline bool TableJsonKindWide( uint8_t kind ) { return kind >= 18 && kind <= 29; } +inline bool TableJsonKindWideSigned( uint8_t kind ) { return kind == 18 || ( kind >= 20 && kind <= 24 ); } +inline bool TableJsonKindFixed( uint8_t kind ) { return kind >= 20 && kind <= 29; } + +inline bool TableJsonWideZero( TableJsonWide v ) { return v.lo == 0 && v.hi == 0; } +inline bool TableJsonWideNegative( TableJsonWide v ) { return ( v.hi >> 63 ) != 0; } + +inline int TableJsonWideCompare( TableJsonWide a, TableJsonWide b, bool is_signed ) +{ + if ( is_signed && TableJsonWideNegative( a ) != TableJsonWideNegative( b ) ) { return TableJsonWideNegative( a ) ? -1 : 1; } + if ( a.hi != b.hi ) { return a.hi < b.hi ? -1 : 1; } + if ( a.lo != b.lo ) { return a.lo < b.lo ? -1 : 1; } + return 0; +} + +inline TableJsonWide TableJsonWideShl( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.hi = v.lo << ( n - 64 ); return r; } + r.hi = ( v.hi << n ) | ( v.lo >> ( 64 - n ) ); + r.lo = v.lo << n; + return r; +} + +inline TableJsonWide TableJsonWideShr( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.lo = v.hi >> ( n - 64 ); return r; } + r.lo = ( v.lo >> n ) | ( v.hi << ( 64 - n ) ); + r.hi = v.hi >> n; + return r; +} + +inline TableJsonWide TableJsonWideNeg( TableJsonWide v ) +{ + TableJsonWide r; + r.lo = ~v.lo + 1; + r.hi = ~v.hi + ( r.lo == 0 ? 1 : 0 ); + return r; +} + +// v = v * m + a; the return is the carry out of 128 bits +inline uint32_t TableJsonWideMulAdd( TableJsonWide * v, uint32_t m, uint32_t a ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t carry = a; + for ( int i = 0; i < 4; i++ ) + { + uint64_t p = limb[i] * m + carry; + limb[i] = p & 0xffffffffull; + carry = p >> 32; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) carry; +} + +// v = v / d; the return is the remainder +inline uint32_t TableJsonWideDiv( TableJsonWide * v, uint32_t d ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t rem = 0; + for ( int i = 3; i >= 0; i-- ) + { + uint64_t cur = ( rem << 32 ) | limb[i]; + limb[i] = cur / d; + rem = cur % d; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) rem; +} + +// The storage of a wide kind, as lanes. A sixteen-byte storage is serialize's +// pair — native __int128 in the host's byte order, or the emulated struct with +// its low lane first — so the lanes are read in the host's order; a narrower +// storage is one lane, sign-extended for a signed kind. +inline TableJsonWide TableJsonWideLoad( const void * storage, uint32_t width, bool is_signed ) +{ + TableJsonWide v = { 0, 0 }; + if ( width == 16 ) + { + uint64_t half[2]; + memcpy( half, storage, 16 ); + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + v.lo = little ? half[0] : half[1]; + v.hi = little ? half[1] : half[0]; + return v; + } + v.lo = is_signed ? (uint64_t) TableJsonGetSigned( storage, width ) : TableJsonGetRaw( storage, width ); + v.hi = ( is_signed && ( v.lo >> 63 ) != 0 ) ? ~uint64_t( 0 ) : 0; + return v; +} + +inline void TableJsonWideStore( void * storage, uint32_t width, TableJsonWide v ) +{ + if ( width == 16 ) + { + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + uint64_t half[2]; + half[0] = little ? v.lo : v.hi; + half[1] = little ? v.hi : v.lo; + memcpy( storage, half, 16 ); + return; + } + TableJsonSetRaw( storage, width, v.lo ); +} + +// a counted field's companion: a string's length, a bytes' length, a counted +// array's count. Bounded by the declared extent on the way out, so a storage +// invariant a caller broke cannot walk off the end of the array. +inline int32_t TableJsonCount( const void * base, const TableFieldInfo * f ) +{ + if ( !f->counted ) + { + return f->array_bound; + } + int32_t count = 0; + memcpy( &count, (const uint8_t *) base + f->count_offset, sizeof( count ) ); + if ( count < 0 ) { count = 0; } + if ( count > f->array_bound ) { count = f->array_bound; } + return count; +} + +inline void TableJsonSetCount( void * base, const TableFieldInfo * f, int32_t count ) +{ + if ( f->counted ) + { + memcpy( (uint8_t *) base + f->count_offset, &count, sizeof( count ) ); + } +} + +// ---- what a field's kind expects to see in the text ---- +// +// One classifier, consulted by both directions, so a reader and a writer can +// never disagree about a kind's JSON form. 'o' object, 'a' array, 's' +// string, 'n' number, 'b' boolean. +// +// A vocabulary field is spelled by NAME: an enum is one name, a flags mask +// is the array of the names of its set bits. The two are told apart by the +// id column — an enum variant rides under a wire id, a flags BIT never does +// (docs/SPEC-TABLES.md §4), so a name function with no id function is flags. +// +// bytes(N) is the one kind whose element kind does not decide its form: it +// shares u8 with a plain array of u8, and rides as base64. The schema type +// name settles it, and "bytes" is a keyword no declaration can claim. +inline bool TableJsonIsBytes( const TableFieldInfo * f ) +{ + return f->is_array && f->kind == 6 && strcmp( f->type_name, "bytes" ) == 0; +} + +// An ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): its JSON form is an OBJECT +// keyed by variant name, not a positional array, because that is what the +// storage is — one slot per variant, addressed by the variant. +inline bool TableJsonIsKeyed( const TableFieldInfo * f ) +{ + return f->key_name != NULL; +} + +// THE KEY A STORAGE SLOT HOLDS (§2.4, §8): the storage shifts left, so slot i +// holds the key i + 1 and nothing is stored for None. This is the ONE place +// the walker spells the shift. +inline uint64_t TableJsonKeyedSlotKey( int64_t slot ) +{ + return (uint64_t) ( slot + 1 ); +} + +// A slot whose key names a variant of the keying enum. Every slot in +// [0, array_bound) does, unless the enum carries max-headroom variants outside +// a table closure, where a reserved value names nothing and its key id is 0 — +// the reserved id no declared name can fold to (§5). +inline bool TableJsonKeyedSlotValid( const TableFieldInfo * f, int64_t slot ) +{ + return f->key_id( TableJsonKeyedSlotKey( slot ) ) != 0; +} + +inline bool TableJsonIsFlags( const TableFieldInfo * f ) +{ + return f->enum_name != NULL && f->variant_id == NULL; +} + +inline bool TableJsonIsEnum( const TableFieldInfo * f ) +{ + return f->variant_id != NULL && f->arms == NULL; +} + +inline char TableJsonShape( const TableFieldInfo * f ) +{ + if ( TableJsonIsMap( f ) ) return 'o'; // a MAP: an object keyed by the KEY (§2.8) + if ( f->kind == 12 ) return 's'; // string + if ( TableJsonIsBytes( f ) ) return 's'; // bytes: base64 + if ( TableJsonIsKeyed( f ) ) return 'o'; // an object keyed by variant NAME + if ( f->is_array ) return 'a'; + if ( f->arms != NULL ) return 'o'; // union: an object with ONE key + if ( f->kind == 13 ) return 'o'; // nested table or type + if ( f->kind == 17 ) return f->table != NULL ? 'o' : 's'; // a pointer: the pointee's object in place, or null (§16.7); a byte buffer's string (§2.5) + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// the ELEMENT shape of an array field — the same classifier one level down +inline char TableJsonElementShape( const TableFieldInfo * f ) +{ + if ( f->arms != NULL ) return 'o'; // an element of an array of unions: one key, the arm (§2.6) + if ( f->kind == 13 ) return 'o'; + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// A guarded group rides only when its guard reads true — the wire's own +// elision (§4), carried into the text so a text and a wire written from one +// instance say the same thing. The guard is spelled as its branch condition +// over bool fields of the SAME type ("at_rest", "!at_rest", +// "active && has_target"), so evaluating it is a walk of the same +// descriptor. Nothing is inferred in the other direction: reading places +// every key it can name, and the guard is a plain bool key (§16.2). +inline bool TableJsonGuardHolds( const void * base, const TableTypeInfo * info, const char * guard ) +{ + const char * p = guard; + for ( ;; ) + { + while ( *p == ' ' || *p == '&' ) { p++; } + if ( *p == 0 ) { return true; } + bool want = true; + if ( *p == '!' ) { want = false; p++; } + const char * start = p; + while ( *p != 0 && *p != ' ' && *p != '&' ) { p++; } + size_t length = (size_t) ( p - start ); + bool value = false; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( strlen( f->name ) == length && strncmp( f->name, start, length ) == 0 ) + { + value = TableJsonGetRaw( (const uint8_t *) base + f->offset, f->elem_size ) != 0; + break; + } + } + if ( value != want ) { return false; } + } +} + +// ---- writing ---- + +// The writer sink MEASURES when the buffer is NULL and WRITES when it is +// not, over one code path — so measure and write agree byte for byte, the +// wire's invariant (§9) carried across. +struct TableJsonOut +{ + char * buffer; + int64_t capacity; + int64_t offset; + bool overflow; + void * graph; // the pointered write's identity map (§16.7); NULL for a fixed table + + void raw( const char * data, int64_t count ) + { + if ( buffer != NULL ) + { + if ( offset + count > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) count ); + } + offset += count; + } + void put( char c ) { raw( &c, 1 ); } + void text( const char * s ) { raw( s, (int64_t) strlen( s ) ); } + void line( int32_t depth ) + { + put( '\n' ); + for ( int32_t i = 0; i < depth; i++ ) { raw( " ", 2 ); } + } +}; + +inline const char * TableJsonBase64Alphabet() +{ + return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; +} + +inline void TableJsonWriteBase64( TableJsonOut & out, const uint8_t * data, int32_t length ) +{ + const char * alphabet = TableJsonBase64Alphabet(); + out.put( '"' ); + int32_t i = 0; + for ( ; i + 3 <= length; i += 3 ) + { + uint32_t triple = ( uint32_t( data[i] ) << 16 ) | ( uint32_t( data[i+1] ) << 8 ) | uint32_t( data[i+2] ); + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], + alphabet[ ( triple >> 6 ) & 0x3f ], alphabet[ triple & 0x3f ] }; + out.raw( quad, 4 ); + } + if ( i < length ) + { + int32_t left = length - i; + uint32_t triple = uint32_t( data[i] ) << 16; + if ( left == 2 ) { triple |= uint32_t( data[i+1] ) << 8; } + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], '=', '=' }; + if ( left == 2 ) { quad[2] = alphabet[ ( triple >> 6 ) & 0x3f ]; } + out.raw( quad, 4 ); + } + out.put( '"' ); +} + +// One UTF-8 sequence at s, or -1 when the bytes there are not one. Rejects +// the lot: a stray continuation, an overlong form, a surrogate half, and +// anything past U+10FFFF. +inline int32_t TableJsonUtf8( const char * s, int32_t remaining, int32_t * width ) +{ + unsigned char lead = (unsigned char) s[0]; + int32_t want = 0; + int32_t code = 0; + if ( lead < 0x80 ) { *width = 1; return lead; } + else if ( lead >= 0xc2 && lead <= 0xdf ) { want = 2; code = lead & 0x1f; } + else if ( lead >= 0xe0 && lead <= 0xef ) { want = 3; code = lead & 0x0f; } + else if ( lead >= 0xf0 && lead <= 0xf4 ) { want = 4; code = lead & 0x07; } + else { return -1; } + if ( remaining < want ) { return -1; } + for ( int32_t i = 1; i < want; i++ ) + { + unsigned char next = (unsigned char) s[i]; + if ( ( next & 0xc0 ) != 0x80 ) { return -1; } + code = ( code << 6 ) | ( next & 0x3f ); + } + if ( want == 3 && code < 0x800 ) { return -1; } // overlong + if ( want == 4 && code < 0x10000 ) { return -1; } // overlong + if ( code >= 0xd800 && code <= 0xdfff ) { return -1; } // a surrogate half + if ( code > 0x10ffff ) { return -1; } + *width = want; + return code; +} + +// A JSON text MUST be valid UTF-8 (RFC 8259 §8.1). The read path is +// byte-transparent — the wire imposes no encoding (§3) and a string may hold +// anything — so the WRITER is where that obligation is met: a byte that is +// not part of a well-formed sequence is written as U+FFFD, one per bad byte, +// and never raw. A text this walk writes is therefore readable by any +// conforming parser, which a raw byte would not be. The cost is stated +// plainly: for a string holding invalid UTF-8, the round trip is NOT +// byte-identical, because the alternative is emitting a text that is not +// JSON. +inline void TableJsonWriteString( TableJsonOut & out, const char * s, int32_t length ) +{ + static const char hex[] = "0123456789abcdef"; + out.put( '"' ); + for ( int32_t i = 0; i < length; i++ ) + { + unsigned char c = (unsigned char) s[i]; + switch ( c ) + { + case '"': out.raw( "\\\"", 2 ); break; + case '\\': out.raw( "\\\\", 2 ); break; + case '\b': out.raw( "\\b", 2 ); break; + case '\f': out.raw( "\\f", 2 ); break; + case '\n': out.raw( "\\n", 2 ); break; + case '\r': out.raw( "\\r", 2 ); break; + case '\t': out.raw( "\\t", 2 ); break; + default: + if ( c < 0x20 ) + { + char escape[6] = { '\\', 'u', '0', '0', hex[ c >> 4 ], hex[ c & 0xf ] }; + out.raw( escape, 6 ); + } + else if ( c < 0x80 ) + { + out.put( (char) c ); + } + else + { + int32_t width = 0; + if ( TableJsonUtf8( s + i, length - i, &width ) < 0 ) + { + out.raw( "\xef\xbf\xbd", 3 ); // U+FFFD, one per bad byte + } + else + { + out.raw( s + i, width ); + i += width - 1; + } + } + break; + } + } + out.put( '"' ); +} + +inline void TableJsonWriteUnsigned( TableJsonOut & out, uint64_t value ) +{ + char digits[24]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) ( value % 10 ) ); + value /= 10; + } while ( value != 0 ); + char text[24]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); +} + +inline void TableJsonWriteSigned( TableJsonOut & out, int64_t value ) +{ + if ( value < 0 ) + { + out.put( '-' ); + TableJsonWriteUnsigned( out, uint64_t( 0 ) - (uint64_t) value ); + return; + } + TableJsonWriteUnsigned( out, (uint64_t) value ); +} + +// A wide kind writes its raw storage as §16.2's text: a 128-bit integer as a +// decimal integer; a fixed value in WHOLE UNITS as the shortest exact decimal +// with at least one fractional digit (1.0, -0.25), the spelling the schema text +// gives a fixed default. The fraction terminates because a dyadic fraction has +// a finite decimal expansion — at most F digits. +inline void TableJsonWriteWide( TableJsonOut & out, const void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + TableJsonWide v = TableJsonWideLoad( storage, f->elem_size, is_signed ); + if ( is_signed && TableJsonWideNegative( v ) ) + { + out.put( '-' ); + v = TableJsonWideNeg( v ); + } + int frac = f->frac_bits; + TableJsonWide whole = TableJsonWideShr( v, frac ); + char digits[40]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) TableJsonWideDiv( &whole, 10 ) ); + } while ( !TableJsonWideZero( whole ) ); + char text[40]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); + if ( !TableJsonKindFixed( f->kind ) ) { return; } + out.put( '.' ); + // the fraction bits alone: v with everything at and above bit F cleared + TableJsonWide fraction = v; + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + if ( frac == 0 ) { fraction.lo = 0; } + if ( TableJsonWideZero( fraction ) ) + { + out.put( '0' ); + return; + } + while ( !TableJsonWideZero( fraction ) ) + { + // ×10: the digit is what lands at and above bit F, including the + // carry out of 128 bits when F leaves no room for it below + uint32_t carry = TableJsonWideMulAdd( &fraction, 10, 0 ); + uint64_t digit = TableJsonWideShr( fraction, frac ).lo; + if ( frac > 64 ) { digit |= uint64_t( carry ) << ( 128 - frac ); } + out.put( (char) ( '0' + (int) digit ) ); + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + } +} + +// A float writes at the SHORTEST precision that reads back as the same value +// at the field's own width, so a round trip is exact and a text stays +// readable. Non-finite values have no JSON spelling at all, and the writer +// REFUSES rather than losing one silently — the same rule measure and save +// already apply to an enum value no variant names (§5). +inline bool TableJsonWriteFloat( TableJsonOut & out, double value, bool single ) +{ + if ( !TableJsonFinite( value ) ) { return false; } + char text[64]; + int low = single ? 6 : 15; + int high = single ? 9 : 17; + int length = 0; + for ( int digits = low; ; digits++ ) + { + length = snprintf( text, sizeof( text ), "%.*g", digits, value ); + if ( length <= 0 || length >= (int) sizeof( text ) ) { return false; } + if ( digits >= high ) { break; } + // the round-trip check runs BEFORE the decimal point is normalised: + // the token still carries whatever point snprintf just produced + if ( single ) + { + if ( (double) strtof( text, NULL ) == value ) { break; } + } + else + { + if ( strtod( text, NULL ) == value ) { break; } + } + } + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int i = 0; i < length; i++ ) + { + if ( text[i] == point ) { text[i] = '.'; } + } + } + out.raw( text, length ); + return true; +} + +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration writes through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ); + +// one scalar, at one storage address: a nested object, a union, a +// vocabulary, or a number +inline bool TableJsonWriteScalar( TableJsonOut & out, const void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; None is {} + const TableUnionInfo * arms = f->arms(); + uint64_t tag = TableJsonGetRaw( (const uint8_t *) storage + arms->tag_offset, arms->tag_size ); + if ( tag == 0 ) + { + out.raw( "{}", 2 ); + return true; + } + if ( (int64_t) tag > f->enum_max ) + { + return false; // a tag no arm names, exactly as measure refuses it + } + const char * arm = f->enum_name( tag ); + // and refuse on the NAME, not merely on the bound: §16.2 says a value + // no variant NAMES is refused, so the check is the name. Writing + // whatever came back would emit "???", a spelling the reader counts + // as unknown — a silent round-trip loss in place of a refusal. + if ( !TableJsonNamed( arm ) ) { return false; } + out.put( '{' ); + out.line( depth + 1 ); + TableJsonWriteString( out, arm, (int32_t) strlen( arm ) ); + out.raw( ": ", 2 ); + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2): an arm that names + // no declaration carries the FIELD descriptor a field of its type + // would carry, offsets taken inside the union storage (§2.6), so the + // value walks through the field writer one key down. + if ( arms->arms[tag].field != NULL ) + { + if ( !TableJsonWriteField( out, storage, arms->arms[tag].field, depth + 1 ) ) + { + return false; + } + } + else if ( arms->arms[tag].table == NULL ) + { + out.raw( "null", 4 ); // a payload-free arm: the name selects it (§2.6) + } + else if ( !TableJsonWriteValue( out, (const uint8_t *) storage + arms->arms[tag].offset, arms->arms[tag].table, depth + 1 ) ) + { + return false; + } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->kind == 13 ) + { + return TableJsonWriteValue( out, storage, f->table, depth ); + } + if ( TableJsonIsEnum( f ) ) + { + uint64_t value = TableJsonGetRaw( storage, f->elem_size ); + // a value no variant names has no text spelling, exactly as it has no + // wire identity: the writer REFUSES rather than writing None over it, + // the rule measure and save already apply (docs/SPEC-TABLES.md §5) + if ( (int64_t) value > f->enum_max ) { return false; } + if ( value != 0 && f->variant_id( value ) == 0 ) { return false; } + const char * name = f->enum_name( value ); + if ( !TableJsonNamed( name ) ) { return false; } + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + return true; + } + if ( TableJsonIsFlags( f ) ) + { + uint64_t bits = TableJsonGetRaw( storage, f->elem_size ); + if ( bits == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + bool first = true; + for ( int64_t bit = 0; bit < 64; bit++ ) + { + if ( ( bits & ( uint64_t( 1 ) << bit ) ) == 0 ) { continue; } + if ( bit > f->enum_max ) + { + return false; // a bit no variant names has no text spelling + } + const char * name = f->enum_name( (uint64_t) bit ); + if ( !TableJsonNamed( name ) ) { return false; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + } + out.line( depth ); + out.put( ']' ); + return true; + } + switch ( f->kind ) + { + case 1: + out.text( TableJsonGetRaw( storage, f->elem_size ) != 0 ? "true" : "false" ); + return true; + case 10: + { + float v = 0.0f; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, (double) v, true ); + } + case 11: + { + double v = 0.0; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, v, false ); + } + case 2: case 3: case 4: case 5: + TableJsonWriteSigned( out, TableJsonGetSigned( storage, f->elem_size ) ); + return true; + default: + if ( TableJsonKindWide( f->kind ) ) + { + TableJsonWriteWide( out, storage, f ); + return true; + } + TableJsonWriteUnsigned( out, TableJsonGetRaw( storage, f->elem_size ) ); + return true; + } +} + +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ) +{ + const uint8_t * storage = (const uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonWriteMap( out, (const void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } + if ( f->kind == 17 && !f->is_array ) + { + return TableJsonWritePointer( out, storage, f, depth ); + } + if ( f->kind == 17 ) + { + // an ARRAY OF POINTERS (§2.1): the pointer row per element — the + // pointee's object in place, null, or `&node` for a shared one (§16.7) + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWritePointer( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; + } + if ( f->kind == 12 ) + { + TableJsonWriteString( out, (const char *) storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + TableJsonWriteBase64( out, storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + // one entry per SLOT, keyed by the variant that owns it, so inserting + // a variant next season moves nothing in the text either. Slot i holds + // the key i + 1: nothing is stored for None, so nothing is written for it. + out.put( '{' ); + bool first = true; + for ( int64_t slot = 0; slot < f->array_bound; slot++ ) + { + if ( !TableJsonKeyedSlotValid( f, slot ) ) { continue; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + const char * key = f->key_name( TableJsonKeyedSlotKey( slot ) ); + TableJsonWriteString( out, key, (int32_t) strlen( key ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteScalar( out, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + if ( first ) { out.raw( "}", 1 ); return true; } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->is_array ) + { + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWriteScalar( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + out.line( depth ); + out.put( ']' ); + return true; + } + return TableJsonWriteScalar( out, storage, f, depth ); +} + +// One instance's fields, in DECLARATION ORDER, defaults included — a text is +// for people and tools, and a text that elides is a text a reader has to know +// the schema to complete. `any` says whether the object is already open on +// entry — a shared node's `&node` opens it before the fields (§16.7) — and +// whether it is open on return. +inline bool TableJsonWriteFields( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth, bool & any ) +{ + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( f->guard[0] != 0 && !TableJsonGuardHolds( base, info, f->guard ) ) { continue; } + // an ABSENT optional writes no key: presence of the key IS the + // presence (§16.2), so an absent field is an absent key and nothing + // else would read back as absent + if ( f->optional && + TableJsonGetRaw( (const uint8_t *) base + f->present_offset, 1 ) == 0 ) + { + continue; + } + if ( !any ) { out.put( '{' ); } + else { out.put( ',' ); } + any = true; + out.line( depth + 1 ); + TableJsonWriteString( out, f->json, (int32_t) strlen( f->json ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, base, f, depth + 1 ) ) { return false; } + } + return true; +} + +// One instance as one object. The writer carries the reader's depth cap +// (§16.2): a pointer chain nests as deep as it is long (§16.7), and a text the +// writer produced past the cap would be a text the reader refuses. +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { return false; } + bool any = false; + if ( !TableJsonWriteFields( out, base, info, depth, any ) ) { return false; } + if ( !any ) + { + out.raw( "{}", 2 ); + return true; + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- reading ---- + +struct TableJsonIn +{ + const char * text; + int64_t size; + int64_t pos; + TableReport * report; + bool bad; // the text is not JSON: the walk stops and keeps what it placed + void * graph; // the pointered read's builder and label map (§16.7); NULL for a fixed table +}; + +inline void TableJsonSpace( TableJsonIn & in ) +{ + while ( in.pos < in.size ) + { + char c = in.text[in.pos]; + if ( c == ' ' || c == '\t' || c == '\n' || c == '\r' ) { in.pos++; continue; } + // comments are not JSON, and a walk that guessed at one would be + // reading a dialect nobody wrote down + if ( c == '/' ) { in.bad = true; } + return; + } +} + +inline char TableJsonPeek( TableJsonIn & in ) +{ + TableJsonSpace( in ); + return in.pos < in.size ? in.text[in.pos] : 0; +} + +// the shape of the value sitting at the cursor, without consuming it +inline char TableJsonValueShape( TableJsonIn & in ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return 'o'; + case '[': return 'a'; + case '"': return 's'; + case 't': case 'f': return 'b'; + case 'n': return 'z'; + case 0: return 0; + default: return 'n'; + } +} + +inline bool TableJsonLiteral( TableJsonIn & in, const char * word ) +{ + int64_t length = (int64_t) strlen( word ); + if ( in.pos + length > in.size || memcmp( in.text + in.pos, word, (size_t) length ) != 0 ) + { + in.bad = true; + return false; + } + in.pos += length; + return true; +} + +// one \uXXXX escape body; -1 when the four hex digits are not there +inline int TableJsonHex4( TableJsonIn & in ) +{ + if ( in.pos + 4 > in.size ) { return -1; } + int value = 0; + for ( int i = 0; i < 4; i++ ) + { + char c = in.text[in.pos + i]; + int digit; + if ( c >= '0' && c <= '9' ) { digit = c - '0'; } + else if ( c >= 'a' && c <= 'f' ) { digit = c - 'a' + 10; } + else if ( c >= 'A' && c <= 'F' ) { digit = c - 'A' + 10; } + else { return -1; } + value = ( value << 4 ) | digit; + } + in.pos += 4; + return value; +} + +inline int32_t TableJsonEncodeUtf8( uint32_t code, char * unit ) +{ + if ( code < 0x80 ) { unit[0] = (char) code; return 1; } + if ( code < 0x800 ) + { + unit[0] = (char) ( 0xc0 | ( code >> 6 ) ); + unit[1] = (char) ( 0x80 | ( code & 0x3f ) ); + return 2; + } + if ( code < 0x10000 ) + { + unit[0] = (char) ( 0xe0 | ( code >> 12 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( code & 0x3f ) ); + return 3; + } + unit[0] = (char) ( 0xf0 | ( code >> 18 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 12 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[3] = (char) ( 0x80 | ( code & 0x3f ) ); + return 4; +} + +// Scan one JSON string into a caller buffer. Bytes are appended ONE CODE +// POINT AT A TIME — an escape's encoding, or a UTF-8 sequence read whole — +// so a string longer than the field is clamped AT A CODE POINT BOUNDARY and +// never cut through a multi-byte character. Clamping is counted, never +// fatal, exactly as it is on the wire (§4). A NULL destination scans past a +// string without keeping it. +inline bool TableJsonScanString( TableJsonIn & in, char * out, int32_t capacity, int32_t * length ) +{ + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + int32_t placed = 0; + bool clamped = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos]; + if ( c == '"' ) { in.pos++; break; } + char unit[4]; + int32_t unit_length = 0; + if ( c == '\\' ) + { + in.pos++; + if ( in.pos >= in.size ) { in.bad = true; return false; } + char escape = in.text[in.pos++]; + switch ( escape ) + { + case '"': unit[0] = '"'; unit_length = 1; break; + case '\\': unit[0] = '\\'; unit_length = 1; break; + case '/': unit[0] = '/'; unit_length = 1; break; + case 'b': unit[0] = '\b'; unit_length = 1; break; + case 'f': unit[0] = '\f'; unit_length = 1; break; + case 'n': unit[0] = '\n'; unit_length = 1; break; + case 'r': unit[0] = '\r'; unit_length = 1; break; + case 't': unit[0] = '\t'; unit_length = 1; break; + case 'u': + { + int high = TableJsonHex4( in ); + if ( high < 0 ) { in.bad = true; return false; } + uint32_t code = (uint32_t) high; + if ( high >= 0xd800 && high <= 0xdbff && in.pos + 2 <= in.size && + in.text[in.pos] == '\\' && in.text[in.pos + 1] == 'u' ) + { + int64_t mark = in.pos; + in.pos += 2; + int low = TableJsonHex4( in ); + if ( low >= 0xdc00 && low <= 0xdfff ) + { + code = 0x10000 + ( ( (uint32_t) high - 0xd800 ) << 10 ) + ( (uint32_t) low - 0xdc00 ); + } + else + { + in.pos = mark; // a lone lead surrogate rides as itself + } + } + // a surrogate half that never found its partner has no + // UTF-8 encoding: encoding it anyway would manufacture + // CESU-8 — invalid UTF-8 — out of input that was valid + // JSON, so it reads as the replacement character + if ( code >= 0xd800 && code <= 0xdfff ) { code = 0xfffd; } + unit_length = TableJsonEncodeUtf8( code, unit ); + break; + } + default: in.bad = true; return false; + } + } + else if ( (unsigned char) c < 0x20 ) + { + in.bad = true; // a raw control character is not a JSON string body + return false; + } + else + { + // a UTF-8 sequence read WHOLE, so the clamp below can only land + // between code points. Only bytes that ACTUALLY look like + // continuations are taken: the wire imposes no encoding (§3), so + // a string may legitimately hold a stray lead byte, and one at + // the end of a text must not swallow the closing quote. + unsigned char lead = (unsigned char) c; + int32_t want = 1; + if ( ( lead & 0xe0 ) == 0xc0 ) { want = 2; } + else if ( ( lead & 0xf0 ) == 0xe0 ) { want = 3; } + else if ( ( lead & 0xf8 ) == 0xf0 ) { want = 4; } + unit[0] = c; + in.pos++; + unit_length = 1; + while ( unit_length < want && in.pos < in.size && + ( (unsigned char) in.text[in.pos] & 0xc0 ) == 0x80 ) + { + unit[unit_length++] = in.text[in.pos++]; + } + } + if ( out == NULL ) + { + placed += unit_length; // measured and not kept: a byte buffer's read sizes its node this way (§2.5) + } + else if ( placed + unit_length <= capacity ) + { + memcpy( out + placed, unit, (size_t) unit_length ); + placed += unit_length; + } + else + { + clamped = true; + } + } + if ( clamped ) { in.report->clamped++; } + if ( length != NULL ) { *length = placed; } + return true; +} + +// the numeric token at the cursor, copied out whole; false = not a number +// Scan one number, to JSON's OWN grammar (RFC 8259 §6) and not to a run of +// number-ish characters: +// +// number = [ "-" ] int [ frac ] [ exp ] +// int = "0" / ( digit1-9 *digit ) +// frac = "." 1*digit +// exp = ( "e" / "E" ) [ "-" / "+" ] 1*digit +// +// Scanning the production is what makes a typo in an authoring file a +// DIAGNOSTIC rather than a value: "1-2" scans as 1 and leaves "-2" where the +// object expects a comma, so the text is malformed — which is what §16.2 +// already promises. A permissive scan would hand "1-2" to a digit loop and +// report a clamp, and a config pipeline would never hear about it. Leading +// "+", leading zeros, ".5" and "3." are not JSON either. +inline bool TableJsonWalkNumber( TableJsonIn & in, bool * integral ) +{ + TableJsonSpace( in ); + bool whole = true; + if ( in.pos < in.size && in.text[in.pos] == '-' ) { in.pos++; } + // int: a lone zero, or a non-zero digit and any digits after it + if ( in.pos >= in.size ) { return false; } + if ( in.text[in.pos] == '0' ) + { + in.pos++; + } + else if ( in.text[in.pos] >= '1' && in.text[in.pos] <= '9' ) + { + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + } + else + { + return false; + } + // frac + if ( in.pos < in.size && in.text[in.pos] == '.' ) + { + in.pos++; + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + // exp + if ( in.pos < in.size && ( in.text[in.pos] == 'e' || in.text[in.pos] == 'E' ) ) + { + in.pos++; + if ( in.pos < in.size && ( in.text[in.pos] == '-' || in.text[in.pos] == '+' ) ) { in.pos++; } + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + *integral = whole; + return true; +} + +// the same production, with the token kept for conversion +inline bool TableJsonScanNumber( TableJsonIn & in, char * token, int32_t capacity, int32_t * length, bool * integral ) +{ + TableJsonSpace( in ); + int64_t start = in.pos; + if ( !TableJsonWalkNumber( in, integral ) ) { return false; } + int64_t count = in.pos - start; + if ( count <= 0 || count >= capacity ) { return false; } + memcpy( token, in.text + start, (size_t) count ); + token[count] = 0; + *length = (int32_t) count; + return true; +} + +// the token's exact double, through the runtime's own converter — which +// speaks the LOCALE's decimal point, so the token crosses back over that +// character on its way in +inline double TableJsonTokenDouble( const char * token, int32_t length, bool single ) +{ + char work[kTableJsonMaxNumber]; + memcpy( work, token, (size_t) length ); + work[length] = 0; + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int32_t i = 0; i < length; i++ ) + { + if ( work[i] == '.' ) { work[i] = point; } + } + } + if ( single ) { return (double) strtof( work, NULL ); } + return strtod( work, NULL ); +} + +// the token's exact integer, parsed digit by digit so no width and no +// locale can move it. Saturation is reported as a clamp, the wire's rule for +// a value outside what the reader can hold (§4). +inline int64_t TableJsonTokenInteger( const char * token, int32_t length, bool is_signed, bool * saturated ) +{ + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) + { + negative = token[i] == '-'; + i++; + } + uint64_t magnitude = 0; + bool over = false; + for ( ; i < length; i++ ) + { + uint64_t digit = (uint64_t) ( token[i] - '0' ); + if ( magnitude > ( UINT64_MAX - digit ) / 10 ) { over = true; break; } + magnitude = magnitude * 10 + digit; + } + if ( !is_signed ) + { + // -0 IS zero, and clamping it would report an event that did not + // happen; only a real negative magnitude is out of range here + if ( negative ) { *saturated = magnitude != 0; return 0; } + if ( over ) { *saturated = true; return (int64_t) UINT64_MAX; } + *saturated = false; + return (int64_t) magnitude; + } + if ( negative ) + { + if ( over || magnitude > ( uint64_t( 1 ) << 63 ) ) { *saturated = true; return INT64_MIN; } + *saturated = false; + if ( magnitude == ( uint64_t( 1 ) << 63 ) ) { return INT64_MIN; } + return -(int64_t) magnitude; + } + if ( over || magnitude > (uint64_t) INT64_MAX ) { *saturated = true; return INT64_MAX; } + *saturated = false; + return (int64_t) magnitude; +} + +// A number token into a wide kind's raw storage (docs/SPEC-TABLES.md §16.2). A +// 128-bit integer takes any token whose VALUE is integral; a fixed field any +// token whose value is EXACTLY representable in its Q I.F — a finer fraction +// is the wrong shape for the field, counted as a kind mismatch and never +// rounded, the rule SPEC.md §4.6 gives a fixed default. A magnitude past 128 +// bits saturates and counts as a clamp, as an int64 field saturates at +// INT64_MAX; the declared range clamps after it, on the RAW scale, as it does +// for every bounded scalar. +// +// The token is normalized to its digits with the decimal point after "point" +// of them. An integer part past 40 digits is above 2^128 whatever the digits +// are, and a value below 10^-40 is finer than 2^-127, the finest fraction any +// F can spell — so outside that band the answer is known without the +// arithmetic, and a token spelling 1e999999999 costs nothing to refuse. +inline bool TableJsonReadWide( TableJsonIn & in, const char * token, int32_t length, void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + int frac = f->frac_bits; + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { negative = token[i] == '-'; i++; } + const char * int_digits = token + i; + int32_t int_len = 0; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { int_len++; i++; } + const char * frac_digits = token + i; + int32_t frac_len = 0; + if ( i < length && token[i] == '.' ) + { + i++; + frac_digits = token + i; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { frac_len++; i++; } + } + int64_t exp = 0; + if ( i < length && ( token[i] == 'e' || token[i] == 'E' ) ) + { + i++; + bool exp_negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { exp_negative = token[i] == '-'; i++; } + while ( i < length && token[i] >= '0' && token[i] <= '9' ) + { + if ( exp < 100000 ) { exp = exp * 10 + ( token[i] - '0' ); } + i++; + } + if ( exp_negative ) { exp = -exp; } + } + // the digits, with the point after "point" of them; leading and trailing + // zeros stripped. digit( k ) reads the k-th of the int and frac runs. + int32_t start = 0, end = int_len + frac_len; + int64_t point = int_len + exp; + while ( start < end && ( start < int_len ? int_digits[start] : frac_digits[start - int_len] ) == '0' ) { start++; point--; } + while ( end > start && ( end - 1 < int_len ? int_digits[end - 1] : frac_digits[end - 1 - int_len] ) == '0' ) { end--; } + + TableJsonWide raw = { 0, 0 }; + bool saturated = false; + TableJsonWide signed_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) >> 1 }; + TableJsonWide signed_min = { 0, uint64_t( 1 ) << 63 }; + TableJsonWide unsigned_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) }; + if ( start == end ) + { + // zero, and -0 IS zero + } + else if ( point > 40 ) + { + saturated = true; + if ( !negative ) { raw = is_signed ? signed_max : unsigned_max; } + else if ( is_signed ) { raw = signed_min; } + } + else if ( point < -40 ) + { + in.report->kind_mismatch++; // finer than any F can spell + return true; + } + else + { + // the fraction FIRST, so an inexact value is the wrong shape whatever + // its magnitude: its digits, with the zeros a negative point puts in + // front, doubled F times; each doubling's carry is the next bit, and + // the value is exact iff nothing is left after the last one + char fd[kTableJsonMaxNumber + 48]; + int32_t fn = 0; + for ( int64_t z = point; z < 0; z++ ) { fd[fn++] = 0; } + for ( int32_t k = (int32_t) ( point > 0 ? point : 0 ) + start; k < end; k++ ) + { + fd[fn++] = (char) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ); + } + TableJsonWide fraction = { 0, 0 }; + for ( int b = 0; b < frac; b++ ) + { + int carry = 0; + for ( int32_t k = fn - 1; k >= 0; k-- ) + { + int d = fd[k] * 2 + carry; + fd[k] = (char) ( d % 10 ); + carry = d / 10; + } + fraction = TableJsonWideShl( fraction, 1 ); + fraction.lo |= (uint64_t) carry; + } + for ( int32_t k = 0; k < fn; k++ ) + { + if ( fd[k] != 0 ) + { + in.report->kind_mismatch++; + return true; + } + } + // then the whole part, saturating past 128 bits + TableJsonWide whole = { 0, 0 }; + for ( int64_t k = start; k < start + point && !saturated; k++ ) + { + uint32_t digit = k < end ? (uint32_t) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ) : 0; + if ( TableJsonWideMulAdd( &whole, 10, digit ) != 0 ) { saturated = true; } + } + if ( !saturated && frac > 0 && !TableJsonWideZero( TableJsonWideShr( whole, 128 - frac ) ) ) { saturated = true; } + if ( !saturated ) + { + raw = TableJsonWideShl( whole, frac ); + raw.lo |= fraction.lo; + raw.hi |= fraction.hi; + } + if ( is_signed ) + { + if ( !saturated && !negative && TableJsonWideNegative( raw ) ) { saturated = true; } + if ( !saturated && negative && TableJsonWideCompare( raw, signed_min, false ) > 0 ) { saturated = true; } + if ( saturated ) { raw = negative ? signed_min : signed_max; } + else if ( negative ) { raw = TableJsonWideNeg( raw ); } + } + else + { + if ( saturated ) { raw = unsigned_max; } + if ( negative && !TableJsonWideZero( raw ) ) { raw.lo = 0; raw.hi = 0; saturated = true; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->wide != NULL ) + { + TableJsonWide lo = { f->wide->lo[0], f->wide->lo[1] }; + TableJsonWide hi = { f->wide->hi[0], f->wide->hi[1] }; + if ( TableJsonWideCompare( raw, lo, is_signed ) < 0 ) { raw = lo; in.report->clamped++; } + else if ( TableJsonWideCompare( raw, hi, is_signed ) > 0 ) { raw = hi; in.report->clamped++; } + } + TableJsonWideStore( storage, f->elem_size, raw ); + return true; +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ); + +inline bool TableJsonSkipContainer( TableJsonIn & in, char close, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; // the opening bracket + bool first = true; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == close ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + if ( close == '}' ) + { + // the key is kept, because a skipped OBJECT may still be a + // pointer's: an `&node` opening it names a node the storage could + // not hold, and the numbering has to survive the drop (§16.7). + // Anywhere but first, the prefix is the reserved key out of place + // — in a pointered unit; a fixed unit skips the value whole. + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( key[0] == '&' && in.graph != NULL ) + { + if ( !first ) { in.report->malformed = true; in.bad = true; return false; } + if ( !TableJsonSkippedAmpersand( in, key, depth ) ) { return false; } + first = false; + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } + } + first = false; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return TableJsonSkipContainer( in, '}', depth ); + case '[': return TableJsonSkipContainer( in, ']', depth ); + case '"': return TableJsonScanString( in, NULL, 0, NULL ); + case 't': return TableJsonLiteral( in, "true" ); + case 'f': return TableJsonLiteral( in, "false" ); + case 'n': return TableJsonLiteral( in, "null" ); + case 0: in.bad = true; return false; + default: + { + // consumed, never converted: skipping needs no buffer, and this + // is the one walk a hostile text drives to the depth cap. It is + // the SAME production the value path scans, so an unknown key + // cannot smuggle past a number a named key would refuse. + bool integral = false; + if ( !TableJsonWalkNumber( in, &integral ) ) { in.bad = true; return false; } + return true; + } + } +} + +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration reads through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ); + +// place one scalar at one storage address +inline bool TableJsonReadScalar( TableJsonIn & in, void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; {} is None, and + // two keys is a text this walk will not guess at + const TableUnionInfo * arms = f->arms(); + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, 0 ); + if ( TableJsonPeek( in ) == '}' ) { in.pos++; return true; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t tag = 0; + for ( int64_t t = 1; t <= f->enum_max; t++ ) + { + if ( strcmp( f->enum_name( (uint64_t) t ), key ) == 0 ) { tag = t; break; } + } + if ( tag == 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + void * payload = (uint8_t *) storage + arms->arms[tag].offset; + const TableFieldInfo * arm = arms->arms[tag].field; + bool placed = true; + if ( arm != NULL ) + { + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2). A value of + // the wrong shape for that row is a KIND MISMATCH: the union + // reads None, the event is counted, and the enclosing object + // continues — the rule a FIELD's value lives under, one key + // down. A pointer arm's null is a null pointer, not a shape + // error, exactly as a pointer field's is (§16.7). + char got = TableJsonValueShape( in ); + if ( arm->kind == 17 && !arm->is_array && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + memset( payload, 0, (size_t) arms->arms[tag].size ); + } + else if ( got != TableJsonShape( arm ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( arm->kind == 17 && !arm->is_array ) + { + // A POINTER ARM'S VALUE IS THE POINTEE IN PLACE, or a + // node reference to one (§16.7) — the read a pointer + // FIELD takes, which is not the scalar walk + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadPointer( in, payload, arm, depth + 1 ) ) { return false; } + } + else + { + // SELECTION ZERO-ESTABLISHES THE ARM (SPEC §5): an arm + // takes no specified default, so zero is the establish + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadField( in, storage, arm, depth + 1 ) ) { return false; } + } + } + else if ( arms->arms[tag].table != NULL ) + { + if ( TableJsonValueShape( in ) != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else + { + arms->arms[tag].table->reset( payload ); + if ( !TableJsonReadTable( in, payload, arms->arms[tag].table, depth + 1 ) ) { return false; } + } + } + else + { + // A PAYLOAD-FREE ARM'S VALUE IS null (§2.6): the arm name + // selects it and there is nothing to place + if ( TableJsonValueShape( in ) != 'z' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( !TableJsonLiteral( in, "null" ) ) + { + return false; + } + } + if ( placed ) + { + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, (uint64_t) tag ); + } + } + char c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; // a second key: a one-of with two arms is not a value + return false; + } + if ( f->kind == 13 ) + { + f->table->reset( storage ); + return TableJsonReadTable( in, storage, f->table, depth + 1 ); + } + if ( TableJsonIsEnum( f ) ) + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + for ( int64_t v = 0; v <= f->enum_max; v++ ) + { + if ( strcmp( f->enum_name( (uint64_t) v ), name ) == 0 ) + { + TableJsonSetRaw( storage, f->elem_size, (uint64_t) v ); + return true; + } + } + // a name this build cannot name reads as None and counts as unknown, + // exactly as an unknown variant id does on the wire (§4) + TableJsonSetRaw( storage, f->elem_size, 0 ); + in.report->unknown++; + return true; + } + if ( TableJsonIsFlags( f ) ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + uint64_t bits = 0; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( c != '"' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + bool found = false; + for ( int64_t bit = 0; bit <= f->enum_max; bit++ ) + { + if ( strcmp( f->enum_name( (uint64_t) bit ), name ) == 0 ) + { + bits |= uint64_t( 1 ) << bit; + found = true; + break; + } + } + if ( !found ) { in.report->unknown++; } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + TableJsonSetRaw( storage, f->elem_size, bits ); + return true; + } + if ( f->kind == 1 ) + { + char c = TableJsonPeek( in ); + if ( c == 't' ) { if ( !TableJsonLiteral( in, "true" ) ) { return false; } TableJsonSetRaw( storage, f->elem_size, 1 ); return true; } + if ( !TableJsonLiteral( in, "false" ) ) { return false; } + TableJsonSetRaw( storage, f->elem_size, 0 ); + return true; + } + char token[kTableJsonMaxNumber]; + int32_t length = 0; + bool integral = false; + if ( !TableJsonScanNumber( in, token, kTableJsonMaxNumber, &length, &integral ) ) + { + in.bad = true; + return false; + } + if ( TableJsonKindWide( f->kind ) ) + { + return TableJsonReadWide( in, token, length, storage, f ); + } + if ( f->kind == 10 || f->kind == 11 ) + { + bool single = f->kind == 10; + double value = TableJsonTokenDouble( token, length, single ); + // A magnitude the field's format cannot hold is the WRONG SHAPE for + // the kind, and it never reaches storage: 1e400 is not a float64 and + // 1e300 is not a float32. Storing the infinity the conversion + // produced would leave an instance this walk called CLEAN that + // ToJsonMeasure then refuses forever (a non-finite float has no JSON + // spelling), and §16.1's one invariant is that a text which reads + // clean writes back. + if ( !TableJsonFinite( value ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( f->has_range ) + { + if ( value < f->range_min ) { value = f->range_min; in.report->clamped++; } + else if ( value > f->range_max ) { value = f->range_max; in.report->clamped++; } + } + if ( single ) + { + float narrow = (float) value; + if ( !TableJsonFinite( (double) narrow ) ) + { + in.report->kind_mismatch++; + return true; + } + memcpy( storage, &narrow, sizeof( narrow ) ); + } + else + { + memcpy( storage, &value, sizeof( value ) ); + } + return true; + } + // JSON HAS ONE NUMBER TYPE. 2.0 IS the integer 2 and 1e3 IS 1000, and a + // library that round-trips numbers through a double emits them that way — + // this walker's own float writer emits 1e+21. So an integer field takes + // any number whose VALUE is integral, however it was spelled; only a + // genuinely fractional value is the wrong shape for it. + bool is_signed = f->kind >= 2 && f->kind <= 5; + bool saturated = false; + int64_t value = 0; + if ( integral ) + { + value = TableJsonTokenInteger( token, length, is_signed, &saturated ); + } + else + { + double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( is_signed ) + { + if ( d >= 9223372036854775808.0 ) { value = INT64_MAX; saturated = true; } + else if ( d < -9223372036854775808.0 ) { value = INT64_MIN; saturated = true; } + else if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) d; } + } + else + { + if ( d < 0.0 ) + { + // a negative for an unsigned field clamps to zero, as the + // exact digit path already does + if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + value = 0; + saturated = true; + } + else if ( d >= 18446744073709551616.0 ) { value = (int64_t) UINT64_MAX; saturated = true; } + else if ( d != (double) (uint64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) (uint64_t) d; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->has_range ) + { + if ( (double) value < f->range_min ) { value = (int64_t) f->range_min; in.report->clamped++; } + else if ( (double) value > f->range_max ) { value = (int64_t) f->range_max; in.report->clamped++; } + } + // the field's own storage width is the last bound: a value past it + // clamps rather than wrapping, which is what the wire does too + if ( f->elem_size < 8 ) + { + if ( is_signed ) + { + int64_t high = ( int64_t( 1 ) << ( f->elem_size * 8 - 1 ) ) - 1; + int64_t low = -high - 1; + if ( value > high ) { value = high; in.report->clamped++; } + else if ( value < low ) { value = low; in.report->clamped++; } + } + else + { + uint64_t high = ( uint64_t( 1 ) << ( f->elem_size * 8 ) ) - 1; + if ( value < 0 ) { value = 0; in.report->clamped++; } + else if ( (uint64_t) value > high ) { value = (int64_t) high; in.report->clamped++; } + } + } + // at eight bytes the storage IS the parser's width, and an unsigned value + // past INT64_MAX rides here as a negative int64 by design — the token + // parser already turned a NEGATIVE token for an unsigned field into a + // clamped zero, so there is nothing left to bound. + TableJsonSetRaw( storage, f->elem_size, (uint64_t) value ); + return true; +} + +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ) +{ + uint8_t * storage = (uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonReadMap( in, (void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + + if ( f->kind == 12 ) + { + int32_t length = 0; + if ( !TableJsonScanString( in, (char *) storage, f->array_bound, &length ) ) { return false; } + storage[length] = 0; + TableJsonSetCount( base, f, length ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + // base64 decodes STRAIGHT INTO the field's storage, six bits at a + // time — no window, no temporary, so a bytes(N) of any declared + // extent reads the same way. A base64 body carries no escapes, so a + // backslash in one is simply not an alphabet character. + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + memset( storage, 0, (size_t) f->array_bound ); + TableJsonSetCount( base, f, 0 ); + const char * alphabet = TableJsonBase64Alphabet(); + int32_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + bool clamped = false; + bool malformed = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + const char * at = c != 0 ? strchr( alphabet, c ) : NULL; + if ( at == NULL ) { malformed = true; continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( at - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < f->array_bound ) + { + storage[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); + } + else + { + clamped = true; + } + } + } + if ( malformed ) + { + // a body that is not base64 is the wrong shape for the kind: the + // field keeps its default and the event is counted + in.report->kind_mismatch++; + return true; + } + if ( clamped ) { in.report->clamped++; } + TableJsonSetCount( base, f, placed ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + // every slot back to its declared defaults first, so a key the text + // omits keeps them and a repeated field key cannot leave an earlier + // occurrence's slots standing + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + void * slot = storage + (int64_t) i * f->elem_size; + if ( f->kind == 13 ) { f->table->reset( slot ); } + else { memset( slot, 0, (size_t) f->elem_size ); } + } + char shape = TableJsonElementShape( f ); + // A KEYED OBJECT'S KEYS ARE KEYS: a variant named twice is a duplicate + // key like any other, last-wins and counted (§16.2). Tracked the way + // a table's own field keys are — a bounded, allocation-free bitmask; + // a vocabulary wider than this still reads, its repeats simply stop + // being counted. + uint64_t seen[8] = {}; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t slot = -1; + for ( int64_t v = 0; v < f->array_bound; v++ ) + { + // nothing is stored for None, so "None" finds no slot and is + // an unknown key like any other name this reader cannot place + if ( !TableJsonKeyedSlotValid( f, v ) ) { continue; } + if ( strcmp( f->key_name( TableJsonKeyedSlotKey( v ) ), key ) == 0 ) { slot = v; break; } + } + if ( slot >= 0 && slot < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( slot & 63 ); + if ( ( seen[slot >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[slot >> 6] |= bit; + } + if ( slot < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; + } + if ( f->is_array ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + // LAST WINS has to be true of a repeated ARRAY key too, and it is + // wire-visible: a fixed array writes every slot, so a second, shorter + // occurrence overlaying a prefix would leave the first occurrence's + // tail standing. The field goes back to its declared defaults before + // this occurrence's elements are placed — the re-establishment a nested + // table and a union arm already get. A table element's defaults are + // its own (the reset hook); every other element kind's storage + // default is zero, which is what the generated array declares. + if ( f->kind == 13 ) + { + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + f->table->reset( storage + (int64_t) i * f->elem_size ); + } + } + else + { + memset( storage, 0, (size_t) f->array_bound * (size_t) f->elem_size ); + } + TableJsonSetCount( base, f, 0 ); + int32_t placed = 0; + char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( placed >= f->array_bound ) + { + // more elements than the reader's bound: the bounded prefix + // is kept and the excess counts, the wire's rule (§4) + in.report->clamped++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( f->kind == 17 ) + { + // an element of an ARRAY OF POINTERS (§2.1): null is a null slot, an + // object is the pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( storage + (int64_t) placed * f->elem_size, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed++; + } + else + { + if ( !TableJsonReadScalar( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + // a fixed array's tail keeps the defaults the prefill left there, + // exactly as a short wire count does + TableJsonSetCount( base, f, placed ); + return true; + } + return TableJsonReadScalar( in, storage, f, depth ); +} + +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ); + +// ONE table object: keys are field keys, unknown ones are skipped and +// counted, a repeated key is last-wins and counted. The instance is already +// at its declared defaults when this is entered, so a key the text never +// mentions keeps the default an absent field takes on the wire (§4). +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + return TableJsonReadTableKeys( in, base, info, depth, NULL ); +} + +// The keys of an object whose brace is already consumed. A pointer's object +// opens the same way a table's does, but its FIRST key may be `&node` (§16.7) +// and the adapter that reads it has to scan the key to know — so it hands the +// key it scanned in as `first_key`, with the colon consumed, and this places +// it before scanning the rest. +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ) +{ + // duplicate tracking, bounded and allocation-free: a table with more + // fields than this still reads, its repeats simply stop being counted + uint64_t seen[8] = {}; + for ( ;; ) + { + char key[kTableJsonMaxKey]; + char c = 0; + if ( first_key != NULL ) + { + memcpy( key, first_key, strlen( first_key ) + 1 ); // scanned into a buffer this size by the caller + first_key = NULL; + } + else + { + c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + } + int32_t index = -1; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + if ( strcmp( info->fields[i].json, key ) == 0 ) { index = i; break; } + } + if ( key[0] == '&' ) + { + // THE AMPERSAND PREFIX IS RESERVED TO THE FORM (docs/SPEC-TABLES.md + // §16.7). No declaration may take a key beginning with it, so this + // is never a field this build lacks — it is the sharing construct + // somewhere it cannot stand: `&node` is the FIRST key of a pointer's + // object and nothing else, and the adapter that reads a pointer + // has consumed it before these keys are read. MALFORMED, refused + // and counted; never counted as unknown, never skipped. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( index < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + const TableFieldInfo * f = &info->fields[index]; + if ( index < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( index & 63 ); + if ( ( seen[index >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[index >> 6] |= bit; + } + // PRESENCE OF THE KEY IS THE PRESENCE (§16.2): reaching this line + // is the key being present, so an optional is set present + // whatever its value — with one exception the page names: a JSON + // null, which reads as ABSENT rather than as a value. + char got = TableJsonValueShape( in ); + if ( f->kind == 17 && !f->is_array ) + { + // a pointer: null is a null pointer, an object is the pointee + // in place or an `&node` reference to one (§16.7), a string is + // a BYTE BUFFER's bytes (§2.5), and anything else is the wrong + // shape for the kind + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) base + f->offset, f->elem_size, 0 ); + } + else if ( got != TableJsonShape( f ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) base + f->offset, f, depth ) ) + { + return false; + } + } + else if ( f->optional && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + // absent, and back at its defaults: a repeated key whose last + // occurrence is null must not leave an earlier value standing + if ( f->table != NULL ) { f->table->reset( (uint8_t *) base + f->offset ); } + else { memset( (uint8_t *) base + f->offset, 0, (size_t) f->elem_size ); } + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 0 ); + } + else + { + if ( got != TableJsonShape( f ) ) + { + // the wrong JSON type for the kind: skipped, never coerced + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, base, f, depth ) ) + { + return false; + } + if ( f->optional ) + { + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 1 ); + } + } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +// ---- the two entry points the per-table wrappers name ---- + +inline bool TableJsonRead( void * value, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = NULL; + info->reset( value ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, value, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +inline int64_t TableJsonWrite( const void * value, const TableTypeInfo * info, char * buffer, int64_t capacity ) +{ + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = NULL; + if ( !TableJsonWriteValue( out, value, info, 0 ) ) { return -1; } + // THE CANONICAL TEXT ENDS WITH EXACTLY ONE NEWLINE (docs/SPEC-TABLES.md + // §16.1). Every writer emits it — this walk, the C# walk and + // "schema unpack" — and every reader accepts a text with or without one, + // because the trailing whitespace a read already skips is what makes the + // two the same text. It is a byte of the FORM rather than a file + // convention: a text that is written to a file, pasted into a diff and + // handed back through a pipe has to be one text in all three places, and a + // buffer whose last byte is a closing brace is the one shape that is not. + out.put( '\n' ); + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json walk: end ---- + +// ---- json graph walk: begin ---- +// +// THE VARIABLE CLASS's half of the text form (docs/SPEC-TABLES.md §16.7). The +// walk above places every kind but one; this defines the three adapters it +// calls for that one, and the two entry points a pointered table's wrappers +// name. The text is the fixed class's — a pointee is an object in place — and a +// node named more than once carries `&node`: defined once, with its fields, +// and referenced after by `{ "&node": N }` alone. + +// ---- the identity map ---- +// +// ONE map shape serves both directions. Writing keys it by a node's ADDRESS and +// counts the slots that name the node, so the second pass knows at a node's +// first occurrence whether it will be named again; reading keys it by the +// text's own label and answers the node it defined. Open addressing, a +// multiply-shift hash and quadrupling growth — TablePackMap's shape (§6.2), on +// the same terms: proportional to nodes, never to bytes, on the authoring +// side, and released before the call returns. + +struct TableJsonGraphEntry +{ + uint64_t key; // a node's address (write) or a label (read); 0 is an empty slot + int64_t count; // write: how many slots name this node + int64_t label; // write: the `&node` label assigned at its first write, 0 until then + uint8_t open; // the descent is still open: a reference here is a cycle (write), a self-reference (read) + uint32_t node; // read: the node's arena offset; 0 for a definition the reader dropped + const TableTypeInfo * type; // read: the node's table; NULL for a dropped one +}; + +struct TableJsonGraphMap +{ + TableJsonGraphEntry * entries; + int64_t capacity; // a power of two, or zero while empty + int64_t count; + TableAllocator allocator; // the caller's pair (§6.5): the builder's on read, the one handed to ToJson on write +}; + +inline void TableJsonGraphMapInit( TableJsonGraphMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TableJsonGraphMapShutdown( TableJsonGraphMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TableJsonGraphMapInit( map, map.allocator ); +} + +inline int64_t TableJsonGraphMapSlot( const TableJsonGraphMap & map, uint64_t key ) +{ + uint64_t hash = key * 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != 0 && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TableJsonGraphEntry * TableJsonGraphMapFind( TableJsonGraphMap & map, uint64_t key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +inline bool TableJsonGraphMapGrow( TableJsonGraphMap & map ) +{ + TableJsonGraphMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 64; + grown.count = 0; + grown.entries = (TableJsonGraphEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TableJsonGraphEntry ) ); // zeroed, by the pair's contract + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == 0 ) { continue; } + grown.entries[ TableJsonGraphMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// the entry for a key, made if it was not there; `taken` says which. NULL is the +// allocator refusing, and the walk refuses with it. +inline TableJsonGraphEntry * TableJsonGraphMapReach( TableJsonGraphMap & map, uint64_t key, bool & taken ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TableJsonGraphMapGrow( map ) ) { return NULL; } + } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + taken = entry->key != key; + if ( taken ) + { + entry->key = key; + map.count++; + } + return entry; +} + +// ---- reading: into a builder ---- + +struct TableJsonGraphIn +{ + TableWorker * worker; // where every node comes from + TableJsonGraphMap labels; // a label -> the node it defined +}; + +// `&node`'s value, the LABEL: a positive integer spelled as one — digits, no sign, no +// fraction, no exponent, no leading zero (§16.7). Anything else is malformed. +inline bool TableJsonScanLabel( TableJsonIn & in, uint64_t & label ) +{ + TableJsonSpace( in ); + if ( in.pos >= in.size || in.text[in.pos] < '1' || in.text[in.pos] > '9' ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + uint64_t value = 0; + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) + { + uint64_t digit = (uint64_t) ( in.text[in.pos] - '0' ); + if ( value > ( UINT64_MAX - digit ) / 10 ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + value = value * 10 + digit; + in.pos++; + } + label = value; + return true; +} + +// A BYTE BUFFER's text (docs/SPEC-TABLES.md §2.5, §16.2): a string. For a +// *string the string's bytes become the blob; for a *bytes the string is base64 +// and its decoded bytes do. The blob is allocated at EXACTLY the decoded +// length — the string is scanned once without keeping it to learn the length, +// and once into the node — so a blob of any size reads with no window and no +// bound to clamp against. A *bytes body that is not base64 is the wrong shape +// for the kind: the reference stays null and the event is counted. +inline bool TableJsonReadBlob( TableJsonIn & in, void * slot, const TableFieldInfo * f ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + TableRef * ref = (TableRef *) slot; + ref->value = 0; + if ( strcmp( f->type_name, "string" ) == 0 ) + { + const int64_t mark = in.pos; + int32_t length = 0; + if ( !TableJsonScanString( in, NULL, 0, &length ) ) { return false; } + in.pos = mark; + char * data = TableStringEmplace( *graph->worker, *ref, NULL, (int64_t) length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int32_t placed = 0; + return TableJsonScanString( in, data, length, &placed ); + } + // base64: the alphabet characters decide the length, six bits apiece + const char * alphabet = TableJsonBase64Alphabet(); + const int64_t mark = in.pos + 1; + int64_t symbols = 0; + bool malformed = false; + in.pos++; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + if ( c == 0 || strchr( alphabet, c ) == NULL ) { malformed = true; continue; } + symbols++; + } + if ( malformed ) + { + in.report->kind_mismatch++; + return true; + } + const int64_t length = ( symbols * 6 ) / 8; + uint8_t * data = TableBytesEmplace( *graph->worker, *ref, length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int64_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + for ( int64_t at = mark; ; at++ ) + { + char c = in.text[at]; + if ( c == '"' ) { break; } + const char * symbol = c != '=' ? strchr( alphabet, c ) : NULL; + if ( symbol == NULL ) { continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( symbol - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < length ) { data[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); } + } + } + return true; +} + +// A pointer's object. Its FIRST key decides what it is: `&node` naming a label not +// yet defined, with fields after it, is a DEFINITION; `&node` naming one already +// defined, alone, is a REFERENCE; any other key is a node named once, its +// object in place. The node comes from the +// builder's arena, and the slot holds its arena offset (§6.3). A pointer whose +// target is a BYTE BUFFER — no table — takes a string instead (§2.5). +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( f->table == NULL ) { return TableJsonReadBlob( in, slot, f ); } + // the pointee nests one level down, exactly as a by-value table does, and + // takes the same cap: a chain nests as deep as it is long (§16.7) + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + char c = TableJsonPeek( in ); + if ( c == '}' ) + { + // an empty object: a node at its defaults, named once + in.pos++; + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + return true; + } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( strcmp( key, "&node" ) != 0 ) + { + // a node named once: the pointee's object in place, and this key is + // its first field — unless it is the reserved prefix under a spelling + // this form does not have, which ReadTableKeys refuses + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return TableJsonReadTableKeys( in, node, f->table, depth + 1, key ); + } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->labels, label, taken ); + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + // ONE SPELLING, and what follows the label says which half it is: fields + // after a label the text has not defined DEFINE it, and a label alone that + // the text has defined REFERS to it. The other two are malformed — a label + // alone that the text never defined, which would otherwise read as a default + // node under a silent report, and a field after a label already defined, + // which would be a second definition. That is what keeps a typo loud. + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + bool bare = c == '}'; + if ( bare == taken ) { in.report->malformed = true; in.bad = true; return false; } + if ( bare ) + { + // A REFERENCE. A label is defined when its object CLOSES, so a + // reference met inside its own definition — at any depth of by-value + // nesting — names a node whose descent is still open: the cycle the + // wire refuses (§3.1), refused here where it is written. A definition + // the reader dropped names no node, so the slot stays null with + // nothing more counted — the drop was counted where it happened. A + // node of another table than the slot declares is a kind mismatch, as + // on the wire. + in.pos++; + if ( entry->open != 0 ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + if ( entry->type == NULL ) + { + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + if ( entry->type != f->table ) + { + memcpy( slot, &ref, sizeof( ref ) ); + in.report->kind_mismatch++; + return true; + } + ref.value = (int64_t) entry->node; + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + // A DEFINITION: the node is allocated, the label is its, and the keys after + // `&node` are its fields. The entry is OPEN until the object closes, so a + // reference to the label from inside the node's own fields is refused as + // the cycle it is; the node and its table are filled in at the close. + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + entry->open = 1; + if ( !TableJsonReadTableKeys( in, node, f->table, depth + 1, NULL ) ) { return false; } + entry = TableJsonGraphMapFind( graph->labels, label ); // the map may have grown under the descent + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + memcpy( &ref, slot, sizeof( ref ) ); + entry->node = (uint32_t) ref.value; + entry->type = f->table; + entry->open = 0; + return true; +} + +// An `&`-prefixed key opening an object the walk is SKIPPING — a value past an +// array's bound, an unknown key's value, a value of the wrong shape. A +// definition in there still takes its label, so the numbering survives whatever +// the storage could not hold (§16.7): the label is registered with no node, and a +// reference to it reads null. Any other prefixed key is the reserved prefix +// out of place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL || strcmp( key, "&node" ) != 0 ) { in.report->malformed = true; in.bad = true; return false; } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + if ( TableJsonGraphMapReach( graph->labels, label, taken ) == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return true; // a fresh entry is node 0, type NULL: a definition with no node +} + +// ---- writing: from a region's const root ---- + +struct TableJsonGraphOut +{ + TableJsonGraphMap nodes; // a node's address -> how many slots name it, and its `&node` once assigned + bool counting; // PASS ONE: count the references, refuse a cycle, emit nothing + int64_t next_label; +}; + +// The node a slot names: null as `null`, a node named once as its object in +// place, and a node named more than once under the construct. Which of the +// last two it is was learned in pass one; pass two spells it. +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphOut * graph = (TableJsonGraphOut *) out.graph; + if ( graph == NULL ) { return false; } + const void * node = f->resolve( slot ); + if ( node == NULL ) + { + out.raw( "null", 4 ); + return true; + } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->nodes, (uint64_t) (uintptr_t) node, taken ); + if ( entry == NULL ) { return false; } + if ( f->table == NULL ) + { + // A BYTE BUFFER (§2.5, §16.7): its text is a string, which has no + // first key to carry `&node`, so a blob named from more than one + // slot has no spelling this form can carry and the graph is refused — + // as a shared node with nothing to write is. A blob named once is its + // bytes in place: base64 for a *bytes, the string itself for a *string. + if ( graph->counting ) { entry->count++; return true; } + if ( entry->count > 1 ) { return false; } + const TableBlob * blob = (const TableBlob *) node; + if ( blob->length > (uint32_t) 0x7fffffff ) { return false; } + if ( strcmp( f->type_name, "string" ) == 0 ) { TableJsonWriteString( out, (const char *) ( blob + 1 ), (int32_t) blob->length ); } + else { TableJsonWriteBase64( out, (const uint8_t *) ( blob + 1 ), (int32_t) blob->length ); } + return true; + } + if ( graph->counting ) + { + // PASS ONE: one visit per node, every slot that names it counted, and + // a reference to a node whose descent is still open is a cycle — + // refused here as the wire refuses it (§3.1) + entry->count++; + if ( !taken ) { return entry->open == 0; } + entry->open = 1; + if ( !TableJsonWriteValue( out, node, f->table, depth ) ) { return false; } + entry = TableJsonGraphMapFind( graph->nodes, (uint64_t) (uintptr_t) node ); // the map may have grown under the descent + if ( entry == NULL ) { return false; } + entry->open = 0; + return true; + } + // PASS TWO: a node named once is its object in place; a node named more + // than once is DEFINED at its first occurrence — `&node` first, then its + // fields — and REFERENCED by `&node` alone after that, spelled the same way at + // every site. Labels run from 1 in first-write order and are the text's own, + // so a stray number in a hand-edited text is most often one never defined. + if ( entry->count <= 1 ) + { + return TableJsonWriteValue( out, node, f->table, depth ); + } + if ( depth > kTableJsonMaxDepth ) { return false; } + if ( entry->label != 0 ) + { + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + out.line( depth ); + out.put( '}' ); + return true; + } + entry->label = ++graph->next_label; + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + bool any = true; + int64_t before = out.offset; + if ( !TableJsonWriteFields( out, node, f->table, depth, any ) ) { return false; } + // a definition carries at least one field, because a label alone is a + // reference: a shared node with nothing to write has no definition this + // form can spell, and the writer refuses it as it refuses any value it + // cannot spell (§16.3) + if ( out.offset == before ) { return false; } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- the two entry points a pointered table's wrappers name ---- + +// The text into the builder's root. Every node the text names is allocated in +// the builder's arena through the field's own Emplace; the label map is the +// walk's, released before this returns. The root itself takes no label — nothing +// may name it (§16.7) — so an `&node` at the root is refused like any other key +// of the prefix. +inline bool TableJsonReadGraph( TableWorker & worker, void * root, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + if ( worker.arena == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } + TableJsonGraphIn graph; + graph.worker = &worker; + TableJsonGraphMapInit( graph.labels, worker.arena->allocator ); + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = &graph; + info->reset( root ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, root, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + TableJsonGraphMapShutdown( graph.labels ); + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +// The text of a region's const root: measured when the buffer is NULL, written +// when it is not, over one code path. Two passes over one walk — the first +// counts how many slots name each node and refuses a cycle, the second writes +// — so a node's first occurrence knows whether it will be named again. The +// ROOT's entry is open for the whole first pass, so a reference back at it is +// the cycle it is (§3.1), and it takes no label. +inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * info, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + if ( root == NULL ) { return -1; } + TableJsonGraphOut graph; + TableJsonGraphMapInit( graph.nodes, allocator ); + graph.counting = true; + graph.next_label = 0; + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph.nodes, (uint64_t) (uintptr_t) root, taken ); + if ( entry == NULL ) { TableJsonGraphMapShutdown( graph.nodes ); return -1; } + entry->open = 1; + TableJsonOut count; + count.buffer = NULL; + count.capacity = 0; + count.offset = 0; + count.overflow = false; + count.graph = &graph; + bool ok = TableJsonWriteValue( count, root, info, 0 ); + graph.counting = false; + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = &graph; + if ( ok ) { ok = TableJsonWriteValue( out, root, info, 0 ); } + TableJsonGraphMapShutdown( graph.nodes ); + if ( !ok ) { return -1; } + out.put( '\n' ); // the canonical text ends with exactly one newline (§16.1) + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json graph walk: end ---- + +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + +// ---- json map walk: begin ---- + +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} + +// the entry's two rows: fields[0] IS the key and fields[1] IS the value, which +// is what makes a user's own table of pairs the same bytes (§2.8) +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } + +inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } +inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } + +// AN INTEGER KEY IS THE INTEGER'S DECIMAL SPELLING, QUOTED, because a JSON +// object's keys are strings. Written digit by digit so no locale can move it. +inline void TableJsonWriteMapIntegerKey( TableJsonOut & out, const void * storage, const TableFieldInfo * key ) +{ + uint64_t magnitude = 0; + bool negative = false; + if ( TableJsonMapKeySigned( key ) ) + { + int64_t value = 0; + switch ( key->kind ) + { + case 2: value = (int64_t) *(const int8_t *) storage; break; + case 3: value = (int64_t) *(const int16_t *) storage; break; + case 4: value = (int64_t) *(const int32_t *) storage; break; + default: value = *(const int64_t *) storage; break; + } + negative = value < 0; + magnitude = negative ? ( ~(uint64_t) value ) + 1 : (uint64_t) value; + } + else + { + switch ( key->kind ) + { + case 6: magnitude = (uint64_t) *(const uint8_t *) storage; break; + case 7: magnitude = (uint64_t) *(const uint16_t *) storage; break; + case 8: magnitude = (uint64_t) *(const uint32_t *) storage; break; + default: magnitude = *(const uint64_t *) storage; break; + } + } + char digits[24]; + int32_t at = (int32_t) sizeof( digits ); + do { digits[--at] = (char) ( '0' + ( magnitude % 10 ) ); magnitude /= 10; } while ( magnitude != 0 ); + if ( negative ) { digits[--at] = '-'; } + TableJsonWriteString( out, digits + at, (int32_t) sizeof( digits ) - at ); +} + +inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const TableFieldInfo * key ) +{ + const uint8_t * storage = (const uint8_t *) entry + key->offset; + if ( TableJsonMapKeyIsString( key ) ) + { + // A STRING KEY IS THE STRING (§2.8): every JSON key of a map object is + // a KEY OF THE MAP and none is a field key, so the `&` prefix §16.7 + // reserves for field keys is ordinary data here. + TableJsonWriteString( out, (const char *) storage, *(const int32_t *) ( (const uint8_t *) entry + key->count_offset ) ); + return; + } + TableJsonWriteMapIntegerKey( out, (const void *) storage, key ); +} + +// ToJson WRITES ENTRIES IN ASCENDING KEY ORDER, so unpack then pack is +// byte-stable and a diff of two texts is a diff of two maps (§2.8, §17.2). +// A region holds them in that order already, so this is the array in place. +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "{}", 2 ); return true; } + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); + out.put( '{' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); + TableJsonWriteMapKey( out, entry, key ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// AN INTEGER KEY IS READ BY §16.2's INTEGER RULE AND BY NOTHING ELSE, so +// "2.0" and "1e3" are the integers 2 and 1000 and "-0" is zero. The token is +// walked as a JSON number over its own bytes; a token that rule calls +// malformed makes the KEY malformed, and a genuinely fractional value, or one +// outside the key kind's range, is kind_mismatch for that entry. +inline bool TableJsonMapKeyValue( const char * token, int32_t length, const TableFieldInfo * key, + int64_t & value, bool & fits ) +{ + fits = false; + TableReport scratch; + TableJsonIn probe = { token, (int64_t) length, 0, &scratch, false, NULL }; + bool integral = false; + if ( !TableJsonWalkNumber( probe, &integral ) ) { return false; } + if ( probe.pos != (int64_t) length ) { return false; } // trailing bytes: not a number + if ( !integral ) + { + const double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) { return true; } // a value no key kind holds + const double whole = d < 0 ? -d : d; + if ( whole != (double) (int64_t) whole ) { return true; } // genuinely fractional + } + bool saturated = false; + const bool is_signed = TableJsonMapKeySigned( key ); + value = integral ? TableJsonTokenInteger( token, length, is_signed, &saturated ) + : (int64_t) TableJsonTokenDouble( token, length, false ); + if ( saturated ) { return true; } // outside every width: kind_mismatch, never clamped + switch ( key->kind ) + { + case 2: fits = value >= -128 && value <= 127; break; + case 3: fits = value >= -32768 && value <= 32767; break; + case 4: fits = value >= -2147483647 - 1 && value <= 2147483647; break; + case 5: fits = true; break; + case 6: fits = value >= 0 && value <= 255; break; + case 7: fits = value >= 0 && value <= 65535; break; + case 8: fits = value >= 0 && (uint64_t) value <= 4294967295ull; break; + default: fits = integral; break; // uint64: the token's own magnitude + } + return true; +} + +// FromJson READS KEYS IN WHATEVER ORDER THE TEXT GIVES THEM. A repeated key is +// last-wins and counted duplicate, the object rule (§16.2) applied inside the +// map. An empty object is an empty map, and null is kind_mismatch. +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const char shape = TableJsonShape( value ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char token[kTableJsonMaxKey]; + int32_t token_length = 0; + if ( !TableJsonScanString( in, token, kTableJsonMaxKey - 1, &token_length ) ) { return false; } + token[token_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t key_value = 0; + bool place = true; + if ( !TableJsonMapKeyIsString( key ) ) + { + bool fits = false; + if ( !TableJsonMapKeyValue( token, token_length, key, key_value, fits ) ) + { + // A MALFORMED KEY STOPS THE READ where §16.1's rule stops it, + // with the instance holding what was placed before the stop. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( !fits ) { in.report->kind_mismatch++; place = false; } + } + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; + if ( place && entry == NULL ) + { + // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the + // wire's rule, because a clamped key is a merged entry (§2.8). + in.report->clamped++; + } + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) + { + in.report->duplicate++; // last-wins, the object rule inside the map + } + const char got = TableJsonValueShape( in ); + if ( entry == NULL ) + { + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( value->kind == 17 && !value->is_array ) + { + // A POINTER VALUE IS SHARED EXACTLY AS A POINTER FIELD IS (§2.8): + // null is a null slot, an object is the pointee in place or an + // &node reference to one (§16.7), anything else is the wrong shape — + // the same three the field-key loop gives a pointer field, because + // an entry's value IS a field line. + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) entry + value->offset, value->elem_size, 0 ); + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) entry + value->offset, value, depth + 1 ) ) + { + return false; + } + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, entry, value, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json map walk: end ---- + +// ---- json list walk: begin ---- + +// an unbounded array is the out-of-line array that is not a map (§8.1) +inline bool TableJsonIsList( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && !TableJsonIsMap( f ); +} + +// ToJson WRITES THE ELEMENTS IN INDEX ORDER, which is the only order there is, +// so unpack then pack is byte-stable without a rule of its own (§2.9, §17.2). +// A region holds the array in place, so this steps it at the descriptor's pitch. +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + const uint8_t * elements = TableJsonExtentElements( slot ); + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const uint8_t * element = elements + (int64_t) i * f->elem_size; + if ( f->kind == 17 ) + { + // a []*T's elements take the pointer row (§16.7): the pointee's + // object in place, null, or `&node` for a shared one + if ( !TableJsonWritePointer( out, element, f, depth + 1 ) ) { return false; } + } + else if ( !TableJsonWriteScalar( out, element, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; +} + +// FromJson READS EVERY ELEMENT THE TEXT CARRIES, appending each through the +// descriptor's place resolver: `[]` is an empty list, and null is +// kind_mismatch, the array row's own rule (§16.2). LAST WINS holds for a +// repeated key: the list goes back to EMPTY before this occurrence's elements +// land, the builder's storage being reclaimed at reset (§2.9). +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) slot, 8, 0 ); + TableJsonSetRaw( (uint8_t *) slot + 8, 4, 0 ); + const char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + void * element = f->place( *graph->worker, slot, NULL, 0, 0 ); + if ( element == NULL ) + { + // NOT ADDED: the arena could not carve another segment, or the + // count met the int32 cap. The text cannot be placed whole, and + // the read stops where §16.1's rule stops it. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( f->kind == 17 ) + { + // an element of a []*T (§2.9): null is a null slot, an object is the + // pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) element, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, element, f, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + // the wrong shape for the element kind: the slot keeps its + // defaults and the event counts, the array row's rule (§16.2) + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, element, f, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json list walk: end ---- + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +bool BytesFromJson( BytesBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Bytes * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, BytesTableType(), text, bytes, report ); +} + +int64_t BytesToJsonMeasure( const Bytes * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, BytesTableType(), NULL, 0, allocator ); +} + +int64_t BytesToJson( const Bytes * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, BytesTableType(), buffer, capacity, allocator ); +} + +bool IntsFromJson( IntsBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Ints * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, IntsTableType(), text, bytes, report ); +} + +int64_t IntsToJsonMeasure( const Ints * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, IntsTableType(), NULL, 0, allocator ); +} + +int64_t IntsToJson( const Ints * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, IntsTableType(), buffer, capacity, allocator ); +} + +bool FloatsFromJson( FloatsBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Floats * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, FloatsTableType(), text, bytes, report ); +} + +int64_t FloatsToJsonMeasure( const Floats * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, FloatsTableType(), NULL, 0, allocator ); +} + +int64_t FloatsToJson( const Floats * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, FloatsTableType(), buffer, capacity, allocator ); +} + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/ReportTable.h b/testdata/golden/tables/lists/ReportTable.h new file mode 100644 index 000000000..5139791b6 --- /dev/null +++ b/testdata/golden/tables/lists/ReportTable.h @@ -0,0 +1,7565 @@ +// Code generated by the schema compiler from Report.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — protocol id 0xa5fbe602c119cdd9 (packets only: tables version by field id, not by protocol id) +// The TABLE wire (evolution-tolerant, docs/SPEC-TABLES.md): no serialize +// dependency — includable from any TU. + +#pragma once + +#include +#include // the prefill's scalar-array fills +#include // offsetof, for the reflection descriptors + +// ---- the hooks (docs/USAGE.md, "the C++ table runtime's hooks") ---- +// +// schema_assert — the runtime's own assert, and the refusal a debugger reads. +// NDEBUG removes it, exactly as it removes assert. A caller who already routes +// serialize's asserts writes `#define schema_assert serialize_assert` before +// including this header and both halves land in one handler. +#ifndef schema_assert +#include +#define schema_assert assert +#endif // #ifndef schema_assert + +// schema_fatal — what stands after the assert on a path that cannot continue. +// NDEBUG does not remove it. Supply it and is never included. +#ifndef schema_fatal +#include // abort +#define schema_fatal abort +#endif // #ifndef schema_fatal + +// schema_allocate / schema_release — what "no allocator handed in" means for +// this program. schema_allocate hands back ZEROED bytes and NULL on failure: +// an arena segment is copied whole, padding included, so anything left +// uninitialized here would reach a packed region. Supply both and +// is never included; hand a TableAllocator to a builder to route one +// structure's allocations somewhere else again. +#ifndef schema_allocate +#include // calloc, free +#define schema_allocate( bytes ) calloc( (size_t) 1, (size_t) ( bytes ) ) +#define schema_release( pointer ) free( pointer ) +#endif // #ifndef schema_allocate +#include // a node's lifetime starts in arena storage (placement new) +#include // one atomic per slab: the arena is lock-free by ownership + +#include "Report.h" + +#ifndef LISTDEMO_SCHEMA_TABLE_PRIMITIVES +#define LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +// THE CODEC DOES NOT DEPEND ON THE COMPILER'S INLINING BUDGET. A table of a +// realistic field count emits one large body per type, and the cursor a body +// writes through lives in the caller's `TableWriter`: across a call boundary +// that cursor round-trips through memory, and a `uint8_t *` store may alias the +// writer itself, so every put reloads it. When a budget runs out mid-body the +// codec silently degrades to that shape. Forcing the primitives and the +// fixed-class bodies inline is what keeps the cursor in registers and lets +// adjacent constant framing bytes merge into one store. +#if defined( _MSC_VER ) +#define LISTDEMO_TABLE_INLINE __forceinline +#elif defined( __GNUC__ ) || defined( __clang__ ) +#define LISTDEMO_TABLE_INLINE inline __attribute__(( always_inline )) +#else +#define LISTDEMO_TABLE_INLINE inline +#endif + +namespace listdemo { + +// WHY A READ WAS REFUSED, by name (docs/SPEC-TABLES.md §3.3, §11). A REFUSAL +// is not one of §4's events: nothing is decoded, no counter moves and no +// damage is reported, so five zero counters and a false flag are what a clean +// read prints too and only the verdict tells them apart. The reason says which +// refusal it was. +// +// This is the MESSAGE PATH's vocabulary and not the cooked form's (§7.4): a +// caller meeting one of these has been refused a MESSAGE on a connection, +// which is a different recovery with a different owner than a file a header +// match turned down. +enum TableMessageReason +{ + newer_form, // a FORM BYTE this reader does not carry (§3) + no_vocabulary, // no table for this connection: the message arrived before the announcement, or after a refused one + second_announcement, // a second announcement on a connection: it sets nothing, amends nothing, and the connection closes + vocabulary_too_large, // an announcement above the receiver's declared bound, refused before an entry is touched + message_form_as_file // a form 2 wire where a FILE was expected: its table is somewhere else +}; + +// The table-wire read report — the permissive contract's ledger. Silence +// (all zero) means the data matched this reader's schema exactly. +struct TableReport +{ + int32_t unknown = 0; // unknown field ids skipped (newer data) + int32_t kind_mismatch = 0; // known id, changed type — skipped, never misdecoded + int32_t clamped = 0; // out-of-range values clamped to declared bounds + // a key the TEXT form saw twice: last wins, and the repeat is counted + // (docs/SPEC-TABLES.md §16.2). The wire never raises it — a body carrying an + // id twice is legal input whose last occurrence wins, silently (§3). + int32_t duplicate = 0; + bool malformed = false; // framing damage; decode stopped, partial result kept + // THE REFUSAL VERDICT, which is not one of §4's events and moves no counter + // (docs/SPEC-TABLES.md §3): a FORM BYTE this reader does not carry. Five + // zero counters and a false flag are what a clean read prints too, so the + // verdict is what tells the two apart. + bool refused = false; + // WHICH refusal, and it is read only when refused is set: a read that + // was not refused has no reason, and this member is the one the caller + // must not look at then (docs/SPEC-TABLES.md §3.3). + TableMessageReason reason = newer_form; +}; + + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; +// ---- reflection (tables only, docs/SPEC-TABLES.md) ---- +// +// Static field descriptors for every type in the table closure: name, wire +// id/kind, storage offset, bounds, ranges, enum names and branch guards — +// enough to walk, print, diff, edit or bind any table value at runtime with +// no RTTI and no schema files. TableType() returns X's descriptor. + +struct TableTypeInfo; + +// One arm of a union field: where its payload sits inside the union's storage +// and what its payload looks like. The arm's NAME and its table-wire id come +// from the field's enum_name/variant_id functions at the same tag, so nothing +// is spelled twice (docs/SPEC-TABLES.md §8). +struct TableFieldInfo; + +struct TableUnionArmInfo +{ + uint32_t offset; // offsetof the arm's payload within the union storage + const TableTypeInfo * table; // the arm payload's descriptor, or NULL + // AN ARM IS A FIELD LINE (docs/SPEC-TABLES.md §2.6): an arm that names no + // declared type or table carries the FIELD descriptor a field of that + // type would carry instead — offsets taken within the union storage — so + // a generic walk meets an arm's kind, width, bounds and companions where + // it meets a field's. Exactly one of the two is non-NULL on a set arm. + const TableFieldInfo * field; + uint32_t size; // the arm's whole storage, which selection zero-establishes +}; + +// A union field's shape: the tag, and the arms indexed by it. Arms run +// [0, enum_max]; index 0 is the EMPTY arm and carries no payload. +struct TableUnionInfo +{ + uint32_t tag_offset; // offsetof the tag within the union storage + uint32_t tag_size; // sizeof the tag + const TableUnionArmInfo * arms; +}; + +// The exact raw range of a wide-kind field (docs/SPEC-TABLES.md §8.2): two 128-bit +// values as 64-bit lanes, low lane first, two's complement for the signed kinds. +struct TableWideRange +{ + uint64_t lo[2]; + uint64_t hi[2]; +}; + +// the arena's allocation front, defined with the variable-length runtime +// below; a descriptor names it only through a pointer parameter. +struct TableWorker; + +struct TableFieldInfo +{ + const char * name; // schema field name, e.g. "health" + const char * json; // the TEXT form's key: the json = "key" attribute, else name (§16.3) + const char * type_name; // schema type name, e.g. "float32", "Grade" + uint64_t id; // table-wire field id: fnv1a64 of the name, of the was alias after a rename (§5) + uint8_t kind; // table-wire kind; for arrays/strings/bytes, the ELEMENT kind + bool is_array; // fixed or counted array (bytes included) + bool is_pointer; // a *T pointer field: storage is an 8-byte TableRef; the target is a table + // THE TWO THE TEXT FORM NEEDS (docs/SPEC-TABLES.md §16.7), and they + // are here for the same reason is_pointer is: the walk is ONE walk + // over descriptors and cannot spell a target's own At or + // Emplace. `resolve` reads a slot in a REGION and answers the + // node it names, or NULL; `emplace` allocates one in a BUILDER's + // arena and points the slot at it. NULL on every field that is not + // a pointer, and emitted only in a unit that declares one. + const void * (*resolve)( const void * slot ); + void * (*emplace)( TableWorker & worker, void * slot ); + bool counted; // a _count/_length int32 companion exists (counted arrays, strings, bytes) + bool optional; // a ?T field: a _present bool companion decides whether it rides + int32_t array_bound; // array capacity / string max length; 0 for plain scalars + uint32_t offset; // offsetof the storage member + uint32_t elem_size; // sizeof the member (element size for arrays) + uint32_t count_offset; // offsetof the _count/_length companion, or 0xffffffff + uint32_t present_offset; // offsetof the _present companion, or 0xffffffff + const TableTypeInfo * table; // nested table's descriptor, or NULL + bool has_range; // a declared [min, max] (int or float) + double range_min; // NOTE: int64 ranges beyond 2^53 lose precision here + double range_max; + // the WIDE kinds (18-29, docs/SPEC-TABLES.md §3, §8.2): frac_bits is a fixed + // field's F — its storage holds units × 2^F — and wide is the declared + // range on that RAW scale, exact, as two 128-bit two's-complement values + // in 64-bit lanes (low lane first). NULL where the declaration bounds + // nothing (a bare uint128) and for every other kind; frac_bits is 0 for + // every kind that is not fixed-point. range_min/range_max still carry + // the declared bounds as doubles — whole units for a fixed field — for + // a walker that only shows them. + uint8_t frac_bits; + const TableWideRange * wide; + int64_t enum_max; // enums: highest valid value (None = 0 always valid); + // unions: the arm count (tag range [0, enum_max]); + // flags: the highest declared BIT INDEX; else -1 + // the vocabulary's names, indexed the same way enum_max bounds: an enum's + // value -> name, a union's tag -> arm name, a FLAGS field's bit index -> + // variant name. NULL for every other kind. + const char * (*enum_name)( uint64_t value ); + // the TABLE-WIRE id of one variant (docs/SPEC-TABLES.md §5): for an enum, the + // hash of the variant's name; for a union, the hash of the arm's name. + // 0 is the reserved id — an enum's None, a union's empty. NULL for every + // other kind — a FLAGS field's variants have no per-variant wire id (§4), + // so a NULL here beside a non-NULL enum_name is what says "flags". + // Walk [0, enum_max] to enumerate a vocabulary and its ids. + uint64_t (*variant_id)( uint64_t value ); + // an ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): the array has one slot per + // variant of key_type_name, indexed by the variant's value, and its slots + // ride under variant ids rather than positions. key_name and key_id are + // the key's vocabulary — walk [0, array_bound) to print slots by name. + // NULL on every other field. + const char * key_type_name; + const char * (*key_name)( uint64_t value ); + uint64_t (*key_id)( uint64_t value ); + // union fields: the tag and its arms, behind a function so the whole + // descriptor stays CONSTANT-INITIALISED (a captureless lambda converts to + // a function pointer at compile time; the arms themselves are a static + // inside it). NULL for every other kind. + const TableUnionInfo * (*arms)(); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded +}; + +struct TableTypeInfo +{ + const char * name; // schema type name + uint32_t size; // sizeof the storage struct + int32_t num_fields; + const TableFieldInfo * fields; + // put one instance back at its declared defaults, in place. A generic + // walker that fills a value has to be able to establish the defaults an + // absent field takes, and it holds no type to spell — this is the one + // thing the descriptors could not express without it. Placement-new + // value-init, exactly what the wire's read path does, and no temporary. + void (*reset)( void * storage ); + // the DERIVED mode (docs/SPEC-TABLES.md): false = fixed-size, a plain + // relocatable struct; true = variable-length, built through a Builder + // and read through a region root. Nobody declares it; the compiler + // works it out. + bool variable; +}; + +struct TableWriter +{ + uint8_t * buffer; + int64_t capacity; + int64_t offset = 0; + bool overflow = false; + + // the parameters do not repeat the member names: a parameter that hides a + // member is a warning the estate's compilers disagree about (gcc's + // -Wshadow and cl's C4458 refuse it, clang's -Wshadow does not), and this + // is a header a consumer compiles under its OWN flags + TableWriter( uint8_t * to_buffer, int64_t to_capacity ) : buffer( to_buffer ), capacity( to_capacity ) {} + + LISTDEMO_TABLE_INLINE void raw( const void * data, int64_t bytes ) + { + if ( offset + bytes > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) bytes ); + offset += bytes; + } + LISTDEMO_TABLE_INLINE void put8( uint8_t v ) { raw( &v, 1 ); } + LISTDEMO_TABLE_INLINE void put16( uint16_t v ) { uint8_t b[2] = { uint8_t( v ), uint8_t( v >> 8 ) }; raw( b, 2 ); } + LISTDEMO_TABLE_INLINE void put32( uint32_t v ) { uint8_t b[4] = { uint8_t( v ), uint8_t( v >> 8 ), uint8_t( v >> 16 ), uint8_t( v >> 24 ) }; raw( b, 4 ); } + LISTDEMO_TABLE_INLINE void put64( uint64_t v ) { put32( uint32_t( v ) ); put32( uint32_t( v >> 32 ) ); } + // a 128-bit value as two lanes, the low half first (docs/SPEC-TABLES.md §3) + LISTDEMO_TABLE_INLINE void put128( uint64_t lo, uint64_t hi ) { put64( lo ); put64( hi ); } + // EVERY LENGTH, COUNT, INDEX AND ID REFERENCE IS ONE CANONICAL UNSIGNED + // LEB128 (docs/SPEC-TABLES.md §3): seven value bits a byte, the lowest + // group first, the high bit set on every byte but the last. One value has + // one spelling, so two conforming writers agree byte for byte. + LISTDEMO_TABLE_INLINE void putleb( uint64_t v ) + { + while ( v >= 0x80 ) { put8( uint8_t( v ) | 0x80 ); v >>= 7; } + put8( uint8_t( v ) ); + } +}; + +// TableLebBytes is one value's spelling length, which a MEASURE needs before +// the bytes exist — the length of a body has to be known before it is written, +// because a length whose own width moves cannot be patched in place. +inline int64_t TableLebBytes( uint64_t v ) +{ + int64_t n = 1; + while ( v >= 0x80 ) { v >>= 7; n++; } + return n; +} + +// THE ID TABLE, WRITER SIDE (docs/SPEC-TABLES.md §3). It holds every id the +// body used, once each, in FIRST-USE order over the whole wire, and the body +// names them by position: reference k is the kth entry, counted from 1, and +// reference 0 names NO ID. +// +// Its capacity is a COMPILE-TIME fact of the unit — the distinct names its +// table closure can spell — so a save allocates nothing: the table is a local +// of Measure and of Save. The bucket chain makes ref constant time and makes +// truncate constant time too, which is what an ELIDED field needs: a field +// that turns out not to ride costs nothing in the id table either, so the walk +// interns its id, builds the payload that decides, and undoes the entry when +// nothing rides. +struct TableIds +{ + static const int32_t kCapacity = 60; + static const int32_t kBuckets = 128; + + uint64_t ids[ kCapacity ]; + int32_t chain[ kCapacity ]; + int32_t head[ kBuckets ]; + int32_t count; + bool overflow; + // THE MESSAGE FORM'S SLOTS (docs/SPEC-TABLES.md §3.3). A form 2 wire + // names ids through the CONNECTION's table, which is the unit's whole + // vocabulary in a compiler-settled order — so every reference is known at + // compile time and rides at the header as a literal beside the id. This + // flag is what selects it: false interns the id in first-use order and + // writes a trailer, true answers the slot and writes none, and the walk + // that decides is one walk. + bool vocabulary; + + TableIds() : count( 0 ), overflow( false ), vocabulary( false ) + { + for ( int32_t i = 0; i < kBuckets; i++ ) { head[i] = -1; } + } + + static LISTDEMO_TABLE_INLINE uint32_t bucket_of( uint64_t id ) + { + return uint32_t( ( id * 0x9E3779B97F4A7C15ull ) >> 57 ) & uint32_t( kBuckets - 1 ); + } + + // the reference an id takes: its message-form SLOT under the connection's + // table, or the file's own first-use entry + LISTDEMO_TABLE_INLINE uint64_t ref( uint64_t id, uint64_t slot ) + { + if ( vocabulary ) { return slot; } + return intern( id ); + } + + // the FILE form's half, appending the id on first use + uint64_t intern( uint64_t id ) + { + const uint32_t b = bucket_of( id ); + for ( int32_t i = head[b]; i >= 0; i = chain[i] ) + { + if ( ids[i] == id ) { return uint64_t( i ) + 1; } + } + if ( count >= kCapacity ) { overflow = true; return 1; } + ids[count] = id; chain[count] = head[b]; head[b] = count; count++; + return uint64_t( count ); + } + + // undo every entry appended since mark. An entry removed is the most + // recent one in its bucket, so it sits at that bucket's head. + void truncate( int32_t mark ) + { + // a SLOT costs no entry, so an elided field has nothing to undo + if ( vocabulary ) { return; } + while ( count > mark ) + { + count--; + head[ bucket_of( ids[count] ) ] = chain[count]; + } + } +}; + +// TableIdsBytes is the trailer's own size: the entries, each a fixed +// little-endian u64, and the ENTRY COUNT, the one fixed-width number on the +// wire (docs/SPEC-TABLES.md §3). +inline int64_t TableIdsBytes( const TableIds & ids ) { return int64_t( ids.count ) * 8 + 8; } + +// TableIdsWrite puts the trailer where the walk ended: a writer never patches, +// because first-use order is known only when the walk ends. +inline void TableIdsWrite( TableWriter & w, const TableIds & ids ) +{ + for ( int32_t i = 0; i < ids.count; i++ ) { w.put64( ids.ids[i] ); } + w.put64( uint64_t( ids.count ) ); +} + +// THE ID TABLE, READER SIDE (docs/SPEC-TABLES.md §3). A reader locates it from +// the END of the wire and resolves it ONCE, at open: the entries are eight +// bytes each and a body names them by position, so every field dispatches +// through an index rather than through a search over hashes. +struct TableIdTable +{ + const uint8_t * entries = NULL; + int64_t count = 0; + + // the id a reference names. ref is 1-based and bounds-checked by the + // caller: a reference ABOVE the entry count is framing damage on the body + // that carries it, and 0 names no id at all. + uint64_t at( uint64_t ref ) const + { + const uint8_t * e = entries + ( ref - 1 ) * 8; + uint64_t lo = uint64_t( e[0] ) | uint64_t( e[1] ) << 8 | uint64_t( e[2] ) << 16 | uint64_t( e[3] ) << 24; + uint64_t hi = uint64_t( e[4] ) | uint64_t( e[5] ) << 8 | uint64_t( e[6] ) << 16 | uint64_t( e[7] ) << 24; + return lo | ( hi << 32 ); + } +}; + +struct TableReader +{ + const uint8_t * buffer; + int64_t size; + int64_t offset = 0; + TableReport * report; + const TableIdTable * ids = NULL; + // ONLY THE ROOT BODY CARRIES THE NODE TABLE (docs/SPEC-TABLES.md §3.1), so + // a body has to know which it is: the reserved id inside a NESTED body is + // malformed, because a second numbering cannot exist. Every reader made + // for a payload is nested; the two the wire surfaces make for a root say so. + bool nested = true; + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report ) + : buffer( from_buffer ), size( from_size ), report( to_report ) {} + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report, const TableIdTable * to_ids ) + : buffer( from_buffer ), size( from_size ), report( to_report ), ids( to_ids ) {} + + LISTDEMO_TABLE_INLINE bool has( int64_t bytes ) const { return offset + bytes <= size; } + // A LENGTH IS A 64-BIT NUMBER AND A BUFFER IS NOT (docs/SPEC-TABLES.md + // §3): every length, count and index on this wire has sixty-four bits of + // capability, so one past what remains must be compared UNSIGNED. Casting + // it to int64 first turns 0xFFFFFFFFFFFFFFFF into -1, and a negative + // length looks like room. + LISTDEMO_TABLE_INLINE bool room( uint64_t bytes ) const { return bytes <= (uint64_t) ( size - offset ); } + LISTDEMO_TABLE_INLINE uint8_t get8() { return buffer[offset++]; } + LISTDEMO_TABLE_INLINE uint16_t get16() { uint16_t v = uint16_t( buffer[offset] ) | uint16_t( buffer[offset+1] ) << 8; offset += 2; return v; } + LISTDEMO_TABLE_INLINE uint32_t get32() { uint32_t v = uint32_t( buffer[offset] ) | uint32_t( buffer[offset+1] ) << 8 | uint32_t( buffer[offset+2] ) << 16 | uint32_t( buffer[offset+3] ) << 24; offset += 4; return v; } + LISTDEMO_TABLE_INLINE uint64_t get64() { uint64_t lo = get32(); uint64_t hi = get32(); return lo | ( hi << 32 ); } + LISTDEMO_TABLE_INLINE void get128( uint64_t & lo, uint64_t & hi ) { lo = get64(); hi = get64(); } + + // ONE CANONICAL UNSIGNED LEB128 (docs/SPEC-TABLES.md §3), and a + // non-minimal spelling is MALFORMED: 0x80 0x00 and 0x00 both spell zero, + // and only the second is legal input. An encoding past ten bytes, or a + // tenth byte with a bit above the 64th value bit, is malformed on the same + // rule. false = framing damage on the body carrying it. + bool getleb( uint64_t & value ) + { + // A NUMBER THIS READER REFUSES LEAVES THE CURSOR WHERE IT WAS. The + // caller's next question is often "did this body end exactly at its + // L", and a rejected number that had moved the cursor would answer + // that question with the damage already stepped over. + const int64_t at = offset; + value = 0; + uint32_t shift = 0; + for ( int32_t i = 0; i < 10; i++ ) + { + if ( !has( 1 ) ) { offset = at; return false; } + const uint8_t b = get8(); + if ( i == 9 && b > 1 ) { offset = at; return false; } + value |= uint64_t( b & 0x7F ) << shift; + if ( ( b & 0x80 ) == 0 ) + { + if ( i > 0 && b == 0 ) { offset = at; return false; } // a redundant continuation + return true; + } + shift += 7; + } + offset = at; + return false; + } + + // resolve one id reference against the file's table. false = a reference + // ABOVE the entry count, or a 0 where an id is required, both of which + // are framing damage on the body that carries it. + bool getid( uint64_t & id ) + { + uint64_t ref = 0; + if ( !getleb( ref ) ) { return false; } + if ( ref == 0 || ids == NULL || ref > (uint64_t) ids->count ) { return false; } + id = ids->at( ref ); + return true; + } + + // skip one payload by kind; false = framing damage. FOUR RULES COVER THE + // SET (docs/SPEC-TABLES.md §3), and a kind outside it is not skippable — + // which is why the set is closed and why kind 31 exists. + bool skip( uint8_t kind ) + { + switch ( kind ) + { + // the fixed-width kinds, each by its width: 18-29 are the 128-bit integers and + // the fixed-point family at every storage width (docs/SPEC-TABLES.md §3) + case 1: case 2: case 6: case 20: case 25: return has( 1 ) ? ( offset += 1, true ) : false; + case 3: case 7: case 21: case 26: return has( 2 ) ? ( offset += 2, true ) : false; + case 4: case 8: case 10: case 22: case 27: return has( 4 ) ? ( offset += 4, true ) : false; + case 5: case 9: case 11: case 23: case 28: return has( 8 ) ? ( offset += 8, true ) : false; + case 18: case 19: case 24: case 29: return has( 16 ) ? ( offset += 16, true ) : false; + case 17: case 30: // a NODE INDEX (§3.1) and an ENUM's variant reference: one LEB128 and stop + { + uint64_t ignored = 0; + return getleb( ignored ); + } + case 12: case 13: case 14: case 16: case 31: case 32: // 31 is the ESCAPE, 32 the payload-free kind + { + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + case 15: // union: the arm id reference, then its kind, its L and its payload (reference 0 = empty) + { + uint64_t arm = 0; + if ( !getleb( arm ) ) return false; + if ( arm == 0 ) return true; + if ( !has( 1 ) ) return false; + offset += 1; // the arm's kind byte + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + } + return false; + } +}; + +// The RESERVED node-table id, the one id the language holds back +// (docs/SPEC-TABLES.md §3.1, §5). It rides in every unit, pointered or not, +// because every body has to know that a NESTED body claiming one is damaged. +static const uint64_t kTableNodeTableFieldId = 0xFFFFFFFFFFFFFFFFull; + +// TableWireForm is the FORM BYTE, and it is the whole header +// (docs/SPEC-TABLES.md §3). A reader that meets a byte it does not know +// refuses the wire by name and never reports damage. +const uint8_t kTableWireForm = 1; + +// TableOpen reads the form byte and the trailer, in that order, and hands back +// the ROOT BODY. It answers one of three verdicts, because five zero counters +// and a false flag are what a clean read prints too: +// +// TableOpenOk the form is known and the table read whole +// TableOpenRefused a FORM BYTE this reader does not carry: nothing is +// decoded, nothing is counted, and no damage is reported +// TableOpenDamaged a table that cannot be read whole — fewer than eight +// bytes, a count whose entries run past the front of the +// file, a count that leaves no room for the form byte, or +// ONE ID IN TWO ENTRIES. The whole wire is malformed, +// nothing is decoded, and one event is counted. +// TableOpenBodyStopped the form and the table were good and the ROOT BODY +// could not be walked to its own terminator. What it +// decoded before that is kept, as everywhere on this wire. +enum TableOpenVerdict { TableOpenOk, TableOpenRefused, TableOpenDamaged, TableOpenBodyStopped }; + +inline TableOpenVerdict TableOpen( const uint8_t * buffer, int64_t bytes, TableIdTable & table, int64_t & body_bytes ) +{ + if ( bytes < 1 ) { return TableOpenDamaged; } + if ( buffer[0] != kTableWireForm ) { return TableOpenRefused; } + if ( bytes < 9 ) { return TableOpenDamaged; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + uint64_t count = lo | ( hi << 32 ); + if ( count > (uint64_t) ( bytes / 8 ) ) { return TableOpenDamaged; } + const int64_t span = (int64_t) count * 8 + 8; + if ( span + 1 > bytes ) { return TableOpenDamaged; } + table.entries = buffer + bytes - span; + table.count = (int64_t) count; + // THE ENTRIES ARE DISTINCT: a table that carries one id twice is malformed + // for the whole wire, because no wire this schema writes carries a repeat + // and it would leave one more shape of table for a hostile writer to aim + // at (docs/SPEC-TABLES.md §3). + for ( int64_t i = 1; i < table.count; i++ ) + { + const uint64_t id = table.at( uint64_t( i ) + 1 ); + for ( int64_t j = 0; j < i; j++ ) + { + if ( table.at( uint64_t( j ) + 1 ) == id ) { return TableOpenDamaged; } + } + } + body_bytes = bytes - span - 1; + return TableOpenOk; +} + +// TableBodyExtent walks a body's framing to the zero reference that ends it, +// so a reader can tell a body that ENDED EARLY — leaving bytes no field claims +// — from one that is merely damaged. ANY BYTE BETWEEN THE ROOT'S TERMINATOR +// AND THE TABLE'S FIRST ENTRY IS MALFORMED, because no field claims it and the +// two ends of the file have met (docs/SPEC-TABLES.md §3). +inline bool TableBodyEndsEarly( const uint8_t * body, int64_t bytes, const TableIdTable & table ) +{ + TableReport ignored; + TableReader r( body, bytes, &ignored, &table ); + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { return false; } + if ( ref == 0 ) { return r.offset != bytes; } + if ( ref > (uint64_t) table.count ) { return false; } + if ( !r.has( 1 ) ) { return false; } + if ( !r.skip( r.get8() ) ) { return false; } + } +} + +// THE MESSAGE FORM (docs/SPEC-TABLES.md §3.3): a FILE carries its own id +// table and a MESSAGE STREAM announces one and then carries none. +// +// A form 2 wire is TWO PARTS, the form byte and the root body: the body ends +// at its own zero reference as it does in a file, there is no trailer, and the +// message's last byte is the body's terminator. Its references resolve against +// the CONNECTION's table, which is the unit's whole vocabulary in the order +// the compiler settled. +const uint8_t kTableWireMessageForm = 2; + +// The RESERVED build-version id, the second id the language holds back (§5, +// §11), beside the node table's. It is the announcement's one required field, +// and a reserved id in any body but the one whose transport it is, is +// malformed (§3.1). +static const uint64_t kTableBuildVersionFieldId = 0xFFFFFFFFFFFFFFFEull; + +// The reserved NODE-TABLE id's own slot in this unit's vocabulary (§3.3). A +// pointered message names the node table through it, exactly as every other +// field header names its id through a slot. +static const uint64_t kTableNodeTableFieldSlot = 38; + +// THE UNIT'S ANNOUNCEMENT, byte for byte: 61 entries and 508 bytes. It is an +// ordinary form 1 FILE — the form byte, a body carrying the BUILD VERSION +// under the reserved id at kind 9, and the trailer that IS the connection's +// table, slot 1 the reserved id and slots 2 and up the vocabulary under one +// numbering. +// +// The vocabulary is the unit's whole closure in the COOK PROJECTION's order +// (§20.2) — each record in the order the projection renders it and each +// record's fields in the order the projection renders them, then each enum's +// variants and each union's arms — followed by the tail the projection does +// not name: the reserved node-table id, the three blob type ids as bytes, +// string and wstring, and every table's own name id in the projection's sorted +// record order. The tail is UNCONDITIONAL, so an ordinary edit only ever grows +// it at its end and never moves a slot a generated field header carries as a +// literal. +static const int64_t kTableAnnounceBytes = 508; +static const uint8_t kTableAnnounce[ kTableAnnounceBytes ] = { + 0x01, 0x01, 0x09, 0xc7, 0x71, 0x45, 0xca, 0xda, 0x0e, 0x7c, 0x8d, 0x00, + 0xfe, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x30, 0xb1, 0x3a, 0xff, + 0x4a, 0xd9, 0xb1, 0x40, 0x20, 0xea, 0x4d, 0x40, 0x8e, 0xa7, 0x19, 0xaa, + 0x26, 0xa9, 0x02, 0x0c, 0x9b, 0x01, 0x48, 0x78, 0xe9, 0xea, 0x71, 0x6f, + 0x0f, 0x01, 0x82, 0xbf, 0x6f, 0x2c, 0x41, 0x4f, 0xbf, 0x84, 0x78, 0x3e, + 0xf3, 0xa4, 0x48, 0x44, 0x19, 0xab, 0xd7, 0x56, 0x05, 0x4a, 0xa3, 0x30, + 0x67, 0x55, 0x5b, 0x85, 0xc9, 0xe2, 0x4e, 0x30, 0x69, 0x6a, 0xb4, 0x81, + 0xfb, 0x67, 0x4d, 0x1a, 0xcf, 0x7b, 0x27, 0x21, 0x74, 0xa2, 0x79, 0x44, + 0x8e, 0xe2, 0xe5, 0xb1, 0x84, 0x76, 0xbc, 0x2e, 0xef, 0x83, 0x76, 0x1e, + 0xc5, 0x99, 0xf7, 0x82, 0x76, 0x4e, 0x0a, 0xd9, 0xa8, 0x2e, 0x86, 0x70, + 0x84, 0xed, 0xf2, 0x4a, 0xbb, 0xf0, 0x0c, 0x9b, 0xcc, 0xfb, 0x2d, 0x73, + 0x68, 0xb7, 0xf0, 0xae, 0x4c, 0x0c, 0xf6, 0x52, 0xbf, 0xe9, 0xd1, 0x2f, + 0x93, 0xcd, 0xda, 0xdb, 0x22, 0x72, 0x34, 0x7d, 0xf6, 0x0b, 0x72, 0x17, + 0x07, 0x17, 0x02, 0x86, 0x4c, 0xf5, 0x63, 0xaf, 0x54, 0x15, 0x02, 0x86, + 0x4c, 0xf4, 0x63, 0xaf, 0x3a, 0x70, 0x6e, 0x3e, 0x93, 0x43, 0xe5, 0x9d, + 0x3d, 0x62, 0xcb, 0x8f, 0xec, 0xfc, 0xf7, 0x39, 0x09, 0x06, 0x02, 0x86, + 0x4c, 0xeb, 0x63, 0xaf, 0x09, 0x4b, 0x4d, 0x57, 0xaa, 0x33, 0x47, 0xd2, + 0x31, 0x54, 0xaf, 0x1d, 0x19, 0x73, 0x50, 0x12, 0xb2, 0x0f, 0x40, 0x27, + 0x0b, 0x6b, 0x98, 0x01, 0x38, 0x81, 0x0a, 0xf1, 0x1f, 0x06, 0xa7, 0xa3, + 0x0f, 0x62, 0xad, 0x07, 0x77, 0x47, 0x82, 0x5f, 0x42, 0x4f, 0x4f, 0x30, + 0x0d, 0x39, 0x84, 0x1c, 0x86, 0x1b, 0x63, 0x8e, 0xba, 0xad, 0xbc, 0xc4, + 0xec, 0x10, 0x5b, 0x36, 0x19, 0x4a, 0xc9, 0x3d, 0xea, 0x0c, 0xe8, 0x30, + 0x94, 0xfd, 0xe4, 0x7c, 0xec, 0x22, 0x02, 0x86, 0x4c, 0xfc, 0x63, 0xaf, + 0x05, 0x28, 0x02, 0x86, 0x4c, 0xff, 0x63, 0xaf, 0x52, 0x26, 0x02, 0x86, + 0x4c, 0xfe, 0x63, 0xaf, 0xb1, 0x45, 0xc3, 0x44, 0x35, 0xab, 0xfe, 0x73, + 0xc0, 0x7f, 0xb3, 0x8a, 0xbe, 0x08, 0x63, 0x7f, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xe4, 0x4f, 0x1c, 0x4f, 0x47, 0xc0, 0x2e, 0x2f, + 0x58, 0xfc, 0xaf, 0xfa, 0xd8, 0xe0, 0x4b, 0x70, 0xc7, 0xd4, 0x7b, 0x26, + 0xb0, 0x9d, 0x29, 0x5f, 0xcc, 0x14, 0x15, 0x7f, 0xcb, 0xc2, 0x58, 0xd8, + 0x84, 0x9e, 0x3a, 0x55, 0x8b, 0x37, 0xe2, 0x06, 0x2a, 0x89, 0xf5, 0x12, + 0x09, 0xc1, 0xca, 0x0a, 0x44, 0xa2, 0x31, 0xc1, 0xad, 0xa7, 0xee, 0xee, + 0xe8, 0xcf, 0xbf, 0x43, 0x73, 0x18, 0x43, 0xd0, 0x42, 0xad, 0xf6, 0xf8, + 0x59, 0x86, 0x63, 0x91, 0xb7, 0xce, 0x00, 0x7c, 0xd1, 0xc5, 0x34, 0x20, + 0x06, 0x68, 0x47, 0x98, 0xd1, 0xa1, 0xcf, 0x52, 0x5f, 0x82, 0x58, 0xac, + 0x36, 0x15, 0x78, 0x5e, 0xb8, 0x8b, 0x59, 0x6f, 0xc9, 0xc6, 0x86, 0xbb, + 0xc3, 0x64, 0x89, 0x50, 0xd2, 0x8d, 0xa7, 0xf1, 0x80, 0xea, 0x3a, 0xb9, + 0xf1, 0x21, 0xf7, 0x41, 0x11, 0xed, 0xd9, 0xce, 0x96, 0x92, 0x43, 0x8a, + 0xfb, 0x06, 0xc9, 0xfe, 0x19, 0xe1, 0x13, 0xa0, 0xa7, 0x0a, 0xc7, 0x54, + 0x12, 0xd6, 0x40, 0xdc, 0x08, 0xf0, 0xf5, 0xc0, 0x24, 0x5f, 0xf8, 0x33, + 0xc8, 0xfb, 0x85, 0x9a, 0xaf, 0xe0, 0xc9, 0x0c, 0x91, 0x0a, 0x55, 0x60, + 0xf7, 0xa2, 0x07, 0xec, 0x8b, 0x6d, 0x02, 0x86, 0x43, 0xf3, 0xc2, 0x2e, + 0x87, 0x27, 0xcc, 0x86, 0xf0, 0xe0, 0x26, 0x8f, 0x3d, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, +}; + +// AnnounceMeasure is the announcement's byte count, which is a constant of the +// unit and not a walk. +inline int64_t AnnounceMeasure() { return kTableAnnounceBytes; } + +// Announce writes the announcement into the caller's buffer and answers the +// bytes written — exactly AnnounceMeasure's answer — or -1 when the buffer is +// too small. It allocates nothing and walks nothing. +inline int64_t Announce( uint8_t * buffer, int64_t capacity ) +{ + if ( buffer == NULL || capacity < kTableAnnounceBytes ) { return -1; } + memcpy( buffer, kTableAnnounce, (size_t) kTableAnnounceBytes ); + return kTableAnnounceBytes; +} + +// TableVocabulary is ONE DIRECTION of ONE CONNECTION's id table (§3.3): the +// entries an announcement carried, whole, under one numbering with slot 1 the +// reserved build-version id. +// +// A peer holds TWO of these for a connection, the one it writes with and the +// one it reads with, and neither is the other's. A restart opens a fresh +// connection with empty tables and nothing is cached across connections, so +// its whole life is one connection's. It BORROWS the announcement's bytes rather than +// copying them, so a receiver holds one table a direction and its memory is +// the bound below and nothing else. +struct TableVocabulary +{ + // THE CONFORMING DEFAULT BOUND (§3.3): 32 KiB a direction, eight times the + // 500-id unit that is already a large one. A connection's table is bounded + // by nothing the wire carries, so the receiver declares the maximum and an + // announcement above it is refused by name before an entry is touched. + static const int64_t kDefaultMaxEntries = 4096; + + TableIdTable table; + uint64_t build_version = 0; + bool announced = false; + int64_t max_entries = kDefaultMaxEntries; +}; + +// AnnounceRead reads an announcement into one direction's table (§3.3). +// +// THE BOUND IS CHECKED BEFORE ANYTHING IS ALLOCATED: the entry count is a +// fixed little-endian u64 at the end, so a receiver reads it, compares it and +// refuses without touching an entry. After that it is §3's ordinary FILE read, +// because the announcement IS a file, with EXACTLY ONE STRICT CHECK over its +// body: the reserved build-version field present, exactly once, under kind 9, +// eight bytes wide. Everything else is an ordinary field under §4's tolerance, +// so an unknown one is skipped and counted and the announcement can GAIN a +// field in a later minor without a lockstep redeploy. +// +// The FIRST announcement sets the table and it is the only one that can. A +// SECOND is refused by name: it does not replace the table, it does not amend +// it and it changes nothing. A refused announcement sets NO TABLE. +inline bool AnnounceRead( TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + if ( vocabulary.announced ) + { + to->refused = true; + to->reason = second_announcement; + return false; + } + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireForm ) + { + to->refused = true; + to->reason = buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + return false; + } + if ( bytes < 9 ) { to->malformed = true; return false; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + if ( ( lo | ( hi << 32 ) ) > (uint64_t) vocabulary.max_entries ) + { + to->refused = true; + to->reason = vocabulary_too_large; + return false; + } + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else { to->refused = true; to->reason = newer_form; } + return false; + } + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) { to->malformed = true; return false; } + // the body, under §4's tolerance and this form's one strict check + TableReader r( buffer + 1, body_bytes, to, &table ); + uint64_t version = 0; + int32_t seen = 0; + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { to->malformed = true; return false; } + if ( ref == 0 ) { break; } + if ( ref > (uint64_t) table.count || !r.has( 1 ) ) { to->malformed = true; return false; } + const uint64_t id = table.at( ref ); + const uint8_t kind = r.get8(); + if ( id != kTableBuildVersionFieldId ) + { + to->unknown++; + if ( !r.skip( kind ) ) { to->malformed = true; return false; } + continue; + } + if ( kind != 9 || !r.has( 8 ) ) { to->refused = true; to->reason = no_vocabulary; return false; } + version = r.get64(); + seen++; + } + if ( seen != 1 ) { to->refused = true; to->reason = no_vocabulary; return false; } + vocabulary.table = table; + vocabulary.build_version = version; + vocabulary.announced = true; + return true; +} + +inline float table_bits_to_float( uint32_t bits ) { float f; memcpy( &f, &bits, 4 ); return f; } +inline uint32_t table_float_to_bits( float f ) { uint32_t b; memcpy( &b, &f, 4 ); return b; } +inline double table_bits_to_double( uint64_t bits ) { double d; memcpy( &d, &bits, 8 ); return d; } +inline uint64_t table_double_to_bits( double d ) { uint64_t b; memcpy( &b, &d, 8 ); return b; } + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +#ifndef LISTDEMO_SCHEMA_TABLE_ARENA +#define LISTDEMO_SCHEMA_TABLE_ARENA + +namespace listdemo { + +// ---- variable-length tables: tuning constants (docs/SPEC-TABLES.md) ---- +// +// The segment size and the count multiply to exactly 2^32: the u32 reference +// is the arena's hard ceiling, and these constants saturate it rather than +// leaving address space unreachable. Slab handout costs one atomic per slab, +// so per-node allocation costs no synchronization at all. + +static const uint32_t kTableSegmentBits = 22; // 4 MiB segments +static const uint32_t kTableSegmentSize = 1u << kTableSegmentBits; +static const uint32_t kTableSegmentMask = kTableSegmentSize - 1u; +static const uint32_t kTableMaxSegments = 1u << ( 32 - kTableSegmentBits ); // 1024 -> 4 GiB +static const uint32_t kTableSlabBytes = 64u * 1024u; // one atomic per slab +static const uint32_t kTableAlign = 8; // every node starts 8-aligned +static const uint32_t kTableAllocFailed = 0xFFFFFFFFu; + +// ---- THE CALLER'S ALLOCATOR (docs/SPEC-TABLES.md §6.5) ---- +// +// Every allocation the variable-length runtime makes goes through one of +// these — the arena's segments, the pack walk's identity map, the numbering's +// entry array, the packed region, and the tool path's node directory. There is +// no other call to the C library on this path, so a counting allocator sees +// every byte and a game's own heap can own all of it. +// +// It is the shape TableBlockAllocator already has (§19.1): two function +// pointers and a context the caller carries. What it adds is a CONTRACT ON +// alloc — the bytes come back ZEROED. Lock copies whole nodes, PADDING +// INCLUDED, so anything left uninitialized reaches a packed region; the default +// pair reaches that through calloc, which costs nothing measurable because a +// fresh segment is untouched pages either way. +struct TableAllocator +{ + void * ( *alloc )( void * context, int64_t bytes ); // ZEROED bytes, NULL on failure + void ( *free )( void * context, void * pointer ); + void * context; +}; + +// The default pair, and it is the one every entry point takes when the caller +// names none. It calls schema_allocate / schema_release, so a program with its +// own C-library replacement can move the floor without writing a struct at all. +inline void * table_default_alloc( void * context, int64_t bytes ) { (void) context; return schema_allocate( bytes ); } +inline void table_default_free( void * context, void * pointer ) { (void) context; schema_release( pointer ); } + +inline TableAllocator TableDefaultAllocator() +{ + TableAllocator allocator; + allocator.alloc = table_default_alloc; + allocator.free = table_default_free; + allocator.context = NULL; + return allocator; +} + +// ---- TableRef: a relocatable reference (never a machine pointer) ---- +// +// Two encodings, one slot, and the FORM says which is in force: +// +// in the arena — the node's arena offset (segment index in the high bits) +// in a region — the SELF-RELATIVE byte delta from this slot's own address, +// so a deref is one add, needs no base pointer, and a whole +// region relocates by memcpy with zero fix-up +// +// 0 is null in both, and a slot can never name the node that contains it, so +// zero names nothing real in either form. +// +// A REGION DELTA HAS NO REQUIRED SIGN (§6.3). A region is packed depth-first, +// so a node's FIRST reference points forward; every LATER reference to that +// same node points BACK at the one body it already has, which is exactly what +// makes one node one node in a region. Sharing and a back-reference are the +// same fact, and nothing validates a reference by its sign. +// +// IT IS EIGHT BYTES, SIGNED, so ONE REGION REACHES EVERYTHING (§6.3, §7): a +// four-byte slot bounded a region at 2 GiB, and the scale a cook exists for is +// *"100mbs or many gigabytes of data in Assets.bin"*. +struct TableRef +{ + int64_t value = 0; + bool null() const { return value == 0; } +}; + +// TableSlot is what Alloc hands back: usable as the node pointer (write +// fields through it) AND as the reference to store in a pointer field. +template struct TableSlot +{ + T * ptr = NULL; + TableRef ref; + T * operator->() const { return ptr; } + T & operator*() const { return *ptr; } + operator T *() const { return ptr; } + operator TableRef() const { return ref; } + bool null() const { return ptr == NULL; } +}; + +inline uint32_t TableAlignUp( uint32_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( kTableAlign - 1 ); } +inline int64_t TableAlignUp64( int64_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( int64_t( kTableAlign ) - 1 ); } + +// ---- a BYTE BUFFER's node (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// A *bytes or *string slot is a TableRef like every pointer slot, and it names +// a BLOB NODE: this eight-byte header and then the bytes, at offset eight so +// the data is eight-aligned. A *string blob carries one more zero byte after +// its data, so a region hands back a C string with no copy. The node's extent +// is the header plus its bytes, rounded to the arena's alignment like every +// node's; on the wire it is a record whose body is the bytes (§3.1). +struct TableBlob +{ + uint32_t length; + uint32_t zero; +}; + +static const int64_t kTableBlobHeader = 8; // length (u32), then four zero bytes +static const int64_t kTableBlobMaxLength = 0xFFFFFFFF; // a record's length is a u32 (§3.1) + +// the node's storage: the header, the bytes, a string's terminator, rounded +// to the arena's alignment like every node +inline int64_t TableBlobStorage( int64_t length, bool terminated ) +{ + return TableAlignUp64( kTableBlobHeader + length + ( terminated ? 1 : 0 ) ); +} + +// What a read answers: a pointer INTO the region and the length, NULL and +// zero for a null slot. Off a locked region, a loaded one or an opened cook +// the pointer is one add from the slot, and nothing is copied. +struct TableBytesView +{ + const uint8_t * data; + int64_t length; +}; + +struct TableStringView +{ + const char * data; // zero-terminated + int64_t length; +}; + +// What AllocBytes and AllocString hand back: the bytes to write through, the +// length asked for, and the reference to store in the slot — the three +// answers TableSlot gives for a table node. +struct TableBytesSlot +{ + uint8_t * data = NULL; + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +struct TableStringSlot +{ + char * data = NULL; // room for length bytes and the terminator, already zero + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +// ---- the arena: segmented, slab-handed, lock-free by ownership ---- +// +// Allocation is thread-local inside a worker's slab — no atomics on the node +// path. A worker takes its next slab with ONE compare-exchange, and a new +// segment is published with one more. Nothing ever moves: a segment, once +// allocated, lives untouched until the arena is torn down, so a T* obtained +// from Alloc stays valid while other workers allocate, and an offset stays +// correct while the arena grows. +// +// The model this DELIBERATELY refuses: one buffer under a lock, grown by +// realloc. A realloc moves the buffer under workers mid-write; offsets fix +// identity but not the raw references already resolved from them, and the +// resulting corruption is invisible until much later. Segments never move, so +// that bug class cannot be written here. +// +// Slack: at most one slab tail per worker plus one slab per segment (a slab +// that will not fit is skipped rather than split), i.e. under 2% of a segment +// plus threads x 64 KiB. That is the price of never synchronizing per node. +struct TableArena +{ + std::atomic segments[ kTableMaxSegments ]; + std::atomic cursor; // (segment << kTableSegmentBits) | bytes handed out + bool locked = false; // MONOTONIC: Lock() is one-way, there is no unlock + // THE ARENA CARRIES ITS OWN, so everything downstream of a builder — + // segments, pack map, numbering, region, node directory — allocates through + // the one pair the caller named, with nothing to thread by hand. + TableAllocator allocator; +}; + +inline void TableArenaInit( TableArena & arena, TableAllocator allocator ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + arena.segments[i].store( NULL, std::memory_order_relaxed ); + } + arena.cursor.store( 0, std::memory_order_relaxed ); + arena.locked = false; + arena.allocator = allocator; +} + +inline void TableArenaShutdown( TableArena & arena ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + uint8_t * segment = arena.segments[i].exchange( NULL, std::memory_order_acq_rel ); + if ( segment != NULL ) { arena.allocator.free( arena.allocator.context, segment ); } + } + arena.cursor.store( 0, std::memory_order_relaxed ); +} + +// one L1 load plus an add: the segment table is 8 KiB and stays hot +inline uint8_t * TableArenaAt( const TableArena & arena, uint32_t offset ) +{ + return arena.segments[ offset >> kTableSegmentBits ].load( std::memory_order_relaxed ) + ( offset & kTableSegmentMask ); +} + +// TableArenaGrabSlab hands one worker its next private slab. Returns +// kTableAllocFailed when the arena's address space or the allocator is +// exhausted — a loud refusal, never a silent smaller slab. +inline uint32_t TableArenaGrabSlab( TableArena & arena ) +{ + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t segment = cursor >> kTableSegmentBits; + uint32_t used = cursor & kTableSegmentMask; + // strictly less: a slab is never split across segments, and the tail + // is the documented slack + if ( used + kTableSlabBytes < kTableSegmentSize ) + { + if ( arena.segments[segment].load( std::memory_order_acquire ) == NULL ) + { + // THE SEGMENT COMES BACK ZEROED, which is the allocator's + // contract and not an extra pass here: Lock copies whole nodes, + // PADDING INCLUDED, so anything uninitialized reaches a packed + // region. Value-initializing a node with placement new zeroes + // its MEMBERS and not its padding, so the zeroing has to happen + // at the segment or not at all. It costs nothing measurable: a + // fresh segment is untouched pages either way, and the default + // pair's calloc has the kernel hand them over zeroed. + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, (int64_t) kTableSegmentSize ); + if ( memory == NULL ) { return kTableAllocFailed; } + uint8_t * expected = NULL; + if ( !arena.segments[segment].compare_exchange_strong( expected, memory, std::memory_order_acq_rel ) ) + { + // another worker published this segment first + arena.allocator.free( arena.allocator.context, memory ); + } + } + if ( arena.cursor.compare_exchange_weak( cursor, cursor + kTableSlabBytes, std::memory_order_acq_rel ) ) + { + return ( segment << kTableSegmentBits ) | used; + } + continue; + } + uint32_t next_segment = segment + 1; + if ( next_segment >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + arena.cursor.compare_exchange_weak( cursor, next_segment << kTableSegmentBits, std::memory_order_acq_rel ); + } +} + +// TableArenaGrabSpan reserves a SPAN of the arena's address space for one node +// larger than a slab — a BYTE BUFFER of any size (docs/SPEC-TABLES.md §2.5) — +// and allocates it as one contiguous block. It takes whole segment indices +// from the cursor, starting at the index after the cursor's so nothing else +// is ever handed out inside the span, and publishes the block under the first +// of them; the indices the span covers past that one stay NULL, which is +// enough, because only a node's START is ever resolved through the segment +// table and a blob's bytes follow its header inside the one allocation. The +// unused tail of the segment the cursor was in is slack, like a slab tail. +// Returns kTableAllocFailed when the address space or the allocator is +// exhausted — a loud refusal, never a smaller blob. +inline uint32_t TableArenaGrabSpan( TableArena & arena, int64_t bytes ) +{ + if ( bytes <= 0 || bytes > ( (int64_t) kTableMaxSegments - 2 ) * (int64_t) kTableSegmentSize ) { return kTableAllocFailed; } + const uint32_t spanned = (uint32_t) ( ( bytes + kTableSegmentSize - 1 ) >> kTableSegmentBits ); + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t start = ( cursor >> kTableSegmentBits ) + 1; + if ( start + spanned >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + uint32_t next = ( start + spanned ) << kTableSegmentBits; + if ( !arena.cursor.compare_exchange_weak( cursor, next, std::memory_order_acq_rel ) ) { continue; } + // the span is this worker's now: nothing else can publish under its + // first index, so a plain store suffices, and the block comes back + // ZEROED like every segment — the blob's bytes and its tail are zeros + // until written + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, bytes ); + if ( memory == NULL ) { return kTableAllocFailed; } + arena.segments[start].store( memory, std::memory_order_release ); + return start << kTableSegmentBits; + } +} + +// ---- TableWorker: one thread's allocation front ---- +// +// The threading contract, stated plainly: +// * Alloc on YOUR OWN worker is safe concurrently with any other worker's. +// No locks, no atomics per node. +// * Writing fields of a node ANOTHER worker allocated is your own +// synchronization problem — this runtime does not arbitrate it. +// * Lock and Save are single-threaded: call them after the workers have +// joined. +struct TableWorker +{ + TableArena * arena = NULL; + uint32_t next = 0; + uint32_t end = 0; + + template TableSlot Alloc() + { + static_assert( alignof( T ) <= kTableAlign, "a table node's alignment must fit the arena's" ); + TableSlot slot; + if ( arena == NULL || arena->locked ) { return slot; } + uint32_t bytes = TableAlignUp( (uint32_t) sizeof( T ) ); + if ( bytes > kTableSlabBytes ) { return slot; } // a node larger than a slab: refused, never split + if ( end == 0 || next + bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return slot; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + uint32_t at = next; + next += bytes; + // A NODE IS BORN IN TWO HALVES: start its lifetime in the raw + // storage, then write the declared defaults ONE MEMBER AT A TIME. + // + // It is "T", not "T{}". Value-initialising the whole aggregate says + // the same thing and costs cl O(BYTES) TO COMPILE — it expands element + // by element in its front end — while both halves here cost + // O(declarations). The slab cap below refuses a large node at RUN + // TIME and bounds nothing at compile time: the cost is paid by + // whatever T a caller instantiates this with. + // Padding is not the difference: value-initialisation zeroes MEMBERS + // and not padding either way, which is why the segment is calloc'd. + // + // TableReset is an OVERLOAD SET, one per closure member, reached from + // this template by argument-dependent lookup on T's own namespace — + // Alloc is a template and cannot spell Reset. + // + // The reset is here because ONE DEFINITION SAYS WHAT THE DECLARED + // DEFAULTS ARE, and it is Reset. Default-initialisation lands on + // the same values today, because a member with a non-zero default + // carries a member initializer that says so — but that is the class + // definition agreeing with Reset, not the arena reading it, and #320's + // fix was itself a pass that MOVED initialisation between the two. + // The arena reads the definition. + slot.ptr = new ( TableArenaAt( *arena, at ) ) T; + TableReset( *slot.ptr ); + slot.ref.value = at; + return slot; + } + + // Alloc a BYTE BUFFER's node of exactly length bytes (docs/SPEC-TABLES.md + // §2.5): the blob header and its bytes, zeroed, in this thread's slab when + // it fits and in a span of the arena's own when it does not. NULL is the + // arena locked, a length below zero or past a record's u32, or the + // allocator refusing. The offset comes back for the reference. + TableBlob * AllocBlob( int64_t length, bool terminated, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( length < 0 || length > kTableBlobMaxLength ) { return NULL; } + const int64_t bytes = TableBlobStorage( length, terminated ); + if ( bytes > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, bytes ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + } + else + { + if ( end == 0 || next + (uint32_t) bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) bytes; + } + TableBlob * blob = (TableBlob *) TableArenaAt( *arena, at ); + blob->length = (uint32_t) length; // the bytes after it are the segment's zeros + blob->zero = 0; + return blob; + } + + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries + // no type id, takes no index and has no Reset, so it goes through the same + // slab and span the blob path uses rather than through Alloc. + uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( bytes <= 0 || align > (int64_t) kTableAlign ) { return NULL; } + const int64_t rounded = TableAlignUp64( bytes ); + if ( rounded > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, rounded ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + return TableArenaAt( *arena, at ); + } + if ( end == 0 || next + (uint32_t) rounded > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) rounded; + return TableArenaAt( *arena, at ); // the segment came back zeroed + } + // a *bytes node: the bytes to write through, and the reference to store + TableBytesSlot AllocBytes( int64_t length ) + { + TableBytesSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, false, at ); + if ( blob == NULL ) { return slot; } + slot.data = (uint8_t *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } + + // a *string node: room for length bytes and the zero byte after them + TableStringSlot AllocString( int64_t length ) + { + TableStringSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, true, at ); + if ( blob == NULL ) { return slot; } + slot.data = (char *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } +}; + +// ---- TablePackMap: the pack walk's identity map (docs/SPEC-TABLES.md §3.1, §6.2) ---- +// +// ONE ENTRY PER REACHABLE NODE, and that map IS identity: a node must know +// where it landed to be named a second time, so Lock packs a shared node ONCE +// and every later reference resolves to the one body it already has. That is +// the same first-visit numbering the wire uses, so the pack order and the node +// order are one order. +// +// COLOURING AN ENTRY WHILE ITS DESCENT IS OPEN COSTS ONE BIT, and it is what +// makes a data cycle free to refuse: a reference to an entry still open is a +// cycle, and Lock returns failure rather than recursing away. The ROOT's entry +// is open for the whole walk. +// +// The map is proportional to NODES, never to bytes, and it lives on the +// AUTHORING side, where §6.5 licenses allocation. Nothing on the reading path +// ever builds one. +struct TablePackEntry +{ + const void * key; // the node's address in the graph being packed + int64_t offset; // where that node landed in the region + uint8_t open; // its descent is still open: a reference here is a cycle +}; + +struct TablePackMap +{ + TablePackEntry * entries = NULL; + int64_t capacity = 0; // a power of two, or zero while empty + int64_t count = 0; + TableAllocator allocator; // the caller's, carried from the walk that built it +}; + +inline void TablePackMapInit( TablePackMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TablePackMapShutdown( TablePackMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TablePackMapInit( map, map.allocator ); +} + +// The two walks behind Lock re-derive the SAME map from the same graph — the +// numbering is never carried between them (§3.1) — so the second starts from +// an empty map and keeps the capacity the first paid for. +inline void TablePackMapReset( TablePackMap & map ) +{ + if ( map.entries != NULL ) { memset( map.entries, 0, (size_t) map.capacity * sizeof( TablePackEntry ) ); } + map.count = 0; +} + +// open addressing, linear probing, a multiply-shift hash over the address: a +// node key is a pointer and its low bits are alignment, so the low bits alone +// would collide on every node of one type +inline int64_t TablePackMapSlot( const TablePackMap & map, const void * key ) +{ + uint64_t hash = (uint64_t) (uintptr_t) key; + hash *= 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != NULL && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TablePackEntry * TablePackMapFind( TablePackMap & map, const void * key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TablePackEntry * entry = &map.entries[ TablePackMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +// QUADRUPLING, not doubling, and the reason is measured: growth rehashes every +// entry, and on a graph of 131,071 nodes the doubling schedule spent 45% of +// Lock in rehashing alone. Quadrupling from 1024 buys 1.35x on that graph and +// keeps the map NODE-proportional (§6.2) — under 128 bytes a node at its +// worst, right after a grow, and about 64 on average. +inline bool TablePackMapGrow( TablePackMap & map ) +{ + TablePackMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 1024; + grown.entries = (TablePackEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TablePackEntry ) ); + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == NULL ) { continue; } + grown.entries[ TablePackMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// REACH a node: one probe answers both questions the walk has. A true "taken" +// says this is a FIRST visit, and the entry is now the node's, coloured open +// at "offset"; otherwise the entry is the one the node already has, and its +// open bit says cycle or sharing. NULL is an allocation failure, and it is a +// refusal like any other: Lock fails rather than packing a graph it cannot +// track. +// +// It is one call and not a find followed by an insert because the walk asks +// this question twice per node — once to measure, once to pack — and every +// probe is a miss into a table larger than L2. +inline TablePackEntry * TablePackMapReach( TablePackMap & map, const void * key, int64_t offset, bool & taken, int64_t & slot ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TablePackMapGrow( map ) ) { return NULL; } + } + slot = TablePackMapSlot( map, key ); + TablePackEntry * entry = &map.entries[slot]; + taken = entry->key != key; // an empty slot is a first visit; the key is never NULL + if ( taken ) + { + entry->key = key; + entry->offset = offset; + entry->open = 1; + map.count++; + } + return entry; +} + +// The descent finished: the node keeps its entry — identity outlives the +// descent — and stops being a cycle. The "hint" is the slot Reach returned, and it +// is checked against the key rather than trusted, so a rehash between the two +// costs a second probe instead of correctness. +inline void TablePackMapClose( TablePackMap & map, const void * key, int64_t hint ) +{ + if ( hint >= 0 && hint < map.capacity && map.entries[hint].key == key ) + { + map.entries[hint].open = 0; + return; + } + TablePackEntry * entry = TablePackMapFind( map, key ); + if ( entry != NULL ) { entry->open = 0; } +} + +// ---- resolution contexts: which encoding a walk is reading ---- + +struct TableArenaCtx { const TableArena * arena; }; +struct TableRegionCtx {}; + +// ---- a BYTE BUFFER's resolution (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// The same two encodings a table pointer has, resolved the same way: a +// self-relative delta in a region — one add, no base — and an arena offset +// while the builder is mutable. The blob is reached through its header, and a +// view is the header plus eight and the header's first word. Nothing here +// allocates and nothing copies: off a locked region, a loaded one or an +// opened cook the view points INTO the region. +inline const TableBlob * TableBlobAt( const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableRegionCtx &, const TableRef & ref ) { return TableBlobAt( ref ); } +inline const TableBlob * TableBlobAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} + +inline TableBytesView TableBytesViewOf( const TableBlob * blob ) +{ + TableBytesView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const uint8_t *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} +inline TableStringView TableStringViewOf( const TableBlob * blob ) +{ + TableStringView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const char *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} + +// the const form's hot path: one add, no base +inline TableBytesView TableBytesAt( const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ref ) ); } +inline TableStringView TableStringAt( const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ref ) ); } +// and the context forms a walk uses: a region context, an arena context, or +// the arena itself while the builder is mutable +template inline TableBytesView TableBytesAt( const Ctx & ctx, const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ctx, ref ) ); } +template inline TableStringView TableStringAt( const Ctx & ctx, const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ctx, ref ) ); } + +// allocate a blob in the arena and point the slot at it; the slot holds the +// arena offset, as every slot does while the builder is mutable +inline uint8_t * TableBytesEmplace( TableWorker & worker, TableRef & slot, int64_t length ) +{ + TableBytesSlot allocated = worker.AllocBytes( length ); + slot = allocated.ref; + return allocated.data; +} +// the text is copied in when one is given; a NULL text leaves the zeros for +// the caller to fill +inline char * TableStringEmplace( TableWorker & worker, TableRef & slot, const char * text, int64_t length ) +{ + TableStringSlot allocated = worker.AllocString( length ); + slot = allocated.ref; + if ( allocated.data != NULL && text != NULL && length > 0 ) { memcpy( allocated.data, text, (size_t) length ); } + return allocated.data; +} + +// ---- the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table, and a +// pointer field rides as an INDEX into it under kind 17. The encoding is +// flat: no pointer edge is a nesting level, so a chain's length is not a depth, +// and two references to one node are one node. +// +// THE FIELD RIDES ONCE: an L with sixty-four bits of capability frames a +// numbering of any size, so the whole numbering is one contiguous payload and a +// save's node bodies have no aggregate ceiling. + +static const uint64_t kTableNodeIndexNull = 0; // absence and null are one value +static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts the table + +// The not-materialized sentinel (§6.3): a record whose type id this build could +// not name. Distinct from every real offset including the root's 0, so an index +// resolving through it yields NULL and can never fabricate the root. +static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; + +// ---- the numbering, on the SAVE side ---- +// +// One entry per reachable node in FIRST-VISIT order, so entry k is node index +// k + 2. The two thunks are what let one loop write a table of mixed types: the +// numbering walk knows each target's type STATICALLY at the site it numbers it, +// so it stores the instantiation there and the loop never asks what a node is. +struct TableNumbering; + +struct TableNodeEntry +{ + const void * node; + uint64_t type_id; + // the type id's MESSAGE-FORM SLOT (docs/SPEC-TABLES.md §3.3), stored where + // the numbering walk stores the id itself and for the same reason: the + // target's type is known STATICALLY at the site that numbers it, so a + // form 2 save reads the slot out of the entry instead of looking an id up. + // Every pointer target's type id is an entry of the announcement, which is + // what makes the slot a compile-time fact of a POINTERED message too. + uint64_t type_slot; + int64_t ( * measure )( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ); + bool ( * save )( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ); +}; + +struct TableNumbering +{ + TablePackMap seen; // node -> index; the ROOT is index 1, open for the whole walk + TableNodeEntry * entries = NULL; + int64_t count = 0; + int64_t capacity = 0; +}; + +// The numbering allocates through the map's pair rather than carrying a second +// copy of it: one numbering is one walk, and a walk has one allocator. +inline void TableNumberingInit( TableNumbering & n, TableAllocator allocator ) +{ + TablePackMapInit( n.seen, allocator ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +inline void TableNumberingShutdown( TableNumbering & n ) +{ + TableAllocator allocator = n.seen.allocator; + TablePackMapShutdown( n.seen ); + allocator.free( allocator.context, n.entries ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +// The index a numbered node was given, for the save that writes it into a +// pointer slot. False means the two walks disagree about the graph, which is a +// refusal and never a guess. +inline bool TableNumberingIndex( const TableNumbering & n, const void * node, uint64_t & index ) +{ + if ( n.seen.capacity == 0 ) { return false; } + const TablePackEntry & entry = n.seen.entries[ TablePackMapSlot( n.seen, node ) ]; + if ( entry.key != node ) { return false; } + index = (uint64_t) entry.offset; + return true; +} + +inline bool TableNumberingAppend( TableNumbering & n, const TableNodeEntry & entry ) +{ + if ( n.count == n.capacity ) + { + // GROW BY COPY, never by realloc: the allocator hook is a PAIR, and a + // game's heap is not required to have a resize primitive at all. The + // schedule quadruples, so the copying is amortized to a constant per + // entry and the growth is the same growth it always was. + int64_t capacity = n.capacity != 0 ? n.capacity * 4 : 256; + TableAllocator allocator = n.seen.allocator; + TableNodeEntry * grown = (TableNodeEntry *) allocator.alloc( allocator.context, capacity * (int64_t) sizeof( TableNodeEntry ) ); + if ( grown == NULL ) { return false; } + if ( n.entries != NULL ) + { + memcpy( grown, n.entries, (size_t) n.count * sizeof( TableNodeEntry ) ); + allocator.free( allocator.context, n.entries ); + } + n.entries = grown; + n.capacity = capacity; + } + n.entries[n.count++] = entry; + return true; +} + +// The thunks the numbering stores. Each resolves to the closure member's own +// MeasureBody / SaveBodyFields through an overload set in the member's DECLARING +// file, reached by argument-dependent lookup at instantiation — the same bridge +// the arena's TableReset uses, and the reason a numbering may span the files of +// one unit without any file naming another's members. +template +inline int64_t TableNodeMeasureThunk( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ) +{ + return TableNodeMeasure( *(const Ctx *) ctx, numbering, ids, *(const T *) node ); +} + +template +inline bool TableNodeSaveThunk( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ) +{ + return TableNodeSave( *(const Ctx *) ctx, numbering, w, ids, *(const T *) node ); +} + +// ---- a BYTE BUFFER's record (docs/SPEC-TABLES.md §2.5, §3.1) ---- +// +// A blob rides as a node record under one of two RESERVED type ids — the fold +// a table's name takes, over the keywords "bytes" and "string", which no table +// can be named — with the bytes as its body and nothing framed inside. These +// two thunks are what the numbering stores for a blob, as it stores a +// member's codec for a table: the length, and the bytes verbatim. +static const uint64_t kTableBytesTypeId = 0x2f2ec0474f1c4fe4ull; // fnv1a64( "bytes" ) +static const uint64_t kTableStringTypeId = 0x704be0d8faaffc58ull; // fnv1a64( "string" ) + +template +inline int64_t TableBlobMeasureThunk( const void *, const TableNumbering &, TableIds &, const void * node ) +{ + return (int64_t) ( (const TableBlob *) node )->length; +} + +template +inline bool TableBlobSaveThunk( const void *, const TableNumbering &, TableWriter & w, TableIds &, const void * node ) +{ + const TableBlob * blob = (const TableBlob *) node; + w.raw( (const void *) ( blob + 1 ), (int64_t) blob->length ); + return true; +} + +// TableNodeTableMeasure and TableNodeTableSave are the framing, and they are +// ONE fill rule written twice — measure derives it from the graph and save +// derives the same one, which is what makes measure == save hold across a +// pointer graph (§3.1). +// +// The field rides ONCE, under the reserved id, kind 12: the payload opens with +// the count and then carries the records back to back, each a type id +// REFERENCE, a length and a body. The reserved id is interned BEFORE the +// records, and a record's type id before its body, which is the first-use order +// the trailer is written in (§3). +template +inline int64_t TableNodeTablePayload( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + int64_t payload = TableLebBytes( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + payload += TableLebBytes( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return -1; } + payload += TableLebBytes( (uint64_t) body ) + body; + } + return payload; +} + +template +inline int64_t TableNodeTableMeasure( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return 0; } // a root that reaches no nodes writes none of them + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return -1; } + return TableLebBytes( ref ) + 1 + TableLebBytes( (uint64_t) payload ) + payload; +} + +template +inline bool TableNodeTableSave( const Ctx & ctx, TableWriter & w, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return true; } + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return false; } + w.putleb( ref ); + w.put8( 12 ); // kind 12 is the opaque byte payload: a reader that cannot name the id skips by L + w.putleb( (uint64_t) payload ); + w.putleb( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + w.putleb( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return false; } + w.putleb( (uint64_t) body ); + if ( !n.entries[k].save( (const void *) &ctx, n, w, ids, n.entries[k].node ) ) { return false; } + } + return true; +} + +// ---- the numbering, on the LOAD side: a region's NODE DIRECTORY (§6.3) ---- +// +// The wire's numbering made resident: one entry per numbered node, in index +// order, position i describing node index i + 1 — so position 0 is the ROOT at +// offset 0. It is ATTRIBUTION, and attribution is separable: nothing that reads +// a structure touches it, a deref is one add on a self-relative offset, and a +// caller may release it once Load returns. +struct TableNodeDirEntry +{ + uint64_t offset; + uint64_t type_id; +}; + +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; + +// TableNodeMap is what a pointer slot resolves through while a body decodes. +struct TableNodeMap +{ + uint8_t * base = NULL; + const TableNodeDirEntry * entries = NULL; + int64_t count = 0; // the ROOT's entry included, so it is records + 1 + bool good = false; // the node table read whole; a numbering that failed resolves nothing + // WHERE THE NODES LIVE, and therefore what a resolved slot holds: a region + // takes the SELF-RELATIVE delta so a deref is one add, and the tool's + // builder path takes the node's ARENA OFFSET (§6.3). + bool arena = false; + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. + TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; +}; + +// TableNodeResolve places one node index in a pointer slot, and every failure +// is one of §4's events with the pointer left null. The declared TARGET type id +// is checked at every index, the root's included: the root carries no record +// and therefore no wire type id, so the READER'S OWN root type is what the +// claim is checked against. +inline void TableNodeResolve( const TableNodeMap & map, TableRef & slot, uint64_t index, uint64_t target, TableReport * report ) +{ + slot.value = 0; + if ( index == kTableNodeIndexNull || !map.good ) { return; } + if ( index - 1 >= (uint64_t) map.count ) + { + report->malformed = true; // an index above node_count + 1 + return; + } + const TableNodeDirEntry & entry = map.entries[index - 1]; + if ( entry.offset == kTableNodeAbsent ) + { + // a node whose type id this build could not name KEEPS ITS INDEX, and + // every pointer naming it reads null. The unknown was counted once, at + // the node, not once per pointer. + return; + } + if ( entry.type_id != target ) + { + report->kind_mismatch++; + return; + } + slot.value = map.arena ? (int64_t) entry.offset + : (int64_t) ( ( map.base + entry.offset ) - (const uint8_t *) &slot ); +} + +// ---- the record SCAN, and it is the whole of load's bound (§3.1) ---- +// +// Reading follows no reference. The scan walks the root body's top-level fields, +// finds the ONE under the reserved id, and reads records out of its payload in +// order — the field rides once, so nothing is copied to make a body contiguous +// and the generated body decoder never learns the transport exists. +struct TableNodeScan +{ + TableReader fields; // over the ROOT body, skipping past everything else + const uint8_t * payload; // the node-table field's payload + int64_t payload_size; + int64_t payload_offset; + bool opened; // the root body has been walked for the field + uint64_t declared; + int64_t records; + bool present; // the root body carries a node table at all + bool malformed; + const TableIdTable * ids; +}; + +inline TableNodeScan TableNodeScanBegin( const uint8_t * body, int64_t size, TableReport * report, const TableIdTable * ids ) +{ + TableNodeScan s = { TableReader( body, size, report, ids ), NULL, 0, 0, false, 0, 0, false, false, ids }; + return s; +} + +// find the node-table field, or answer false when the root body has none. A +// body carrying an id more than once is legal input and THE LAST OCCURRENCE +// WINS (docs/SPEC-TABLES.md §3), so the walk runs to the terminator and keeps +// the last rather than stopping at the first. +inline bool TableNodeScanOpen( TableNodeScan & s ) +{ + if ( s.opened ) { return false; } + s.opened = true; + for ( ;; ) + { + uint64_t ref = 0; + if ( !s.fields.getleb( ref ) ) { break; } + if ( ref == 0 ) { break; } // the terminator + if ( s.ids == NULL || ref > (uint64_t) s.ids->count ) { break; } + const uint64_t id = s.ids->at( ref ); + if ( !s.fields.has( 1 ) ) { break; } + const uint8_t kind = s.fields.get8(); + if ( id == kTableNodeTableFieldId ) + { + s.present = true; + if ( kind != 12 ) { s.malformed = true; return false; } + uint64_t length = 0; + if ( !s.fields.getleb( length ) || !s.fields.room( length ) ) { s.malformed = true; return false; } + s.payload = s.fields.buffer + s.fields.offset; + s.payload_size = (int64_t) length; + s.fields.offset += (int64_t) length; + continue; + } + if ( !s.fields.skip( kind ) ) { break; } + } + if ( s.payload == NULL ) { return false; } + TableReader head( s.payload, s.payload_size, s.fields.report, s.ids ); + if ( !head.getleb( s.declared ) ) { s.malformed = true; return false; } + s.payload_offset = head.offset; + return true; +} + +// the next record, or false at the end of the table — s.malformed says whether +// the end was the end or the framing giving out +inline bool TableNodeScanNext( TableNodeScan & s, uint64_t & type_id, const uint8_t * & body, int64_t & length ) +{ + if ( !s.opened && !TableNodeScanOpen( s ) ) { return false; } + if ( s.payload == NULL || s.payload_offset >= s.payload_size ) { return false; } + TableReader rec( s.payload, s.payload_size, s.fields.report, s.ids ); + rec.offset = s.payload_offset; + uint64_t ref = 0; + if ( !rec.getleb( ref ) || ref == 0 || s.ids == NULL || ref > (uint64_t) s.ids->count ) + { + s.malformed = true; // a type id reference of 0, or one past the table + return false; + } + type_id = s.ids->at( ref ); + uint64_t declared_length = 0; + if ( !rec.getleb( declared_length ) ) + { + s.malformed = true; // a record whose length is damaged + return false; + } + if ( declared_length > (uint64_t) ( s.payload_size - rec.offset ) ) + { + s.malformed = true; // a record whose length runs past its field + return false; + } + body = s.payload + rec.offset; + length = (int64_t) declared_length; + s.payload_offset = rec.offset + length; + s.records++; + return true; +} + +// The record scan is AUTHORITATIVE: node_count is data from the wire, and a +// count that disagrees with the scan is malformed. Nothing is sized from it +// before the scan has confirmed it. +inline bool TableNodeScanWhole( TableNodeScan & s ) +{ + if ( s.malformed ) { return false; } + if ( !s.present ) { return true; } // no node table at all is not a broken one + return s.declared == (uint64_t) s.records; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_ARENA + +#ifndef LISTDEMO_SCHEMA_TABLE_EXTENT +#define LISTDEMO_SCHEMA_TABLE_EXTENT + +namespace listdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_EXTENT + +#ifndef LISTDEMO_SCHEMA_TABLE_MAP +#define LISTDEMO_SCHEMA_TABLE_MAP + +namespace listdemo { + +// ---- a MAP: a sorted entry array, and the lookup over it (§2.8) ---- +// +// On the wire, in a region and in a cook a map is an array of one generated +// ENTRY table held in ascending key order. What this adds is Find — a binary +// search over that array where it lies — and a builder that inserts, replaces +// and erases by key. Nothing here is stored: a region and a cook carry the +// array and the count, and not one byte about a hash or a probe. + +// entries carved from ONE call to the allocator pair; a new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableMapSegmentEntries = 32; + +// TableDeclRef names a type in an unevaluated context and is never defined — +// what 's declval is for, without the include the generated corpus +// refuses to pay for (the iterator_traits note, §13.9). +template T & TableDeclRef(); + +// THE ORDER IS TOTAL, AND IT IS THE SAME IN NINE LANGUAGES (§2.8). Integers +// compare by VALUE, signed for the signed kinds and unsigned for the unsigned. +// Strings compare by BYTES, unsigned, a shorter string that is a prefix of a +// longer one first: memcmp over the common length, then the lengths. Never a +// locale, never a code point, never a case fold. +inline int TableKeyOrder( uint64_t a, uint64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( int64_t a, int64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( const char * a, int32_t a_length, const char * b, int32_t b_length ) +{ + const int32_t common = a_length < b_length ? a_length : b_length; + if ( common > 0 ) + { + const int order = memcmp( (const void *) a, (const void *) b, (size_t) common ); + if ( order != 0 ) { return order < 0 ? -1 : 1; } + } + return a_length < b_length ? -1 : ( a_length > b_length ? 1 : 0 ); +} + +// the length of a NUL-terminated key at a call site, bounded by the storage it +// has to fit: a key one byte longer than the bound is refused, never truncated +inline int32_t TableKeyLength( const char * key, int32_t bound ) +{ + if ( key == NULL ) { return 0; } + for ( int32_t i = 0; i <= bound; i++ ) { if ( key[i] == 0 ) { return i; } } + return bound + 1; // longer than the bound: the caller refuses it +} + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.8, §7.2) ---- +// +// An int64 self-relative reference to the entry array and an int32 count, then +// padding to eight. The reference is a TableRef like a pointer's: in the arena +// it names the builder's HEAD, in a region it is the delta from the slot to +// the first entry, and 0 is the empty map in both. +template struct TableMap +{ + TableRef entries; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Entry * Entries() const + { + return entries.value != 0 ? (const Entry *) ( (const uint8_t *) &entries + entries.value ) : NULL; + } + int32_t size() const { return count; } + + // FIND: floor( log2 n ) + 1 key compares, in place, no allocation. NULL + // when absent, and on a map[K]*T the RESOLVED pointer, which is what a + // pointer field's accessor answers. + template const Entry * FindEntry( Key key ) const + { + const Entry * base = Entries(); + int32_t low = 0, high = count; + while ( low < high ) + { + const int32_t mid = low + ( high - low ) / 2; + const int order = TableEntryOrder( base[mid], key ); + if ( order == 0 ) { return base + mid; } + if ( order < 0 ) { low = mid + 1; } else { high = mid; } + } + return NULL; + } + // the return type is DEDUCED, so it is worked out when a call site + // instantiates Find and not when the holder's record declares the slot — + // which is what lets the entry's own overloads be declared after it + template auto Find( Key key ) const + { + return TableEntryFound( FindEntry( key ) ); + } + + // ---- iteration: ASCENDING key order, the key beside the value ---- + // + // A proxy BY VALUE, the keyed array's shape (§2.4): for ( auto [ key, + // value ] : map ). It carries no iterator_traits, for the reason + // TableKeyed's does not (§13.9). + struct ConstEntry + { + decltype( TableEntryKey( TableDeclRef() ) ) key; + decltype( TableEntryFound( (const Entry *) NULL ) ) value; + }; + + struct ConstIterator + { + const Entry * at; + ConstEntry operator*() const { return ConstEntry{ TableEntryKey( *at ), TableEntryFound( at ) }; } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Entries() }; } + ConstIterator end() const { return ConstIterator{ Entries() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.8, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first entry is inserted. Each +// segment is a fixed number of entries carved from one call to the allocator +// pair. An entry's address is stable for the arena's life, so a value handed +// back by an insert stays valid while other entries arrive. +struct TableMapHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an insert appends into + int32_t live; + int32_t dead; +}; + +template struct TableMapSegment +{ + TableRef next; + int32_t used; // entries carved from this segment + int32_t padding; + uint32_t dead[ ( kTableMapSegmentEntries + 31 ) / 32 ]; // Erase marks one bit, never the entry + Entry entries[ kTableMapSegmentEntries ]; +}; + +inline bool TableMapSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// ---- the ORDERED CURSOR the four writing walks read (§2.8) ---- +// +// Measure, Save, Lock and Cook each write a map's entries in ascending key +// order with no key twice, deriving the order from the builder's entries as +// each walk derives the numbering (§3.1). Nothing passes between them, so +// measure == save over a map is a real check on two sorts agreeing. +// +// A REGION is already sorted, so its cursor is the array in place and +// allocates nothing. The BUILDER's is the sort: an array of entry pointers +// allocated through the pair and released before the walk returns, because +// sorting the segments themselves would move entries whose addresses a caller +// holds. +template struct TableMapCursor +{ + const Entry * const * order = NULL; // the builder's form: sorted pointers + const Entry * entries = NULL; // the region's form: the array in place + int32_t count = 0; + TableAllocator allocator; + bool ok = false; + const Entry * operator[]( int32_t index ) const + { + return order != NULL ? order[index] : entries + index; + } +}; + +// heapsort: O( n log n ) once per map, no recursion, no allocation past the +// pointer array the caller already paid for +template inline void TableMapSort( const Entry ** order, int32_t count ) +{ + for ( int32_t start = count / 2 - 1; start >= 0; start-- ) + { + int32_t root = start; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= count ) { break; } + if ( child + 1 < count && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * swap = order[root]; order[root] = order[child]; order[child] = swap; + root = child; + } + } + for ( int32_t end = count - 1; end > 0; end-- ) + { + const Entry * swap = order[0]; order[0] = order[end]; order[end] = swap; + int32_t root = 0; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= end ) { break; } + if ( child + 1 < end && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * hold = order[root]; order[root] = order[child]; order[child] = hold; + root = child; + } + } +} + +// the REGION form: the array is already sorted, so the cursor is the array +template +inline TableMapCursor TableMapOrder( const TableRegionCtx &, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.entries = map.Entries(); + cursor.count = map.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: gather the LIVE entries out of the segment chain in +// insertion order, then sort. A dead entry costs nothing on any wire (§2.8). +template +inline TableMapCursor TableMapOrder( const TableArena & arena, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.allocator = arena.allocator; + cursor.count = map.count; + if ( map.entries.value == 0 || map.count <= 0 ) { cursor.ok = map.count == 0; cursor.count = 0; return cursor; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + if ( head->live != map.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + const Entry ** order = (const Entry **) arena.allocator.alloc( arena.allocator.context, (int64_t) map.count * (int64_t) sizeof( const Entry * ) ); + if ( order == NULL ) { return cursor; } + int32_t at = 0; + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 && at < map.count ) + { + const TableMapSegment * segment = (const TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used && at < map.count; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + order[at++] = segment->entries + i; + } + segment_ref = segment->next; + } + if ( at != map.count ) + { + arena.allocator.free( arena.allocator.context, order ); + return cursor; + } + TableMapSort( order, map.count ); + cursor.order = order; + cursor.ok = true; + return cursor; +} + +template +inline TableMapCursor TableMapOrder( const TableArenaCtx & ctx, const TableMap & map ) +{ + return TableMapOrder( *ctx.arena, map ); +} + +template inline void TableMapRelease( TableMapCursor & cursor ) +{ + if ( cursor.order != NULL ) { cursor.allocator.free( cursor.allocator.context, (void *) cursor.order ); } + cursor.order = NULL; +} + +// ---- the builder's five (§2.8) ---- +// +// Insert APPENDS after one LINEAR SCAN of the live entries for the key it may +// replace, Find is that same scan, and Erase is the scan and one bit. The +// builder builds NO INDEX, and that is a rule: the sort happens once, at Lock, +// Save or Cook, and every lookup that matters runs over the sorted region. + +// the head, allocated when the first entry is inserted +template +inline TableMapHead * TableMapReach( TableWorker & worker, TableMap & map ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( map.entries.value != 0 ) { return (TableMapHead *) TableArenaAt( *worker.arena, (uint32_t) map.entries.value ); } + uint32_t at = 0; + TableMapHead * head = (TableMapHead *) worker.AllocRaw( (int64_t) sizeof( TableMapHead ), (int64_t) alignof( TableMapHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + map.entries.value = (int64_t) at; + return head; +} + +// one entry's storage, appended: the current segment when it has room, a new +// one carved from one call to the pair when it does not +template +inline Entry * TableMapAppend( TableWorker & worker, TableMapHead * head, TableMap & map ) +{ + TableMapSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableMapSegmentEntries ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableMapSegment *) worker.AllocRaw( (int64_t) sizeof( TableMapSegment ), (int64_t) alignof( TableMapSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableMapSegment * previous = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Entry * entry = segment->entries + segment->used; + segment->used++; + head->live++; + map.count++; + return entry; +} + +// the LINEAR SCAN: the live entries in insertion order, O( n ) key compares +template +inline Entry * TableMapScan( const TableArena & arena, const TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) == 0 ) { return segment->entries + i; } + } + segment_ref = segment->next; + } + return NULL; +} + +// ERASE marks the entry DEAD, one bit in the segment's slot and not in the +// entry table, and decrements the live count. Its storage is reclaimed at +// RESET and never reused mid-build, because reusing a slot would make "an +// entry's address is stable" false for exactly one case. +template +inline bool TableMapErase( TableArena & arena, TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return false; } + TableMapHead * head = (TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) != 0 ) { continue; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + map.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INSERTION order, live entries only (§2.8) ---- +template struct TableMapEach +{ + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableMapSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableMapSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + auto operator*() const { return TableEntryEach( segment->entries + index ); } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableMapEach TableMapEachOf( const TableArena & arena, const TableMap & map ) +{ + TableMapEach each = { &arena, TableRef() }; + if ( map.entries.value != 0 ) + { + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + each.first = head->first; + } + return each; +} + +// ---- the LOAD side: where a decoded entry lands (§2.8) ---- +// +// THE READER TRUSTS NOTHING and spends one compare per entry. Every load path +// applies the same rules and produces one report (§4), so the region load of +// §6.5 and LoadBuilder never disagree about a wire. These two shapes are what +// makes that true with one generated decoder: a REGION carves the entry array +// out of the holder node's own extent, and the TOOL's path appends into the +// builder's arena, and the decoder above them cannot tell which it has. + +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. + +// TableMapFill is one map field being decoded: where the next entry lands, and +// the entry that last LANDED, which is what the ascending check compares +// against. +template struct TableMapFill +{ + TableMap * map = NULL; + Entry * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; +}; + +template +inline TableMapFill TableMapFillBegin( const TableNodeMap & nodes, TableMap & map, uint32_t n ) +{ + TableMapFill fill; + fill.map = ↦ + map.entries.value = 0; + map.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Entry ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Entry ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Entry *) base; + fill.capacity = (int32_t) n; + map.entries.value = (int64_t) ( base - (const uint8_t *) &map.entries ); + fill.ok = true; + return fill; +} + +// the entry that last LANDED — NULL before the first +template inline Entry * TableMapFillLast( TableMapFill & fill ) +{ + if ( fill.map->count <= 0 ) { return NULL; } + if ( fill.array != NULL ) { return fill.array + ( fill.map->count - 1 ); } + return TableMapLive( *fill.worker->arena, *fill.map, fill.map->count - 1 ); +} + +// the next slot, at the entry type's declared defaults +template inline Entry * TableMapFillNext( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + if ( fill.map->count >= fill.capacity ) { return NULL; } + Entry * entry = fill.array + fill.map->count; + TableReset( *entry ); + fill.map->count++; + return entry; + } + TableMapHead * head = TableMapReach( *fill.worker, *fill.map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( *fill.worker, head, *fill.map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// A MAP WITH HALF ITS KEYS IS NOT A MAP (§2.8): at the first entry whose key +// kind disagrees with the reader's declaration the map resets to EMPTY, one +// kind_mismatch is counted for the map, and its remaining bytes are skipped. +template inline void TableMapFillReset( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + fill.map->entries.value = 0; + fill.map->count = 0; + return; + } + if ( fill.map->entries.value != 0 ) + { + TableMapHead * head = (TableMapHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.map->entries.value ); + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + } + fill.map->count = 0; +} + +// an EMPTY map's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableMapFillEnd( TableMapFill & fill ) +{ + if ( fill.array != NULL && fill.map->count == 0 ) { fill.map->entries.value = 0; } +} + +// the k-th LIVE entry of a builder map, in insertion order — what the tool +// path's ascending check compares against +template +inline Entry * TableMapLive( const TableArena & arena, const TableMap & map, int32_t index ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + int32_t at = 0; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( at == index ) { return segment->entries + i; } + at++; + } + segment_ref = segment->next; + } + return NULL; +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.8, §6.5) ---- +// +// LoadMeasure's term for a map is N x sizeof( Entry ) rounded to +// alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this +// reads no field: it walks the map's own header and, where an entry's value +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. +// A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its +// own L and the body's terminator, and under this form's variable lengths that +// footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a +// map's L can carry, and therefore what a LoadMeasure may be asked for. +static const int64_t kTableMapEntryFloor = 2; + +inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry + at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); + at += (int64_t) n * entry_size; + if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// ---- the TEXT form's placement (docs/SPEC-TABLES.md §2.8, §16) ---- +// +// The text is a plain JSON object keyed by the KEY, and the generic walk fills +// it through the ENTRY'S OWN descriptor — so all it needs from here is one +// entry at one key, handed back at its defaults. It is the builder's Insert +// with the ENTRY returned rather than its value, because the walk writes the +// value through a field row and not through a typed pointer. +template +inline Entry * TableMapPlace( TableWorker & worker, TableMap & map, Key key ) +{ + if ( worker.arena == NULL ) { return NULL; } + Entry * found = TableMapScan( *worker.arena, map, key ); + if ( found != NULL ) + { + TableResetMapValue( *found ); // a repeated key is LAST-WINS, whole + return found; + } + TableMapHead * head = TableMapReach( worker, map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( worker, head, map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// ---- the OPTIONAL RUNTIME INDEX (§2.8) ---- +// +// Open addressing with LINEAR PROBING over the sorted array, built AT LOAD for +// a map large enough that log n compares over a cold array cost more than one +// hash and a probe. IT IS NEVER STORED: the caller measures it, owns its +// storage, builds it in one pass and releases it whenever. +// +// ITS HASH AND ITS LOAD FACTOR ARE NOT A CROSS-PORT CONTRACT, and that is a +// rule. What a port is held to is the CONTRACT of the lookup: the same value +// the sorted array's Find returns for the same key, and no allocation past the +// storage the caller handed in. +struct TableMapIndex +{ + int32_t * slots = NULL; // entry indices, +1; 0 is an empty slot + int32_t capacity = 0; + bool good = false; +}; + +// this runtime's own, and no port reproduces it: fnv1a64 over the key's bytes +inline uint64_t TableMapHash( const void * bytes, int32_t length ) +{ + uint64_t hash = 0xCBF29CE484222325ull; + const uint8_t * at = (const uint8_t *) bytes; + for ( int32_t i = 0; i < length; i++ ) { hash ^= (uint64_t) at[i]; hash *= 0x100000001B3ull; } + return hash; +} +inline uint64_t TableMapHash( uint64_t key ) { return TableMapHash( (const void *) &key, (int32_t) sizeof( key ) ); } + +// this runtime's own load factor, and no port reproduces it either: the next +// power of two at or above twice the count, so a probe run stays short +inline int32_t TableMapIndexSlots( int32_t count ) +{ + int32_t slots = 8; + while ( slots < count * 2 ) { slots *= 2; } + return slots; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_MAP + +#ifndef LISTDEMO_SCHEMA_TABLE_LIST +#define LISTDEMO_SCHEMA_TABLE_LIST + +namespace listdemo { + +// ---- an UNBOUNDED ARRAY: a counted array whose count the data decides (§2.9) ---- +// +// On the wire, in a region and in a cook a list is the kind 14 body a [..N]T +// writes, its elements by-value records inside the holder's node extent. What +// this adds is the slot, a builder that appends into segments that never +// move, and a const surface that indexes and iterates in place. There is no +// sort, no key and no lookup: the order is INSERTION order, and it is +// identity the way position is identity in a fixed array. + +// elements carved from ONE call to the allocator pair. A new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableListSegmentElements = 32; + +// THE ELEMENT STORAGE: T itself, and a TableRef slot for a []*T, whose +// elements are references exactly as a pointer field's slot is (§2.1) +template struct TableListStorage { typedef T Element; }; +template struct TableListStorage { typedef TableRef Element; }; + +// WHAT THE CONST FORM ANSWERS: the element by reference, and on a []*T the +// RESOLVED pointer, one add on the self-relative delta, NULL for a null slot, +// exactly as At answers it (§6.2, §6.3) +template struct TableListConst +{ + typedef const T & Result; + static Result At( const T * element ) { return *element; } +}; +template struct TableListConst +{ + typedef const T * Result; + static Result At( const TableRef * element ) + { + return element->value != 0 ? (const T *) ( (const uint8_t *) element + element->value ) : NULL; + } +}; + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.9, §7.2) ---- +// +// An int64 self-relative reference to the element array and an int32 count, +// then padding to eight. The reference is a TableRef like a pointer's: in the +// arena it names the builder's HEAD, in a region it is the delta from the slot +// to the first element, and 0 is the empty list in both. It is the map's slot +// exactly, because it is the same two facts. +template struct TableList +{ + typedef typename TableListStorage::Element Element; + + TableRef elements; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Element * Elements() const + { + return elements.value != 0 ? (const Element *) ( (const uint8_t *) &elements + elements.value ) : NULL; + } + int32_t size() const { return count; } + + // INDEXING IS BOUNDS-CHECKED IN EVERY BUILD (§2.4, §2.9): the extent is a + // number that CAME FROM A FILE, so an index past it is not a mistake a + // release build gets to make cheaply. There is no undefined-behavior path + // here in any configuration. The assert carries the message where a + // debugger can read it and NDEBUG removes that. The fatal is what stands + // after it. Both go through the hooks: define schema_assert and + // schema_fatal and this refusal lands in your own handler. + void RefuseIndex( int32_t index ) const + { + if ( (uint32_t) index >= (uint32_t) count ) + { + schema_assert( false && "an unbounded array is indexed inside its count, which came from a file" ); + schema_fatal(); + } + } + typename TableListConst::Result operator[]( int32_t index ) const + { + RefuseIndex( index ); + return TableListConst::At( Elements() + index ); + } + + // ---- iteration: INDEX order, the element and no key ---- + // + // It carries no iterator_traits, for the reason TableKeyed's does not + // (§13.9). + struct ConstIterator + { + const Element * at; + typename TableListConst::Result operator*() const { return TableListConst::At( at ); } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Elements() }; } + ConstIterator end() const { return ConstIterator{ Elements() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.9, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first element is added. Each +// segment is a fixed number of elements carved from one call to the allocator +// pair. An element's address is stable for the arena's life, so a T * handed +// back by Add stays valid while other elements arrive. +struct TableListHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an Add appends into + int32_t live; + int32_t dead; +}; + +template struct TableListSegment +{ + TableRef next; + int32_t used; // elements carved from this segment + int32_t padding; + uint32_t dead[ ( kTableListSegmentElements + 31 ) / 32 ]; // Erase marks one bit, never the element + Element elements[ kTableListSegmentElements ]; +}; + +inline bool TableListSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// the head, allocated when the first element is added +template +inline TableListHead * TableListReach( TableWorker & worker, TableList & list ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( list.elements.value != 0 ) { return (TableListHead *) TableArenaAt( *worker.arena, (uint32_t) list.elements.value ); } + uint32_t at = 0; + TableListHead * head = (TableListHead *) worker.AllocRaw( (int64_t) sizeof( TableListHead ), (int64_t) alignof( TableListHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + list.elements.value = (int64_t) at; + return head; +} + +// one element's storage, appended: the current segment when it has room, a +// new one carved from one call to the pair when it does not. NULL means NOT +// ADDED: an arena that cannot carve another segment, or a count at the int32 +// cap (§2.2, §2.9). +template +inline typename TableList::Element * TableListAppend( TableWorker & worker, TableListHead * head, TableList & list ) +{ + typedef typename TableList::Element Element; + if ( list.count >= INT32_MAX ) { return NULL; } // the int32 storage cap + TableListSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableListSegmentElements ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableListSegment *) worker.AllocRaw( (int64_t) sizeof( TableListSegment ), (int64_t) alignof( TableListSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableListSegment * previous = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Element * element = segment->elements + segment->used; + segment->used++; + head->live++; + list.count++; + return element; +} + +// ADD, whole: the head, the append, and the element at its declared defaults +// (§2.9). The text form's placement is this same call, because a list has no +// key to place under (§16). +template +inline typename TableList::Element * TableListPlace( TableWorker & worker, TableList & list ) +{ + typedef typename TableList::Element Element; + TableListHead * head = TableListReach( worker, list ); + if ( head == NULL ) { return NULL; } + Element * element = TableListAppend( worker, head, list ); + if ( element == NULL ) { return NULL; } + new ( element ) Element(); // value-init: the declared defaults, and null for a slot + return element; +} + +// ERASE, ADDRESSED BY THE POINTER (§2.9): the element Add handed back is the +// handle, because a list has no key and the address is the one thing the +// builder promises never moves (§6.4). It marks the element DEAD, one bit in +// the segment's slot and not in the element storage, and decrements the live +// count. False when the pointer is not this list's. Its storage is reclaimed +// at RESET and never reused mid-build, the map's rule for the map's reason. +template +inline bool TableListErase( TableArena & arena, TableList & list, const typename TableList::Element * element ) +{ + typedef typename TableList::Element Element; + if ( list.elements.value == 0 || element == NULL ) { return false; } + TableListHead * head = (TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableListSegment * segment = (TableListSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + if ( element >= segment->elements && element < segment->elements + segment->used ) + { + const int32_t i = (int32_t) ( element - segment->elements ); + if ( TableListSegmentDead( segment->dead, i ) ) { return false; } // already erased + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + list.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INDEX order, live elements only (§2.9) ---- +template struct TableListEach +{ + typedef typename TableList::Element Element; + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableListSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableListSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + Element * operator*() const { return segment->elements + index; } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableListEach TableListEachOf( const TableArena & arena, const TableList & list ) +{ + TableListEach each = { &arena, TableRef() }; + if ( list.elements.value != 0 ) + { + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + each.first = head->first; + } + return each; +} + +// ---- the INDEX-ORDER CURSOR the four writing walks read (§2.9) ---- +// +// Measure, Save, Lock and Cook each visit a list's live elements in the order +// they were added, and they allocate nothing to do it: a region's cursor is +// the array in place, and the builder's walks the segment chain. Indexing the +// builder's form is SEQUENTIAL by construction, every walk steps i, i + 1, +// i + 2, so the cursor remembers where the last access landed and moves one +// live slot per step. An access behind the memo restarts from the first +// segment, which no walk here does. +template struct TableListCursor +{ + const Element * elements = NULL; // the region's form: the array in place + const TableArena * arena = NULL; // the builder's form: the segments + TableRef first; + int32_t count = 0; + bool ok = false; + // the memo: the segment and slot the last access landed on, and the live + // index that slot holds + mutable const TableListSegment * segment = NULL; + mutable int32_t within = -1; + mutable int32_t logical = -1; + + const Element * At( int32_t index ) const + { + if ( elements != NULL ) { return elements + index; } + if ( segment == NULL || index < logical ) + { + segment = first.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL; + within = -1; + logical = -1; + } + while ( logical < index ) + { + for ( ;; ) + { + within++; + while ( segment != NULL && within >= segment->used ) + { + segment = segment->next.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + within = 0; + } + if ( segment == NULL ) { return NULL; } // the slot and the head disagree + if ( !TableListSegmentDead( segment->dead, within ) ) { break; } + } + logical++; + } + return segment->elements + within; + } + const Element & operator[]( int32_t index ) const { return *At( index ); } +}; + +// the REGION form: the array is the cursor +template +inline TableListCursor::Element> TableListElements( const TableRegionCtx &, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.elements = list.Elements(); + cursor.count = list.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: the live elements out of the segment chain, in the +// order they were added. A dead element costs nothing on any wire (§2.9). +template +inline TableListCursor::Element> TableListElements( const TableArena & arena, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.arena = &arena; + cursor.count = list.count; + if ( list.elements.value == 0 || list.count <= 0 ) { cursor.ok = list.count == 0; cursor.count = 0; return cursor; } + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + if ( head->live != list.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + cursor.first = head->first; + cursor.ok = true; + return cursor; +} + +template +inline TableListCursor::Element> TableListElements( const TableArenaCtx & ctx, const TableList & list ) +{ + return TableListElements( *ctx.arena, list ); +} + +// ---- the LOAD side: where a decoded element lands (§2.9) ---- +// +// The same two shapes the map's fill takes, because the decoder above them +// cannot tell which it has: a REGION carves the element array out of the +// holder node's own extent, PRE-ORDER, and the TOOL's path appends into the +// builder's arena. +template struct TableListFill +{ + typedef typename TableList::Element Element; + TableList * list = NULL; + Element * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; + bool refused = false; // a count above the int32 cap on the tool's path: LoadBuilder answers NULL +}; + +template +inline TableListFill TableListFillBegin( const TableNodeMap & nodes, TableList & list, uint64_t n ) +{ + typedef typename TableList::Element Element; + TableListFill fill; + fill.list = &list; + list.elements.value = 0; + list.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( n > (uint64_t) INT32_MAX ) + { + // A COUNT ABOVE THE int32 STORAGE CAP (§2.2, §2.9): into a region it was + // refused by LoadMeasure before this ran, and into a builder it is the + // refusal LoadBuilder answers NULL for, moving no counter + fill.refused = nodes.carve->worker != NULL; + return fill; + } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Element ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Element ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Element *) base; + fill.capacity = (int32_t) n; + list.elements.value = (int64_t) ( base - (const uint8_t *) &list.elements ); + fill.ok = true; + return fill; +} + +// the next slot, at the element's declared defaults. NULL when the arena +// could not carve, which the decoder reports as framing damage +template inline typename TableList::Element * TableListFillNext( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count >= fill.capacity ) { return NULL; } + Element * element = fill.array + fill.list->count; + new ( element ) Element(); + fill.list->count++; + return element; + } + return TableListPlace( *fill.worker, *fill.list ); +} + +// A SLOT WHOSE ELEMENT NEVER LANDED is given back (§2.9, §4): the array keeps +// what it decoded, and an element whose own framing gave out before one byte +// of it decoded was not decoded. The region's form uncounts it, and the builder's +// marks it dead, which is what the storage rule allows mid-build. +template inline void TableListFillDrop( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count > 0 ) { fill.list->count--; } + return; + } + if ( fill.list->elements.value == 0 ) { return; } + TableListHead * head = (TableListHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.list->elements.value ); + if ( head->last.value == 0 ) { return; } + TableListSegment * segment = (TableListSegment *) TableArenaAt( *fill.worker->arena, (uint32_t) head->last.value ); + if ( segment->used <= 0 ) { return; } + const int32_t i = segment->used - 1; + if ( TableListSegmentDead( segment->dead, i ) ) { return; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + fill.list->count--; +} + +// an EMPTY list's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableListFillEnd( TableListFill & fill ) +{ + if ( fill.array != NULL && fill.list->count == 0 ) { fill.list->elements.value = 0; } +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.9, §6.5) ---- +// +// N x sizeof( T ) rounded to alignof( T ), AT EVERY DEPTH. N is framing and +// not a value, so this reads no field: it walks the list's own header and, +// where a table element holds a list or a map of its own, the elements' +// headers under it. Every -1 carries its REASON (§6.5): the int32 cap first, +// because a count past it cannot fit any body, and then the body's own L. +inline bool TableListWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t elem_size, int64_t elem_align, uint8_t elem_kind, int64_t elem_floor, + TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != elem_kind ) { return true; } // another element kind: §4's ordinary kind mismatch, the field reads empty + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; } // an N the list's L cannot carry + at = ( at + elem_align - 1 ) & ~( elem_align - 1 ); + at += (int64_t) n * elem_size; + if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_LIST + +#ifndef LISTDEMO_SCHEMA_BUILD_VERSION +#define LISTDEMO_SCHEMA_BUILD_VERSION + +namespace listdemo { + +// THE BUILD VERSION (docs/SPEC-TABLES.md §20): one digest over every fact the bytes +// this build produces depend on — the type wire's protocol id, every record's +// layout as the compiler's own C ABI model computes it, and the facts that +// decide what a load PUTS in those slots. It is the number a cook's header +// carries and the number Open compares, and the number a block's prologue +// carries and BlockOpen compares: a build version answers "which build?" and +// not "which form?", and what separates the two forms is their MAGIC. +// +// There are TWO ids in the design and they are not interchangeable: the +// PROTOCOL ID is the type wire's and nothing else, and the BUILD VERSION is +// what everything cooked or blocked is keyed by. A table edit moves this and +// never the protocol id; a type edit moves both. +static const uint64_t BuildVersion = 0x8d7c0edaca4571c7ull; + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_BUILD_VERSION + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK +#define LISTDEMO_SCHEMA_TABLE_COOK + +namespace listdemo { + +// ---- the cooked form (docs/SPEC-TABLES.md §7) ---- +// +// A cooked file is a HEADER, a DATA part and an ATTRIBUTION part, in that +// order. Every word of the header is a u64 written in the byte order the cook +// was produced in, and the header is 64 bytes: +// +// 0 magic 0x4b4f4f434d484353, read BYTEWISE before anything else +// 8 build_version the unit's id (docs/SPEC-TABLES.md §20) +// 16 byte_order 1 little, 2 big — the order that WROTE the file +// 24 data_length the region's bytes, rounded up to alignment +// 32 attribution_length the directory's bytes, or 0 +// 40 alignment the region's alignment, never below eight +// 48 reserved zero +// 56 reserved zero +// +// The DATA part is Lock's region written verbatim (§7.2) — the root at its +// base — and it is what a runtime points at. The ATTRIBUTION part is the node +// directory (§6.3), and NOTHING THAT READS THE STRUCTURE TOUCHES IT: it is +// written beside the data for schema cook-check, so a build that ships no +// tooling need not carry it at all. +static const int64_t kTableCookHeaderBytes = 64; + +// THE MAGIC'S VALUE, and a consumer written from the page needs the constant +// rather than a description of one. It is "SCHMCOOK" read as ASCII in the byte +// order a little-endian store produces — the same shape the block form's +// SCHMABLK takes, so a hex dump of a little-endian cook is legible and the two +// accelerators sit in one vocabulary. +// +// IT IS STORED IN THE PRODUCER'S ORDER, which is what makes it the byte-order +// check as well as the form check: a consumer reads back this build's +// constant, or that constant byte-reversed — which identifies a cook of the +// OTHER order — or something that is not a cook. All three answers but the +// first refuse, and a cook and a BLOCK are separated here too, because a +// form's identity belongs in its magic rather than in a second digest. +static const uint64_t TableCookMagic = 0x4b4f4f434d484353ull; + +// THIS BUILD's byte order, as the header's own word carries it. The magic is +// what REFUSES a foreign order; this word is what RECORDS which order wrote +// the file, so a refusal names the order rather than inferring it and a tool +// dumping a cook reads the fact. A file whose magic matched and whose order +// word did not is corrupt, and there is no reading that recovers it. +// +// The BUILD VERSION cannot do either job: §20.1 digests byteorder as a +// GENERATION input, little for every target schema generates for today, so +// two builds of one schema for two orders emit the same id. +#if defined( __BYTE_ORDER__ ) && defined( __ORDER_BIG_ENDIAN__ ) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__ +static const uint64_t TableCookByteOrder = 2; // big +#else +static const uint64_t TableCookByteOrder = 1; // little +#endif + +// The greatest region alignment a cooked file may name. The DATA part begins +// at align_up( 64, alignment ), which is 64 for every unit this language can +// declare — the largest alignment it has is sixteen — so a word past this cap +// describes a file no build of this schema wrote (docs/SPEC-TABLES.md §7.1). +static const uint64_t TableCookMaxAlign = 64; + +// The header read, BYTEWISE. memcpy is the portable spelling of "these eight +// bytes, in this machine's order"; every compiler this repo builds under folds +// it to one load, and it is the only read in the whole of Open that is not a +// comparison. +inline uint64_t table_cook_read64( const uint8_t * p ) +{ + uint64_t v; + memcpy( &v, p, sizeof( v ) ); + return v; +} + +// TableCookOpen: THE WHOLE CHECK, in one place, because §7 states the +// enumeration once and every generated Open is that one enumeration plus +// its own root's two layout facts. +// +// THE CHECK, in order: the magic read bytewise, the byte order it establishes, +// the build version against this build's own, both RESERVED words zero, the +// region alignment the header names, the two part lengths against the length +// the caller passed — a truncated file and a file with trailing bytes are the +// same refusal — the root's own storage inside the data part, and the +// alignment of the base. +// +// AND THAT IS ALL OF IT. On a match the bytes ARE what this build wrote, in +// this build's layout and this build's byte order, so there is nothing to +// validate and nothing to fix up: the caller gets the root. Nothing per node +// happens here, which is what makes open O(1) in the file's size; a walk of +// any shape would forfeit that, and validating an untrusted file is schema +// cook-check's job and a person's decision (§7.4). +// +// EVERY NUMBER BELOW COMES OUT OF THE FILE, so the arithmetic is unsigned and +// each term is BOUNDED BEFORE IT IS ADDED: a forged length near 2^64 must +// refuse, and an addition that wrapped would be the defect the comparison +// after it was supposed to catch. Nothing past length is read on any path, +// including every refusing one. +inline const uint8_t * TableCookOpen( const void * bytes, uint64_t length, uint64_t root_size, uint64_t root_align ) +{ + if ( bytes == NULL ) { return NULL; } + if ( length < (uint64_t) kTableCookHeaderBytes ) { return NULL; } + const uint8_t * raw = (const uint8_t *) bytes; + // the MAGIC, bytewise and first: it is what establishes the byte order + // every other header word is read in, so nothing else may be read before + // it. A byte-reversed constant is a cook of the other order and refuses + // here, which is why the order never reaches a fix-up pass. + if ( table_cook_read64( raw ) != TableCookMagic ) { return NULL; } + if ( table_cook_read64( raw + 16 ) != TableCookByteOrder ) { return NULL; } + if ( table_cook_read64( raw + 8 ) != BuildVersion ) { return NULL; } + // the RESERVED words: a non-zero one means a writer used a form this build + // does not understand, and Open refuses rather than ignoring it. + if ( table_cook_read64( raw + 48 ) != 0 ) { return NULL; } + if ( table_cook_read64( raw + 56 ) != 0 ) { return NULL; } + const uint64_t data_length = table_cook_read64( raw + 24 ); + const uint64_t attribution_length = table_cook_read64( raw + 32 ); + const uint64_t alignment = table_cook_read64( raw + 40 ); + // THE ALIGNMENT WORD IS DATA, and it is the one header field the rest of + // the check does arithmetic WITH rather than only comparison against. A + // region's alignment is a power of two, never below eight (the floor that + // puts the attribution part on an eight-byte boundary without a second + // padding rule) and never past the cap above; a word that is none of those + // rounds nothing and aligns nothing, so it is refused before it is used. + if ( alignment < 8 || alignment > TableCookMaxAlign ) { return NULL; } + if ( ( alignment & ( alignment - 1 ) ) != 0 ) { return NULL; } + // and it must be an alignment THE ROOT CAN SIT AT, since the root is at + // the region's base: both are powers of two, so "at least the root's" + // is one division. + if ( ( alignment % root_align ) != 0 ) { return NULL; } + // The DATA part begins at align_up( 64, alignment ). It is DERIVED and not + // a header field, because a fact a reader computes is a fact two writers + // cannot disagree about. + const uint64_t data_offset = ( (uint64_t) kTableCookHeaderBytes + alignment - 1 ) & ~( alignment - 1 ); + if ( length < data_offset ) { return NULL; } + // the two part lengths against the length the caller passed. The whole + // file is data_offset + data_length + attribution_length, and a length + // that is not EXACTLY that refuses — truncation and trailing bytes are one + // refusal, and both terms are subtracted rather than added so no sum can + // carry. + if ( data_length > length - data_offset ) { return NULL; } + if ( attribution_length != length - data_offset - data_length ) { return NULL; } + // the ROOT sits at the region's base, so the region has to hold it: a + // shorter data part describes a root partly outside the file, which is the + // one way a match-and-point reader could hand back storage it never + // received. + if ( data_length < root_size ) { return NULL; } + const uint8_t * base = raw + data_offset; + // the alignment of the BASE. The header pads the data part to the region's + // alignment, so a base an allocator or mmap gave you is already aligned — + // mmap gives page alignment for free — and a base that is not is a caller's + // buffer this form cannot be read out of. + if ( ( (uintptr_t) base % (uintptr_t) alignment ) != 0 ) { return NULL; } + return base; +} + +// ---- the cooked form, the WRITE side (docs/SPEC-TABLES.md §7.6) ---- +// +// THE BYTE ORDER IS THE TARGET'S, NOT THE HOST'S. A cook is produced in the +// byte order of the build that will read it (§7), so the fixing happens here — +// offline, once, on the writing side — and never at Open. Passing +// TableByteOrder::Big on a little-endian machine produces a big-endian build's +// file, and nothing about the writing host reaches the bytes. +enum class TableByteOrder +{ + Little = 1, // the header's byte_order word, and the order every scalar is written in + Big = 2, +}; + +// One store, width as an argument. Every call site passes a literal width, so +// the loop folds to a store (and a byte swap on the foreign order); a name per +// width would claim four §11 names to save nothing. +inline void table_cook_put( uint8_t * at, uint64_t value, int32_t width, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * i ) ); } + } + else + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * ( width - 1 - i ) ) ); } + } +} + +// A 128-bit store as two lanes: sixteen bytes, the low lane first in the +// little order and the high lane first — each lane big-endian — in the big +// order, exactly as a u64 is one lane of eight (docs/SPEC-TABLES.md §7.2). +inline void table_cook_put128( uint8_t * at, uint64_t lo, uint64_t hi, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) { table_cook_put( at, lo, 8, order ); table_cook_put( at + 8, hi, 8, order ); } + else { table_cook_put( at, hi, 8, order ); table_cook_put( at + 8, lo, 8, order ); } +} + +// A buffer piece: the USED bytes and nothing else. The tail is already zero — +// the whole extent was zeroed before any field was written — so this copies the +// used prefix and leaves the rest, which is what makes a string's unused tail a +// consequence of one memset rather than a rule per buffer. A used length past +// the buffer, or below zero, is a value no reader could have produced and it is +// clamped rather than trusted: this writes inside the caller's buffer on every +// input. +inline void table_cook_bytes( uint8_t * at, const void * source, int64_t used, int64_t capacity ) +{ + if ( used <= 0 ) { return; } + const int64_t n = used < capacity ? used : capacity; + memcpy( at, source, (size_t) n ); +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE +#define LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// ---- the cooked form's WRITE side for a POINTERED root (docs/SPEC-TABLES.md §7.6) ---- +// +// A pointered root's cook is the region of §7.2: every node the numbering +// reached (§3.1), once, at its own type's alignment, in index order, the root +// at offset zero. This is that region while it is being laid out and written — +// the tool's own Layout and Write, in one struct. +// +// The OFFSETS are one per node, the root's zero at position 0 and node index k +// at position k - 1, which is the directory's own order (§6.3); they are the +// one allocation the write makes beyond the numbering, and they go through the +// same pair. A measure needs no offsets and leaves the pointer NULL. +struct TableCookRegion +{ + const TableNumbering * numbering = NULL; // node -> index, from the walk that placed it + int64_t * offsets = NULL; // index - 1 -> the node's region offset; NULL while measuring + int64_t count = 0; // nodes, the root included + int64_t bytes = 0; // the data part's length, rounded to align + int64_t align = 0; // the region's alignment: the nodes' greatest, never below eight + uint8_t * base = NULL; // where the data part is being written; NULL while measuring +}; + +// A reference slot: the SELF-RELATIVE delta from the slot's own address to the +// node's start (§6.3), and zero for null. The node is found by the address the +// numbering keyed it under, which is the same address the walk resolved through +// the same context — so a reference the numbering does not carry is a slot the +// walk never reached (a counted array's slot past its count, an absent +// optional's value) holding a node the region will not hold, and it is refused +// rather than written as a delta to nowhere. +inline bool table_cook_ref( const TableCookRegion & region, uint8_t * at, const void * pointee, TableByteOrder order ) +{ + if ( pointee == NULL ) { table_cook_put( at, 0, 8, order ); return true; } + uint64_t index = 0; + if ( !TableNumberingIndex( *region.numbering, pointee, index ) ) { return false; } + if ( index == 0 || index > (uint64_t) region.count ) { return false; } + const int64_t delta = region.offsets[index - 1] - (int64_t) ( at - region.base ); + table_cook_put( at, (uint64_t) delta, 8, order ); + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// table Bytes — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Bytes { + TableList data; // uint8: the element array, empty until an Add + int32_t after = 0; +}; + +// table Ints — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Ints { + TableList values; // int32: the element array, empty until an Add + int32_t after = 0; +}; + +// table Floats — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Floats { + TableList values; // float32: the element array, empty until an Add + int32_t after = 0; +}; + +// ---- prefill: the declared defaults, in place (docs/SPEC-TABLES.md) ---- + +inline void BytesReset( Bytes & value ); +inline void IntsReset( Ints & value ); +inline void FloatsReset( Floats & value ); + +inline void BytesReset( Bytes & value ) +{ + value.data.elements.value = 0; // uint8: empty + value.data.count = 0; + value.data.padding = 0; + value.after = 0; +} + +inline void IntsReset( Ints & value ) +{ + value.values.elements.value = 0; // int32: empty + value.values.count = 0; + value.values.padding = 0; + value.after = 0; +} + +inline void FloatsReset( Floats & value ) +{ + value.values.elements.value = 0; // float32: empty + value.values.count = 0; + value.values.padding = 0; + value.after = 0; +} + +// ---- the arena's reset hook (docs/SPEC-TABLES.md §6) ---- +// +// TableWorker::Alloc is a template and cannot name a member's Reset, so +// the arena reaches it through this overload set by argument-dependent +// lookup. It is how a node born in raw arena storage comes to hold the +// declared defaults without value-initialising the whole aggregate. + +inline void TableReset( Bytes & value ) { BytesReset( value ); } +inline void TableReset( Ints & value ) { IntsReset( value ); } +inline void TableReset( Floats & value ) { FloatsReset( value ); } + +// ---- pointer targets: allocation and resolution (docs/SPEC-TABLES.md §2) ---- +// +// A reference resolves differently in the two forms, and the CONTEXT says +// which: in the arena it is an offset; in a region it is a self-relative +// delta, so the const deref below is one add and needs no base pointer. + +// ---- codecs: measure/save/load per closure member ---- + +template inline int64_t BytesMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Bytes & value ); +template inline bool BytesSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Bytes & value ); +template inline bool BytesSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Bytes & value ); +inline bool BytesLoadBody( TableReader & r, const TableNodeMap & nodes, Bytes & value ); +template inline int64_t IntsMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Ints & value ); +template inline bool IntsSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Ints & value ); +template inline bool IntsSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Ints & value ); +inline bool IntsLoadBody( TableReader & r, const TableNodeMap & nodes, Ints & value ); +template inline int64_t FloatsMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Floats & value ); +template inline bool FloatsSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Floats & value ); +template inline bool FloatsSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Floats & value ); +inline bool FloatsLoadBody( TableReader & r, const TableNodeMap & nodes, Floats & value ); + +// ---- pointer-graph walkers: number (measure/save), pack (Lock) ---- + +template inline bool BytesNumber( const Ctx & ctx, TableNumbering & numbering, const Bytes & value ); +template inline int64_t BytesPackMeasure( const Ctx & ctx, TablePackMap & seen, const Bytes & value ); +template inline bool BytesPack( const Ctx & ctx, TablePackMap & seen, const Bytes & src, Bytes & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool IntsNumber( const Ctx & ctx, TableNumbering & numbering, const Ints & value ); +template inline int64_t IntsPackMeasure( const Ctx & ctx, TablePackMap & seen, const Ints & value ); +template inline bool IntsPack( const Ctx & ctx, TablePackMap & seen, const Ints & src, Ints & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool FloatsNumber( const Ctx & ctx, TableNumbering & numbering, const Floats & value ); +template inline int64_t FloatsPackMeasure( const Ctx & ctx, TablePackMap & seen, const Floats & value ); +template inline bool FloatsPack( const Ctx & ctx, TablePackMap & seen, const Floats & src, Floats & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +// ---- the numbering's bridge to each member's codec (docs/SPEC-TABLES.md §3.1) ---- + +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Bytes & value ) { return BytesMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Bytes & value ) { return BytesSaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Ints & value ) { return IntsMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Ints & value ) { return IntsSaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Floats & value ) { return FloatsMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Floats & value ) { return FloatsSaveBody( ctx, numbering, w, ids, value ); } + +template +inline int64_t BytesMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Bytes & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // data: a kind 14 array of kind 6 elements, INDEX order (§2.9) + TableListCursor cursor_data = TableListElements( ctx, value.data ); + if ( !cursor_data.ok ) { return -1; } // the slot and the head disagree + if ( cursor_data.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_data = ids.ref( 0x855b556730a34a05ull, 8 ); + int64_t body_data = 0; + body_data += 1 + TableLebBytes( (uint64_t) ( cursor_data.count ) ); // the element kind byte and the count + body_data += (int64_t) ( cursor_data.count ) * 1; + bytes += TableLebBytes( ref_data ) + 1 + TableLebBytes( (uint64_t) ( body_data ) ) + ( body_data ); + } + } + if ( value.after != 0 ) { bytes += TableLebBytes( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ) + 1 + 4; } // after + return bytes; +} + +template +inline bool BytesSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Bytes & value ) +{ + (void) ctx; (void) numbering; + { + TableListCursor cursor_data = TableListElements( ctx, value.data ); // data + if ( !cursor_data.ok ) { return false; } + if ( cursor_data.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_data = ids.ref( 0x855b556730a34a05ull, 8 ); + int64_t body_data = 0; + body_data += 1 + TableLebBytes( (uint64_t) ( cursor_data.count ) ); // the element kind byte and the count + body_data += (int64_t) ( cursor_data.count ) * 1; + w.putleb( ref_data ); w.put8( 14 ); w.putleb( (uint64_t) body_data ); // data + w.put8( 6 ); w.putleb( (uint64_t) ( cursor_data.count ) ); + for ( int32_t elem_i_data = 0; elem_i_data < cursor_data.count; elem_i_data++ ) + { + w.put8( uint8_t( cursor_data[elem_i_data] ) ); + } + } + } + if ( value.after != 0 ) + { + w.putleb( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ); w.put8( 4 ); // after + w.put32( uint32_t( value.after ) ); + } + return !w.overflow; +} + +template +inline bool BytesSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Bytes & value ) +{ + if ( !BytesSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool BytesLoadBody( TableReader & r, const TableNodeMap & nodes, Bytes & value ) +{ + (void) nodes; + BytesReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x855b556730a34a05ull: // data + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 6 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.data, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + uint8_t * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + if ( !sub.has( 1 ) ) { r.report->malformed = true; break; } + uint8_t decoded_v = uint8_t( sub.get8( ) ); + ( *slot ) = decoded_v; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xbf82010f6f71eae9ull: // after + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.after = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +template +inline int64_t IntsMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Ints & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // values: a kind 14 array of kind 4 elements, INDEX order (§2.9) + TableListCursor cursor_values = TableListElements( ctx, value.values ); + if ( !cursor_values.ok ) { return -1; } // the slot and the head disagree + if ( cursor_values.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_values = ids.ref( 0x21277bcf1a4d67fbull, 10 ); + int64_t body_values = 0; + body_values += 1 + TableLebBytes( (uint64_t) ( cursor_values.count ) ); // the element kind byte and the count + body_values += (int64_t) ( cursor_values.count ) * 4; + bytes += TableLebBytes( ref_values ) + 1 + TableLebBytes( (uint64_t) ( body_values ) ) + ( body_values ); + } + } + if ( value.after != 0 ) { bytes += TableLebBytes( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ) + 1 + 4; } // after + return bytes; +} + +template +inline bool IntsSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Ints & value ) +{ + (void) ctx; (void) numbering; + { + TableListCursor cursor_values = TableListElements( ctx, value.values ); // values + if ( !cursor_values.ok ) { return false; } + if ( cursor_values.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_values = ids.ref( 0x21277bcf1a4d67fbull, 10 ); + int64_t body_values = 0; + body_values += 1 + TableLebBytes( (uint64_t) ( cursor_values.count ) ); // the element kind byte and the count + body_values += (int64_t) ( cursor_values.count ) * 4; + w.putleb( ref_values ); w.put8( 14 ); w.putleb( (uint64_t) body_values ); // values + w.put8( 4 ); w.putleb( (uint64_t) ( cursor_values.count ) ); + for ( int32_t elem_i_values = 0; elem_i_values < cursor_values.count; elem_i_values++ ) + { + w.put32( uint32_t( cursor_values[elem_i_values] ) ); + } + } + } + if ( value.after != 0 ) + { + w.putleb( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ); w.put8( 4 ); // after + w.put32( uint32_t( value.after ) ); + } + return !w.overflow; +} + +template +inline bool IntsSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Ints & value ) +{ + if ( !IntsSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool IntsLoadBody( TableReader & r, const TableNodeMap & nodes, Ints & value ) +{ + (void) nodes; + IntsReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x21277bcf1a4d67fbull: // values + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 4 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.values, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + int32_t * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + if ( !sub.has( 4 ) ) { r.report->malformed = true; break; } + int32_t decoded_v = int32_t( sub.get32( ) ); + ( *slot ) = decoded_v; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xbf82010f6f71eae9ull: // after + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.after = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +template +inline int64_t FloatsMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Floats & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // values: a kind 14 array of kind 10 elements, INDEX order (§2.9) + TableListCursor cursor_values = TableListElements( ctx, value.values ); + if ( !cursor_values.ok ) { return -1; } // the slot and the head disagree + if ( cursor_values.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_values = ids.ref( 0x21277bcf1a4d67fbull, 10 ); + int64_t body_values = 0; + body_values += 1 + TableLebBytes( (uint64_t) ( cursor_values.count ) ); // the element kind byte and the count + body_values += (int64_t) ( cursor_values.count ) * 4; + bytes += TableLebBytes( ref_values ) + 1 + TableLebBytes( (uint64_t) ( body_values ) ) + ( body_values ); + } + } + if ( value.after != 0 ) { bytes += TableLebBytes( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ) + 1 + 4; } // after + return bytes; +} + +template +inline bool FloatsSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Floats & value ) +{ + (void) ctx; (void) numbering; + { + TableListCursor cursor_values = TableListElements( ctx, value.values ); // values + if ( !cursor_values.ok ) { return false; } + if ( cursor_values.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_values = ids.ref( 0x21277bcf1a4d67fbull, 10 ); + int64_t body_values = 0; + body_values += 1 + TableLebBytes( (uint64_t) ( cursor_values.count ) ); // the element kind byte and the count + body_values += (int64_t) ( cursor_values.count ) * 4; + w.putleb( ref_values ); w.put8( 14 ); w.putleb( (uint64_t) body_values ); // values + w.put8( 10 ); w.putleb( (uint64_t) ( cursor_values.count ) ); + for ( int32_t elem_i_values = 0; elem_i_values < cursor_values.count; elem_i_values++ ) + { + w.put32( table_float_to_bits( cursor_values[elem_i_values] ) ); + } + } + } + if ( value.after != 0 ) + { + w.putleb( ids.ref( 0xbf82010f6f71eae9ull, 5 ) ); w.put8( 4 ); // after + w.put32( uint32_t( value.after ) ); + } + return !w.overflow; +} + +template +inline bool FloatsSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Floats & value ) +{ + if ( !FloatsSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool FloatsLoadBody( TableReader & r, const TableNodeMap & nodes, Floats & value ) +{ + (void) nodes; + FloatsReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x21277bcf1a4d67fbull: // values + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 10 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.values, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + float * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + if ( !sub.has( 4 ) ) { r.report->malformed = true; break; } + ( *slot ) = table_bits_to_float( sub.get32() ); + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xbf82010f6f71eae9ull: // after + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.after = decoded_v; + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// BytesWireExtent: the extent Bytes's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool BytesWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x855b556730a34a05ull && field_kind == 14 ) // data: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( uint8_t ), (int64_t) alignof( uint8_t ), 6, 1, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// BytesExtentAt: the node extent Bytes's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as BytesExtentPack advances it (§2.8, §2.9). +template +inline bool BytesExtentAt( const Ctx & ctx, const Bytes & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.data ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( uint8_t ) - 1 ) & ~( (int64_t) alignof( uint8_t ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( uint8_t ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t BytesExtent( const Ctx & ctx, const Bytes & value ) +{ + int64_t at = 0; + if ( !BytesExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// BytesExtentPack: carve Bytes's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset BytesExtentAt advances (§2.8, §2.9). +template +inline bool BytesExtentPack( const Ctx & ctx, const Bytes & src, Bytes & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.data ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( uint8_t ) - 1 ) & ~( (int64_t) alignof( uint8_t ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( uint8_t ); + if ( at + bytes > capacity ) { return false; } + uint8_t * placed = (uint8_t *) ( extent + at ); + at += bytes; + dst.data.count = cursor.count; + dst.data.padding = 0; + dst.data.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.data.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( uint8_t ) ); // trivially copyable, by construction + } + } + return true; +} + +// IntsWireExtent: the extent Ints's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool IntsWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x21277bcf1a4d67fbull && field_kind == 14 ) // values: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( int32_t ), (int64_t) alignof( int32_t ), 4, 4, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// IntsExtentAt: the node extent Ints's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as IntsExtentPack advances it (§2.8, §2.9). +template +inline bool IntsExtentAt( const Ctx & ctx, const Ints & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.values ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( int32_t ) - 1 ) & ~( (int64_t) alignof( int32_t ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( int32_t ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t IntsExtent( const Ctx & ctx, const Ints & value ) +{ + int64_t at = 0; + if ( !IntsExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// IntsExtentPack: carve Ints's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset IntsExtentAt advances (§2.8, §2.9). +template +inline bool IntsExtentPack( const Ctx & ctx, const Ints & src, Ints & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.values ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( int32_t ) - 1 ) & ~( (int64_t) alignof( int32_t ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( int32_t ); + if ( at + bytes > capacity ) { return false; } + int32_t * placed = (int32_t *) ( extent + at ); + at += bytes; + dst.values.count = cursor.count; + dst.values.padding = 0; + dst.values.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.values.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( int32_t ) ); // trivially copyable, by construction + } + } + return true; +} + +// FloatsWireExtent: the extent Floats's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool FloatsWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x21277bcf1a4d67fbull && field_kind == 14 ) // values: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( float ), (int64_t) alignof( float ), 10, 4, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// FloatsExtentAt: the node extent Floats's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as FloatsExtentPack advances it (§2.8, §2.9). +template +inline bool FloatsExtentAt( const Ctx & ctx, const Floats & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.values ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( float ) - 1 ) & ~( (int64_t) alignof( float ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( float ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t FloatsExtent( const Ctx & ctx, const Floats & value ) +{ + int64_t at = 0; + if ( !FloatsExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// FloatsExtentPack: carve Floats's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset FloatsExtentAt advances (§2.8, §2.9). +template +inline bool FloatsExtentPack( const Ctx & ctx, const Floats & src, Floats & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.values ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( float ) - 1 ) & ~( (int64_t) alignof( float ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( float ); + if ( at + bytes > capacity ) { return false; } + float * placed = (float *) ( extent + at ); + at += bytes; + dst.values.count = cursor.count; + dst.values.padding = 0; + dst.values.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.values.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( float ) ); // trivially copyable, by construction + } + } + return true; +} + +// ---- Bytes.data: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline uint8_t * BytesDataAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool BytesDataErase( TableArena & arena, TableList & list, const uint8_t * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach BytesDataEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Ints.values: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline int32_t * IntsValuesAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool IntsValuesErase( TableArena & arena, TableList & list, const int32_t * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach IntsValuesEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Floats.values: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline float * FloatsValuesAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool FloatsValuesErase( TableArena & arena, TableList & list, const float * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach FloatsValuesEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// BytesNumber: number everything Bytes POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool BytesNumber( const Ctx & ctx, TableNumbering & numbering, const Bytes & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// BytesPackMeasure: the packed region bytes of everything Bytes POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t BytesPackMeasure( const Ctx & ctx, TablePackMap & seen, const Bytes & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// BytesPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool BytesPackEdges( const Ctx & ctx, TablePackMap & seen, const Bytes & src, Bytes & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool BytesPack( const Ctx & ctx, TablePackMap & seen, const Bytes & src, Bytes & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Bytes ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Bytes ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !BytesExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return BytesPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool BytesPackEdges( const Ctx & ctx, TablePackMap & seen, const Bytes & src, Bytes & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// IntsNumber: number everything Ints POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool IntsNumber( const Ctx & ctx, TableNumbering & numbering, const Ints & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// IntsPackMeasure: the packed region bytes of everything Ints POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t IntsPackMeasure( const Ctx & ctx, TablePackMap & seen, const Ints & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// IntsPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool IntsPackEdges( const Ctx & ctx, TablePackMap & seen, const Ints & src, Ints & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool IntsPack( const Ctx & ctx, TablePackMap & seen, const Ints & src, Ints & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Ints ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Ints ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !IntsExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return IntsPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool IntsPackEdges( const Ctx & ctx, TablePackMap & seen, const Ints & src, Ints & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// FloatsNumber: number everything Floats POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool FloatsNumber( const Ctx & ctx, TableNumbering & numbering, const Floats & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// FloatsPackMeasure: the packed region bytes of everything Floats POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t FloatsPackMeasure( const Ctx & ctx, TablePackMap & seen, const Floats & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// FloatsPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool FloatsPackEdges( const Ctx & ctx, TablePackMap & seen, const Floats & src, Floats & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool FloatsPack( const Ctx & ctx, TablePackMap & seen, const Floats & src, Floats & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Floats ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Floats ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !FloatsExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return FloatsPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool FloatsPackEdges( const Ctx & ctx, TablePackMap & seen, const Floats & src, Floats & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// ---- Bytes: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: BytesBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Bytes is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct BytesBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + BytesBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~BytesBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + BytesBuilder( const BytesBuilder & ) = delete; + BytesBuilder & operator=( const BytesBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Bytes * GetRoot() { return arena.locked ? NULL : (Bytes *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Bytes * AsConst() const { return (const Bytes *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool BytesBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Bytes & root = *(const Bytes *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = BytesPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = BytesExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + Bytes * destination = new ( packed ) Bytes; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !BytesPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Bytes on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// BytesNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t BytesNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// BytesNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void BytesNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// BytesNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t BytesNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// BytesNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t BytesNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// BytesNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void BytesNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = BytesNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? BytesNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool BytesNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Bytes & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return BytesNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t BytesMeasureWire( const Ctx & ctx, const Bytes & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( BytesNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = BytesMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t BytesSaveWire( const Ctx & ctx, const Bytes & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !BytesNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = BytesSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == BytesMeasure( root ) +} + +inline int64_t BytesMeasure( const Bytes * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return BytesMeasureWire( ctx, *root, allocator ); +} + +inline int64_t BytesSave( const Bytes * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return BytesSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t BytesMeasure( const BytesBuilder & builder ) +{ + if ( builder.region != NULL ) { return BytesMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return BytesMeasureWire( ctx, *(const Bytes *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t BytesSave( const BytesBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return BytesSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return BytesSaveWire( ctx, *(const Bytes *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t BytesMeasureMessage( const Bytes * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return BytesMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t BytesSaveMessage( const Bytes * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return BytesSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t BytesMeasureMessage( const BytesBuilder & builder ) +{ + if ( builder.region != NULL ) { return BytesMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return BytesMeasureWire( ctx, *(const Bytes *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t BytesSaveMessage( const BytesBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return BytesSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return BytesSaveWire( ctx, *(const Bytes *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// BytesLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t BytesLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !BytesWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = BytesNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// BytesLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Bytes * BytesLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Bytes ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !BytesWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = BytesNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xeeeea7adc131a244ull; + Bytes * root = new ( region ) Bytes; // lifetime only: LoadBody's first act is BytesReset + BytesReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = BytesNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + BytesNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + BytesNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Bytes ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + BytesLoadBody( r, nodes, *root ); + return root; +} + +// BytesLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t BytesLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !BytesWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = BytesNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// BytesLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Bytes * BytesLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Bytes ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !BytesWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = BytesNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xeeeea7adc131a244ull; + Bytes * root = new ( region ) Bytes; // lifetime only: LoadBody's first act is BytesReset + BytesReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Bytes ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = BytesNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + BytesNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + BytesNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Bytes ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + BytesLoadBody( r, nodes, *root ); + return root; +} + +// BytesLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool BytesLoadBuilder( BytesBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Bytes * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0xeeeea7adc131a244ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = BytesNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + BytesNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = BytesLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Ints: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: IntsBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Ints is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct IntsBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + IntsBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~IntsBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + IntsBuilder( const IntsBuilder & ) = delete; + IntsBuilder & operator=( const IntsBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Ints * GetRoot() { return arena.locked ? NULL : (Ints *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Ints * AsConst() const { return (const Ints *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool IntsBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Ints & root = *(const Ints *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = IntsPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = IntsExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + Ints * destination = new ( packed ) Ints; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !IntsPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Ints on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// IntsNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t IntsNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// IntsNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void IntsNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// IntsNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t IntsNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// IntsNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t IntsNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// IntsNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void IntsNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = IntsNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? IntsNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool IntsNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Ints & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return IntsNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t IntsMeasureWire( const Ctx & ctx, const Ints & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( IntsNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = IntsMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t IntsSaveWire( const Ctx & ctx, const Ints & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !IntsNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = IntsSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == IntsMeasure( root ) +} + +inline int64_t IntsMeasure( const Ints * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return IntsMeasureWire( ctx, *root, allocator ); +} + +inline int64_t IntsSave( const Ints * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return IntsSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t IntsMeasure( const IntsBuilder & builder ) +{ + if ( builder.region != NULL ) { return IntsMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return IntsMeasureWire( ctx, *(const Ints *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t IntsSave( const IntsBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return IntsSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return IntsSaveWire( ctx, *(const Ints *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t IntsMeasureMessage( const Ints * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return IntsMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t IntsSaveMessage( const Ints * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return IntsSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t IntsMeasureMessage( const IntsBuilder & builder ) +{ + if ( builder.region != NULL ) { return IntsMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return IntsMeasureWire( ctx, *(const Ints *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t IntsSaveMessage( const IntsBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return IntsSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return IntsSaveWire( ctx, *(const Ints *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// IntsLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t IntsLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !IntsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = IntsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// IntsLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Ints * IntsLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Ints ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !IntsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = IntsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x2034c5d17c00ceb7ull; + Ints * root = new ( region ) Ints; // lifetime only: LoadBody's first act is IntsReset + IntsReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = IntsNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + IntsNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + IntsNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Ints ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + IntsLoadBody( r, nodes, *root ); + return root; +} + +// IntsLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t IntsLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !IntsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = IntsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// IntsLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Ints * IntsLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Ints ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !IntsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = IntsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x2034c5d17c00ceb7ull; + Ints * root = new ( region ) Ints; // lifetime only: LoadBody's first act is IntsReset + IntsReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Ints ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = IntsNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + IntsNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + IntsNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Ints ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + IntsLoadBody( r, nodes, *root ); + return root; +} + +// IntsLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool IntsLoadBuilder( IntsBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Ints * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0x2034c5d17c00ceb7ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = IntsNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + IntsNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = IntsLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Floats: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: FloatsBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Floats is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct FloatsBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + FloatsBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~FloatsBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + FloatsBuilder( const FloatsBuilder & ) = delete; + FloatsBuilder & operator=( const FloatsBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Floats * GetRoot() { return arena.locked ? NULL : (Floats *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Floats * AsConst() const { return (const Floats *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool FloatsBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Floats & root = *(const Floats *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = FloatsPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = FloatsExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + Floats * destination = new ( packed ) Floats; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !FloatsPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Floats on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// FloatsNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t FloatsNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// FloatsNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void FloatsNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// FloatsNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t FloatsNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// FloatsNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t FloatsNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// FloatsNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void FloatsNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = FloatsNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? FloatsNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool FloatsNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Floats & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return FloatsNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t FloatsMeasureWire( const Ctx & ctx, const Floats & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( FloatsNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = FloatsMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t FloatsSaveWire( const Ctx & ctx, const Floats & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !FloatsNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = FloatsSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == FloatsMeasure( root ) +} + +inline int64_t FloatsMeasure( const Floats * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return FloatsMeasureWire( ctx, *root, allocator ); +} + +inline int64_t FloatsSave( const Floats * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return FloatsSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t FloatsMeasure( const FloatsBuilder & builder ) +{ + if ( builder.region != NULL ) { return FloatsMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return FloatsMeasureWire( ctx, *(const Floats *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t FloatsSave( const FloatsBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return FloatsSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return FloatsSaveWire( ctx, *(const Floats *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t FloatsMeasureMessage( const Floats * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return FloatsMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t FloatsSaveMessage( const Floats * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return FloatsSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t FloatsMeasureMessage( const FloatsBuilder & builder ) +{ + if ( builder.region != NULL ) { return FloatsMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return FloatsMeasureWire( ctx, *(const Floats *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t FloatsSaveMessage( const FloatsBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return FloatsSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return FloatsSaveWire( ctx, *(const Floats *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// FloatsLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t FloatsLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !FloatsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = FloatsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// FloatsLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Floats * FloatsLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Floats ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !FloatsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = FloatsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x91638659f8f6ad42ull; + Floats * root = new ( region ) Floats; // lifetime only: LoadBody's first act is FloatsReset + FloatsReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = FloatsNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + FloatsNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + FloatsNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Floats ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + FloatsLoadBody( r, nodes, *root ); + return root; +} + +// FloatsLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t FloatsLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !FloatsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = FloatsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// FloatsLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Floats * FloatsLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Floats ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !FloatsWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = FloatsNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x91638659f8f6ad42ull; + Floats * root = new ( region ) Floats; // lifetime only: LoadBody's first act is FloatsReset + FloatsReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Floats ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = FloatsNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + FloatsNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + FloatsNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Floats ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + FloatsLoadBody( r, nodes, *root ); + return root; +} + +// FloatsLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool FloatsLoadBuilder( FloatsBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Floats * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0x91638659f8f6ad42ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = FloatsNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + FloatsNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = FloatsLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- the cooked form: point at a cook (docs/SPEC-TABLES.md §7) ---- + +// BytesOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH BytesAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Bytes * BytesOpen( const void * bytes, uint64_t length ) +{ + return (const Bytes *) TableCookOpen( bytes, length, (uint64_t) sizeof( Bytes ), (uint64_t) alignof( Bytes ) ); +} + +// IntsOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH IntsAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Ints * IntsOpen( const void * bytes, uint64_t length ) +{ + return (const Ints *) TableCookOpen( bytes, length, (uint64_t) sizeof( Ints ), (uint64_t) alignof( Ints ) ); +} + +// FloatsOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH FloatsAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Floats * FloatsOpen( const void * bytes, uint64_t length ) +{ + return (const Floats *) TableCookOpen( bytes, length, (uint64_t) sizeof( Floats ), (uint64_t) alignof( Floats ) ); +} + +// ---- the cooked form: WRITE a cook (docs/SPEC-TABLES.md §7.6) ---- +// +// The bytes are `schema cook`'s, and the tool stays the reference: the two +// writers are held to one file, byte for byte, in both byte orders. A cook is +// content-addressed by (asset hash, build version), so two writers of one +// instance produce ONE artifact or the pair means nothing. + +template inline bool BytesCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Bytes & value, TableByteOrder order ); +template inline bool IntsCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Ints & value, TableByteOrder order ); +template inline bool FloatsCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Floats & value, TableByteOrder order ); + +template inline bool BytesCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Bytes & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // data: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.after, 4, order ); + return true; +} + +template inline bool IntsCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Ints & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // values: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.after, 4, order ); + return true; +} + +template inline bool FloatsCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Floats & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // values: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, (uint64_t) value.after, 4, order ); + return true; +} + +template inline bool BytesCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Bytes & value, TableByteOrder order ); +template inline bool IntsCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Ints & value, TableByteOrder order ); +template inline bool FloatsCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Floats & value, TableByteOrder order ); + +// BytesCookExtent: Bytes's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool BytesCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Bytes & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // data: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.data ); + if ( !cursor.ok ) { return false; } + at = ( at + 0 ) & ~(int64_t) 0; // at alignof( uint8_t ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 1; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + table_cook_put( array + i * 1, (uint64_t) cursor[i], 1, order ); + } + } + return true; +} + +// IntsCookExtent: Ints's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool IntsCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Ints & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // values: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.values ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( int32_t ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 4; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + table_cook_put( array + i * 4, (uint64_t) cursor[i], 4, order ); + } + } + return true; +} + +// FloatsCookExtent: Floats's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FloatsCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Floats & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // values: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.values ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( float ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 4; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + { uint32_t bits = 0; memcpy( &bits, &cursor[i], 4 ); table_cook_put( array + i * 4, (uint64_t) bits, 4, order ); } + } + } + return true; +} + +// BytesCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool BytesCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Bytes & value, TableByteOrder order ) +{ + if ( !BytesCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return BytesCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// IntsCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool IntsCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Ints & value, TableByteOrder order ) +{ + if ( !IntsCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return IntsCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// FloatsCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool FloatsCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Floats & value, TableByteOrder order ) +{ + if ( !FloatsCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return FloatsCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// BytesCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool BytesCookLayout( const Ctx & ctx, const Bytes & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = BytesExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// BytesCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t BytesCookMeasureFrom( const Ctx & ctx, const Bytes & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( BytesNumberFrom( ctx, numbering, root ) && BytesCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// BytesCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool BytesCookFrom( const Ctx & ctx, const Bytes & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = BytesNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && BytesCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = BytesCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0xeeeea7adc131a244ull, 8, order ); // the root: fnv1a64( "Bytes" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// BytesCookMeasure / BytesCook over a REGION root — a locked builder's AsConst, a +// region BytesLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t BytesCookMeasure( const Bytes * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return BytesCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool BytesCook( const Bytes * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return BytesCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t BytesCookMeasure( const BytesBuilder & builder ) +{ + if ( builder.region != NULL ) { return BytesCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return BytesCookMeasureFrom( ctx, *(const Bytes *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool BytesCook( const BytesBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return BytesCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return BytesCookFrom( ctx, *(const Bytes *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// IntsCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool IntsCookLayout( const Ctx & ctx, const Ints & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = IntsExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// IntsCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t IntsCookMeasureFrom( const Ctx & ctx, const Ints & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( IntsNumberFrom( ctx, numbering, root ) && IntsCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// IntsCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool IntsCookFrom( const Ctx & ctx, const Ints & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = IntsNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && IntsCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = IntsCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0x2034c5d17c00ceb7ull, 8, order ); // the root: fnv1a64( "Ints" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// IntsCookMeasure / IntsCook over a REGION root — a locked builder's AsConst, a +// region IntsLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t IntsCookMeasure( const Ints * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return IntsCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool IntsCook( const Ints * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return IntsCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t IntsCookMeasure( const IntsBuilder & builder ) +{ + if ( builder.region != NULL ) { return IntsCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return IntsCookMeasureFrom( ctx, *(const Ints *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool IntsCook( const IntsBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return IntsCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return IntsCookFrom( ctx, *(const Ints *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// FloatsCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool FloatsCookLayout( const Ctx & ctx, const Floats & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = FloatsExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// FloatsCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t FloatsCookMeasureFrom( const Ctx & ctx, const Floats & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( FloatsNumberFrom( ctx, numbering, root ) && FloatsCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// FloatsCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool FloatsCookFrom( const Ctx & ctx, const Floats & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = FloatsNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && FloatsCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = FloatsCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0x91638659f8f6ad42ull, 8, order ); // the root: fnv1a64( "Floats" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// FloatsCookMeasure / FloatsCook over a REGION root — a locked builder's AsConst, a +// region FloatsLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t FloatsCookMeasure( const Floats * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return FloatsCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool FloatsCook( const Floats * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return FloatsCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t FloatsCookMeasure( const FloatsBuilder & builder ) +{ + if ( builder.region != NULL ) { return FloatsCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return FloatsCookMeasureFrom( ctx, *(const Floats *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool FloatsCook( const FloatsBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return FloatsCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return FloatsCookFrom( ctx, *(const Floats *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ---- relocatability, enforced: the wire is a pure length-prefixed +// stream AND the decoded storage is pointer-free — every closure type +// must stay trivially copyable and standard-layout, so instances can be +// memcpy'd, mmap'd, shared across processes, and walked through +// descriptor offsets. A failure here means a pointer, virtual or +// non-trivial member crept into generated storage. +// +// They ask the COMPILER ITSELF, which is what every C++ standard library +// answers the same two questions with — and it costs this header no +// include at all. +// A pointer FIELD is a TableRef — eight bytes and no address — so the +// property holds in BOTH forms: a fixed-size table is one relocatable +// struct, and a packed region is one relocatable block whose references +// are self-relative and therefore survive a plain memcpy. +static_assert( __is_trivially_copyable( Bytes ), "Bytes must stay relocatable" ); +static_assert( __is_standard_layout( Bytes ), "Bytes must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Ints ), "Ints must stay relocatable" ); +static_assert( __is_standard_layout( Ints ), "Ints must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Floats ), "Floats must stay relocatable" ); +static_assert( __is_standard_layout( Floats ), "Floats must stay standard-layout for offsetof" ); + +// ---- the cook's layout contract (docs/SPEC-TABLES.md §20.3) ---- +// +// The compiler derived every number below from the declaration and folded it +// into the BUILD VERSION; these asserts are this compiler saying whether it +// agrees. The model is not self-evidently right — on 32-bit System V +// alignof(uint64_t) is 4, not 8 — which is precisely why it is asserted +// rather than assumed. +static_assert( sizeof( Bytes ) == 24, "Bytes's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Bytes ) == 8, "Bytes's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Bytes, data ) == 0, "Bytes's field data moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Bytes, after ) == 16, "Bytes's field after moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Ints ) == 24, "Ints's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Ints ) == 8, "Ints's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Ints, values ) == 0, "Ints's field values moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Ints, after ) == 16, "Ints's field after moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Floats ) == 24, "Floats's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Floats ) == 8, "Floats's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Floats, values ) == 0, "Floats's field values moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Floats, after ) == 16, "Floats's field after moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); + +static_assert( alignof( uint8_t ) <= kTableAlign, "Bytes.data: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( int32_t ) <= kTableAlign, "Ints.values: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( float ) <= kTableAlign, "Floats.values: an unbounded array's element alignment must fit the arena's" ); + +// ---- reflection descriptors (tables only, docs/SPEC-TABLES.md) ---- + +inline const TableTypeInfo * BytesTableType(); +inline const TableTypeInfo * IntsTableType(); +inline const TableTypeInfo * FloatsTableType(); +// The descriptors are CONSTANT-INITIALISED data, and a field's target is +// the ADDRESS of another descriptor. These declarations are what let a +// self- or mutually-referential graph — Node naming itself through *Node — +// be expressed as constant data instead of a lazy link, which could not +// have been written race-free OR recursion-safe. The whole reflection +// surface is therefore immutable: read it from any thread, any time. +extern const TableTypeInfo BytesTableInfo; +extern const TableTypeInfo IntsTableInfo; +extern const TableTypeInfo FloatsTableInfo; + +inline const TableFieldInfo BytesTableFields[] = { + { "data", "data", "uint8", 0x855b556730a34a05ull, 6, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Bytes, data ), (uint32_t) sizeof( uint8_t ), (uint32_t) offsetof( Bytes, data.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Bytes, after ), (uint32_t) sizeof( Bytes::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo BytesTableInfo = { "Bytes", (uint32_t) sizeof( Bytes ), 2, BytesTableFields, +[]( void * p ) { BytesReset( *(Bytes *) p ); }, true }; +inline const TableTypeInfo * BytesTableType() { return &BytesTableInfo; } + +inline const TableFieldInfo IntsTableFields[] = { + { "values", "values", "int32", 0x21277bcf1a4d67fbull, 4, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Ints, values ), (uint32_t) sizeof( int32_t ), (uint32_t) offsetof( Ints, values.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Ints, after ), (uint32_t) sizeof( Ints::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo IntsTableInfo = { "Ints", (uint32_t) sizeof( Ints ), 2, IntsTableFields, +[]( void * p ) { IntsReset( *(Ints *) p ); }, true }; +inline const TableTypeInfo * IntsTableType() { return &IntsTableInfo; } + +inline const TableFieldInfo FloatsTableFields[] = { + { "values", "values", "float32", 0x21277bcf1a4d67fbull, 10, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Floats, values ), (uint32_t) sizeof( float ), (uint32_t) offsetof( Floats, values.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Floats, after ), (uint32_t) sizeof( Floats::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo FloatsTableInfo = { "Floats", (uint32_t) sizeof( Floats ), 2, FloatsTableFields, +[]( void * p ) { FloatsReset( *(Floats *) p ); }, true }; +inline const TableTypeInfo * FloatsTableType() { return &FloatsTableInfo; } + +// ---- the text form (docs/SPEC-TABLES.md §16) ---- + +// Bytes in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in ReportTable.cpp; link it to use them. +bool BytesFromJson( BytesBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t BytesToJsonMeasure( const Bytes * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t BytesToJson( const Bytes * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Ints in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in ReportTable.cpp; link it to use them. +bool IntsFromJson( IntsBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t IntsToJsonMeasure( const Ints * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t IntsToJson( const Ints * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Floats in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in ReportTable.cpp; link it to use them. +bool FloatsFromJson( FloatsBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t FloatsToJsonMeasure( const Floats * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t FloatsToJson( const Floats * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/SaveTable.cpp b/testdata/golden/tables/lists/SaveTable.cpp new file mode 100644 index 000000000..70e6c40f4 --- /dev/null +++ b/testdata/golden/tables/lists/SaveTable.cpp @@ -0,0 +1,3181 @@ +// Code generated by the schema compiler from Save.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — the TABLE wire's text form (docs/SPEC-TABLES.md §16). +// Compile this file to use FromJson / ToJson; a project that +// never reads or writes a text does not compile it and pays nothing. + +#include "SaveTable.h" + +#include // the text form: number formatting +#include // the text form: exact number conversion +#include // the text form: the runtime's decimal point + +// The guard is not vestigial. Several listdemo Table.cpp files may be +// concatenated into ONE translation unit — a unity build — and without it +// each would redefine the walk. It is also why the walk's functions may be +// weak (vague linkage) across separate objects: ODR requires their +// definitions to be token-identical, and the generic-walk gate is what +// proves that, byte for byte, across every generated .cpp. +#ifndef LISTDEMO_SCHEMA_TABLE_JSON +#define LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +// ---- the pointer adapters (docs/SPEC-TABLES.md §16.7) ---- +// +// The walk below is ONE walk, byte-identical in every generated .cpp, and a +// pointer is the one kind it cannot walk alone: reading one needs the +// builder's arena and writing one needs a region's deref, and neither exists +// in a unit that declares no pointer. So the walk calls these three and does +// not define them. A unit with no pointer defines them as stubs no field ever +// reaches; a pointered unit defines them in the graph half that follows the +// walk. + +struct TableJsonIn; +struct TableJsonOut; + +// a pointer field's object, or the `&node` reference standing in for it, into +// the slot; the cursor is on the opening brace +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// the node a pointer slot names, in place — or as `&node` when it is shared +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// the FIRST key of an object the walk is skipping begins with `&`: the cursor is +// on its value. A dropped definition still takes its label (§16.7); a fixed reader +// skips the value whole, as it skips everything else it does not place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); + +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- +// +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map +inline bool TableJsonIsMap( const TableFieldInfo * f ); +// the map as a plain JSON object keyed by the KEY, in ASCENDING key order +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that object back into the slot, in whatever order the text gives it +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); + +// ---- json walk: begin ---- +// +// The TEXT form (docs/SPEC-TABLES.md §16): one table, one text, one walk over the +// reflection descriptors (§8). Reading fills ONE caller-owned instance and +// allocates nothing beyond it; writing targets a caller buffer with the +// wire's measure/write symmetry. Everything AROUND this — which file goes +// with which instance, what key an instance is filed under, how instances +// link into a root table's collections — is a packer's opinion and stays +// with the tool that holds it. +// +// The dialect: trailing commas are accepted on read (the authoring files +// this exists for carry them) and never written; comments are not JSON and +// are refused; unknown keys are skipped and counted; a duplicate key is +// last-wins and counted; a key present with the wrong JSON type is skipped +// and counted, never coerced. + +static const int32_t kTableJsonMaxDepth = 128; + +// A key longer than this cannot name a field, so it is skipped as unknown. +static const int32_t kTableJsonMaxKey = 256; + +// The longest numeric token the walk will convert. Anything longer is a +// value no field can hold and counts as a kind mismatch. +static const int32_t kTableJsonMaxNumber = 512; + +// The decimal point the C runtime is CURRENTLY using. Number conversion is +// the one locale-sensitive corner of the grammar — JSON's point is always +// '.', the runtime's is whatever the program set — so every number crosses +// this one character on the way out and on the way back in. Nothing else in +// the walk consults the locale. +inline char TableJsonDecimalPoint() +{ + const struct lconv * conv = localeconv(); + if ( conv != NULL && conv->decimal_point != NULL && conv->decimal_point[0] != 0 ) + { + return conv->decimal_point[0]; + } + return '.'; +} + +// ---- storage access: the descriptors give an offset and a width, and the +// ---- storage is the HOST's, so every load and store goes through a width +// ---- switch rather than a memcpy into the low bytes of a wider word + +// finite: not a NaN, not an infinity. Written without — the walk's +// runtime surface stays the handful of functions it already names. +// A vocabulary entry the descriptor could not spell. The generated name +// functions answer "???" for a value outside the declared set, and that is +// not a name — writing it would put a spelling in the text that the reader +// then counts as unknown, turning a refusal into a silent loss. +inline bool TableJsonNamed( const char * name ) +{ + return name != NULL && strcmp( name, "???" ) != 0; +} + +inline bool TableJsonFinite( double v ) +{ + return v == v && v <= 1.7976931348623157e308 && v >= -1.7976931348623157e308; +} + +inline uint64_t TableJsonGetRaw( const void * storage, uint32_t width ) +{ + switch ( width ) + { + case 1: { uint8_t v = 0; memcpy( &v, storage, 1 ); return v; } + case 2: { uint16_t v = 0; memcpy( &v, storage, 2 ); return v; } + case 4: { uint32_t v = 0; memcpy( &v, storage, 4 ); return v; } + case 8: { uint64_t v = 0; memcpy( &v, storage, 8 ); return v; } + } + return 0; +} + +inline void TableJsonSetRaw( void * storage, uint32_t width, uint64_t value ) +{ + switch ( width ) + { + case 1: { uint8_t v = (uint8_t) value; memcpy( storage, &v, 1 ); break; } + case 2: { uint16_t v = (uint16_t) value; memcpy( storage, &v, 2 ); break; } + case 4: { uint32_t v = (uint32_t) value; memcpy( storage, &v, 4 ); break; } + case 8: { uint64_t v = value; memcpy( storage, &v, 8 ); break; } + } +} + +inline int64_t TableJsonGetSigned( const void * storage, uint32_t width ) +{ + uint64_t raw = TableJsonGetRaw( storage, width ); + if ( width < 8 ) + { + uint64_t sign = uint64_t( 1 ) << ( width * 8 - 1 ); + if ( ( raw & sign ) != 0 ) + { + raw |= ~( ( sign << 1 ) - 1 ); + } + } + return (int64_t) raw; +} + +// ---- the WIDE kinds (docs/SPEC-TABLES.md §3, §16.2) ---- +// +// The 128-bit integers and the fixed-point family convert EXACTLY, over two +// 64-bit lanes: a 128-bit integer is a decimal integer, a fixed value a +// decimal in WHOLE UNITS (1.0, -0.25, 3.0000152587890625) and nothing +// on either path passes through a double. Nothing here needs a 128-bit type +// either, which is what keeps this walk one text for every unit. +struct TableJsonWide +{ + uint64_t lo; + uint64_t hi; +}; + +inline bool TableJsonKindWide( uint8_t kind ) { return kind >= 18 && kind <= 29; } +inline bool TableJsonKindWideSigned( uint8_t kind ) { return kind == 18 || ( kind >= 20 && kind <= 24 ); } +inline bool TableJsonKindFixed( uint8_t kind ) { return kind >= 20 && kind <= 29; } + +inline bool TableJsonWideZero( TableJsonWide v ) { return v.lo == 0 && v.hi == 0; } +inline bool TableJsonWideNegative( TableJsonWide v ) { return ( v.hi >> 63 ) != 0; } + +inline int TableJsonWideCompare( TableJsonWide a, TableJsonWide b, bool is_signed ) +{ + if ( is_signed && TableJsonWideNegative( a ) != TableJsonWideNegative( b ) ) { return TableJsonWideNegative( a ) ? -1 : 1; } + if ( a.hi != b.hi ) { return a.hi < b.hi ? -1 : 1; } + if ( a.lo != b.lo ) { return a.lo < b.lo ? -1 : 1; } + return 0; +} + +inline TableJsonWide TableJsonWideShl( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.hi = v.lo << ( n - 64 ); return r; } + r.hi = ( v.hi << n ) | ( v.lo >> ( 64 - n ) ); + r.lo = v.lo << n; + return r; +} + +inline TableJsonWide TableJsonWideShr( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.lo = v.hi >> ( n - 64 ); return r; } + r.lo = ( v.lo >> n ) | ( v.hi << ( 64 - n ) ); + r.hi = v.hi >> n; + return r; +} + +inline TableJsonWide TableJsonWideNeg( TableJsonWide v ) +{ + TableJsonWide r; + r.lo = ~v.lo + 1; + r.hi = ~v.hi + ( r.lo == 0 ? 1 : 0 ); + return r; +} + +// v = v * m + a; the return is the carry out of 128 bits +inline uint32_t TableJsonWideMulAdd( TableJsonWide * v, uint32_t m, uint32_t a ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t carry = a; + for ( int i = 0; i < 4; i++ ) + { + uint64_t p = limb[i] * m + carry; + limb[i] = p & 0xffffffffull; + carry = p >> 32; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) carry; +} + +// v = v / d; the return is the remainder +inline uint32_t TableJsonWideDiv( TableJsonWide * v, uint32_t d ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t rem = 0; + for ( int i = 3; i >= 0; i-- ) + { + uint64_t cur = ( rem << 32 ) | limb[i]; + limb[i] = cur / d; + rem = cur % d; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) rem; +} + +// The storage of a wide kind, as lanes. A sixteen-byte storage is serialize's +// pair — native __int128 in the host's byte order, or the emulated struct with +// its low lane first — so the lanes are read in the host's order; a narrower +// storage is one lane, sign-extended for a signed kind. +inline TableJsonWide TableJsonWideLoad( const void * storage, uint32_t width, bool is_signed ) +{ + TableJsonWide v = { 0, 0 }; + if ( width == 16 ) + { + uint64_t half[2]; + memcpy( half, storage, 16 ); + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + v.lo = little ? half[0] : half[1]; + v.hi = little ? half[1] : half[0]; + return v; + } + v.lo = is_signed ? (uint64_t) TableJsonGetSigned( storage, width ) : TableJsonGetRaw( storage, width ); + v.hi = ( is_signed && ( v.lo >> 63 ) != 0 ) ? ~uint64_t( 0 ) : 0; + return v; +} + +inline void TableJsonWideStore( void * storage, uint32_t width, TableJsonWide v ) +{ + if ( width == 16 ) + { + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + uint64_t half[2]; + half[0] = little ? v.lo : v.hi; + half[1] = little ? v.hi : v.lo; + memcpy( storage, half, 16 ); + return; + } + TableJsonSetRaw( storage, width, v.lo ); +} + +// a counted field's companion: a string's length, a bytes' length, a counted +// array's count. Bounded by the declared extent on the way out, so a storage +// invariant a caller broke cannot walk off the end of the array. +inline int32_t TableJsonCount( const void * base, const TableFieldInfo * f ) +{ + if ( !f->counted ) + { + return f->array_bound; + } + int32_t count = 0; + memcpy( &count, (const uint8_t *) base + f->count_offset, sizeof( count ) ); + if ( count < 0 ) { count = 0; } + if ( count > f->array_bound ) { count = f->array_bound; } + return count; +} + +inline void TableJsonSetCount( void * base, const TableFieldInfo * f, int32_t count ) +{ + if ( f->counted ) + { + memcpy( (uint8_t *) base + f->count_offset, &count, sizeof( count ) ); + } +} + +// ---- what a field's kind expects to see in the text ---- +// +// One classifier, consulted by both directions, so a reader and a writer can +// never disagree about a kind's JSON form. 'o' object, 'a' array, 's' +// string, 'n' number, 'b' boolean. +// +// A vocabulary field is spelled by NAME: an enum is one name, a flags mask +// is the array of the names of its set bits. The two are told apart by the +// id column — an enum variant rides under a wire id, a flags BIT never does +// (docs/SPEC-TABLES.md §4), so a name function with no id function is flags. +// +// bytes(N) is the one kind whose element kind does not decide its form: it +// shares u8 with a plain array of u8, and rides as base64. The schema type +// name settles it, and "bytes" is a keyword no declaration can claim. +inline bool TableJsonIsBytes( const TableFieldInfo * f ) +{ + return f->is_array && f->kind == 6 && strcmp( f->type_name, "bytes" ) == 0; +} + +// An ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): its JSON form is an OBJECT +// keyed by variant name, not a positional array, because that is what the +// storage is — one slot per variant, addressed by the variant. +inline bool TableJsonIsKeyed( const TableFieldInfo * f ) +{ + return f->key_name != NULL; +} + +// THE KEY A STORAGE SLOT HOLDS (§2.4, §8): the storage shifts left, so slot i +// holds the key i + 1 and nothing is stored for None. This is the ONE place +// the walker spells the shift. +inline uint64_t TableJsonKeyedSlotKey( int64_t slot ) +{ + return (uint64_t) ( slot + 1 ); +} + +// A slot whose key names a variant of the keying enum. Every slot in +// [0, array_bound) does, unless the enum carries max-headroom variants outside +// a table closure, where a reserved value names nothing and its key id is 0 — +// the reserved id no declared name can fold to (§5). +inline bool TableJsonKeyedSlotValid( const TableFieldInfo * f, int64_t slot ) +{ + return f->key_id( TableJsonKeyedSlotKey( slot ) ) != 0; +} + +inline bool TableJsonIsFlags( const TableFieldInfo * f ) +{ + return f->enum_name != NULL && f->variant_id == NULL; +} + +inline bool TableJsonIsEnum( const TableFieldInfo * f ) +{ + return f->variant_id != NULL && f->arms == NULL; +} + +inline char TableJsonShape( const TableFieldInfo * f ) +{ + if ( TableJsonIsMap( f ) ) return 'o'; // a MAP: an object keyed by the KEY (§2.8) + if ( f->kind == 12 ) return 's'; // string + if ( TableJsonIsBytes( f ) ) return 's'; // bytes: base64 + if ( TableJsonIsKeyed( f ) ) return 'o'; // an object keyed by variant NAME + if ( f->is_array ) return 'a'; + if ( f->arms != NULL ) return 'o'; // union: an object with ONE key + if ( f->kind == 13 ) return 'o'; // nested table or type + if ( f->kind == 17 ) return f->table != NULL ? 'o' : 's'; // a pointer: the pointee's object in place, or null (§16.7); a byte buffer's string (§2.5) + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// the ELEMENT shape of an array field — the same classifier one level down +inline char TableJsonElementShape( const TableFieldInfo * f ) +{ + if ( f->arms != NULL ) return 'o'; // an element of an array of unions: one key, the arm (§2.6) + if ( f->kind == 13 ) return 'o'; + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// A guarded group rides only when its guard reads true — the wire's own +// elision (§4), carried into the text so a text and a wire written from one +// instance say the same thing. The guard is spelled as its branch condition +// over bool fields of the SAME type ("at_rest", "!at_rest", +// "active && has_target"), so evaluating it is a walk of the same +// descriptor. Nothing is inferred in the other direction: reading places +// every key it can name, and the guard is a plain bool key (§16.2). +inline bool TableJsonGuardHolds( const void * base, const TableTypeInfo * info, const char * guard ) +{ + const char * p = guard; + for ( ;; ) + { + while ( *p == ' ' || *p == '&' ) { p++; } + if ( *p == 0 ) { return true; } + bool want = true; + if ( *p == '!' ) { want = false; p++; } + const char * start = p; + while ( *p != 0 && *p != ' ' && *p != '&' ) { p++; } + size_t length = (size_t) ( p - start ); + bool value = false; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( strlen( f->name ) == length && strncmp( f->name, start, length ) == 0 ) + { + value = TableJsonGetRaw( (const uint8_t *) base + f->offset, f->elem_size ) != 0; + break; + } + } + if ( value != want ) { return false; } + } +} + +// ---- writing ---- + +// The writer sink MEASURES when the buffer is NULL and WRITES when it is +// not, over one code path — so measure and write agree byte for byte, the +// wire's invariant (§9) carried across. +struct TableJsonOut +{ + char * buffer; + int64_t capacity; + int64_t offset; + bool overflow; + void * graph; // the pointered write's identity map (§16.7); NULL for a fixed table + + void raw( const char * data, int64_t count ) + { + if ( buffer != NULL ) + { + if ( offset + count > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) count ); + } + offset += count; + } + void put( char c ) { raw( &c, 1 ); } + void text( const char * s ) { raw( s, (int64_t) strlen( s ) ); } + void line( int32_t depth ) + { + put( '\n' ); + for ( int32_t i = 0; i < depth; i++ ) { raw( " ", 2 ); } + } +}; + +inline const char * TableJsonBase64Alphabet() +{ + return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; +} + +inline void TableJsonWriteBase64( TableJsonOut & out, const uint8_t * data, int32_t length ) +{ + const char * alphabet = TableJsonBase64Alphabet(); + out.put( '"' ); + int32_t i = 0; + for ( ; i + 3 <= length; i += 3 ) + { + uint32_t triple = ( uint32_t( data[i] ) << 16 ) | ( uint32_t( data[i+1] ) << 8 ) | uint32_t( data[i+2] ); + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], + alphabet[ ( triple >> 6 ) & 0x3f ], alphabet[ triple & 0x3f ] }; + out.raw( quad, 4 ); + } + if ( i < length ) + { + int32_t left = length - i; + uint32_t triple = uint32_t( data[i] ) << 16; + if ( left == 2 ) { triple |= uint32_t( data[i+1] ) << 8; } + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], '=', '=' }; + if ( left == 2 ) { quad[2] = alphabet[ ( triple >> 6 ) & 0x3f ]; } + out.raw( quad, 4 ); + } + out.put( '"' ); +} + +// One UTF-8 sequence at s, or -1 when the bytes there are not one. Rejects +// the lot: a stray continuation, an overlong form, a surrogate half, and +// anything past U+10FFFF. +inline int32_t TableJsonUtf8( const char * s, int32_t remaining, int32_t * width ) +{ + unsigned char lead = (unsigned char) s[0]; + int32_t want = 0; + int32_t code = 0; + if ( lead < 0x80 ) { *width = 1; return lead; } + else if ( lead >= 0xc2 && lead <= 0xdf ) { want = 2; code = lead & 0x1f; } + else if ( lead >= 0xe0 && lead <= 0xef ) { want = 3; code = lead & 0x0f; } + else if ( lead >= 0xf0 && lead <= 0xf4 ) { want = 4; code = lead & 0x07; } + else { return -1; } + if ( remaining < want ) { return -1; } + for ( int32_t i = 1; i < want; i++ ) + { + unsigned char next = (unsigned char) s[i]; + if ( ( next & 0xc0 ) != 0x80 ) { return -1; } + code = ( code << 6 ) | ( next & 0x3f ); + } + if ( want == 3 && code < 0x800 ) { return -1; } // overlong + if ( want == 4 && code < 0x10000 ) { return -1; } // overlong + if ( code >= 0xd800 && code <= 0xdfff ) { return -1; } // a surrogate half + if ( code > 0x10ffff ) { return -1; } + *width = want; + return code; +} + +// A JSON text MUST be valid UTF-8 (RFC 8259 §8.1). The read path is +// byte-transparent — the wire imposes no encoding (§3) and a string may hold +// anything — so the WRITER is where that obligation is met: a byte that is +// not part of a well-formed sequence is written as U+FFFD, one per bad byte, +// and never raw. A text this walk writes is therefore readable by any +// conforming parser, which a raw byte would not be. The cost is stated +// plainly: for a string holding invalid UTF-8, the round trip is NOT +// byte-identical, because the alternative is emitting a text that is not +// JSON. +inline void TableJsonWriteString( TableJsonOut & out, const char * s, int32_t length ) +{ + static const char hex[] = "0123456789abcdef"; + out.put( '"' ); + for ( int32_t i = 0; i < length; i++ ) + { + unsigned char c = (unsigned char) s[i]; + switch ( c ) + { + case '"': out.raw( "\\\"", 2 ); break; + case '\\': out.raw( "\\\\", 2 ); break; + case '\b': out.raw( "\\b", 2 ); break; + case '\f': out.raw( "\\f", 2 ); break; + case '\n': out.raw( "\\n", 2 ); break; + case '\r': out.raw( "\\r", 2 ); break; + case '\t': out.raw( "\\t", 2 ); break; + default: + if ( c < 0x20 ) + { + char escape[6] = { '\\', 'u', '0', '0', hex[ c >> 4 ], hex[ c & 0xf ] }; + out.raw( escape, 6 ); + } + else if ( c < 0x80 ) + { + out.put( (char) c ); + } + else + { + int32_t width = 0; + if ( TableJsonUtf8( s + i, length - i, &width ) < 0 ) + { + out.raw( "\xef\xbf\xbd", 3 ); // U+FFFD, one per bad byte + } + else + { + out.raw( s + i, width ); + i += width - 1; + } + } + break; + } + } + out.put( '"' ); +} + +inline void TableJsonWriteUnsigned( TableJsonOut & out, uint64_t value ) +{ + char digits[24]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) ( value % 10 ) ); + value /= 10; + } while ( value != 0 ); + char text[24]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); +} + +inline void TableJsonWriteSigned( TableJsonOut & out, int64_t value ) +{ + if ( value < 0 ) + { + out.put( '-' ); + TableJsonWriteUnsigned( out, uint64_t( 0 ) - (uint64_t) value ); + return; + } + TableJsonWriteUnsigned( out, (uint64_t) value ); +} + +// A wide kind writes its raw storage as §16.2's text: a 128-bit integer as a +// decimal integer; a fixed value in WHOLE UNITS as the shortest exact decimal +// with at least one fractional digit (1.0, -0.25), the spelling the schema text +// gives a fixed default. The fraction terminates because a dyadic fraction has +// a finite decimal expansion — at most F digits. +inline void TableJsonWriteWide( TableJsonOut & out, const void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + TableJsonWide v = TableJsonWideLoad( storage, f->elem_size, is_signed ); + if ( is_signed && TableJsonWideNegative( v ) ) + { + out.put( '-' ); + v = TableJsonWideNeg( v ); + } + int frac = f->frac_bits; + TableJsonWide whole = TableJsonWideShr( v, frac ); + char digits[40]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) TableJsonWideDiv( &whole, 10 ) ); + } while ( !TableJsonWideZero( whole ) ); + char text[40]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); + if ( !TableJsonKindFixed( f->kind ) ) { return; } + out.put( '.' ); + // the fraction bits alone: v with everything at and above bit F cleared + TableJsonWide fraction = v; + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + if ( frac == 0 ) { fraction.lo = 0; } + if ( TableJsonWideZero( fraction ) ) + { + out.put( '0' ); + return; + } + while ( !TableJsonWideZero( fraction ) ) + { + // ×10: the digit is what lands at and above bit F, including the + // carry out of 128 bits when F leaves no room for it below + uint32_t carry = TableJsonWideMulAdd( &fraction, 10, 0 ); + uint64_t digit = TableJsonWideShr( fraction, frac ).lo; + if ( frac > 64 ) { digit |= uint64_t( carry ) << ( 128 - frac ); } + out.put( (char) ( '0' + (int) digit ) ); + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + } +} + +// A float writes at the SHORTEST precision that reads back as the same value +// at the field's own width, so a round trip is exact and a text stays +// readable. Non-finite values have no JSON spelling at all, and the writer +// REFUSES rather than losing one silently — the same rule measure and save +// already apply to an enum value no variant names (§5). +inline bool TableJsonWriteFloat( TableJsonOut & out, double value, bool single ) +{ + if ( !TableJsonFinite( value ) ) { return false; } + char text[64]; + int low = single ? 6 : 15; + int high = single ? 9 : 17; + int length = 0; + for ( int digits = low; ; digits++ ) + { + length = snprintf( text, sizeof( text ), "%.*g", digits, value ); + if ( length <= 0 || length >= (int) sizeof( text ) ) { return false; } + if ( digits >= high ) { break; } + // the round-trip check runs BEFORE the decimal point is normalised: + // the token still carries whatever point snprintf just produced + if ( single ) + { + if ( (double) strtof( text, NULL ) == value ) { break; } + } + else + { + if ( strtod( text, NULL ) == value ) { break; } + } + } + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int i = 0; i < length; i++ ) + { + if ( text[i] == point ) { text[i] = '.'; } + } + } + out.raw( text, length ); + return true; +} + +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration writes through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ); + +// one scalar, at one storage address: a nested object, a union, a +// vocabulary, or a number +inline bool TableJsonWriteScalar( TableJsonOut & out, const void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; None is {} + const TableUnionInfo * arms = f->arms(); + uint64_t tag = TableJsonGetRaw( (const uint8_t *) storage + arms->tag_offset, arms->tag_size ); + if ( tag == 0 ) + { + out.raw( "{}", 2 ); + return true; + } + if ( (int64_t) tag > f->enum_max ) + { + return false; // a tag no arm names, exactly as measure refuses it + } + const char * arm = f->enum_name( tag ); + // and refuse on the NAME, not merely on the bound: §16.2 says a value + // no variant NAMES is refused, so the check is the name. Writing + // whatever came back would emit "???", a spelling the reader counts + // as unknown — a silent round-trip loss in place of a refusal. + if ( !TableJsonNamed( arm ) ) { return false; } + out.put( '{' ); + out.line( depth + 1 ); + TableJsonWriteString( out, arm, (int32_t) strlen( arm ) ); + out.raw( ": ", 2 ); + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2): an arm that names + // no declaration carries the FIELD descriptor a field of its type + // would carry, offsets taken inside the union storage (§2.6), so the + // value walks through the field writer one key down. + if ( arms->arms[tag].field != NULL ) + { + if ( !TableJsonWriteField( out, storage, arms->arms[tag].field, depth + 1 ) ) + { + return false; + } + } + else if ( arms->arms[tag].table == NULL ) + { + out.raw( "null", 4 ); // a payload-free arm: the name selects it (§2.6) + } + else if ( !TableJsonWriteValue( out, (const uint8_t *) storage + arms->arms[tag].offset, arms->arms[tag].table, depth + 1 ) ) + { + return false; + } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->kind == 13 ) + { + return TableJsonWriteValue( out, storage, f->table, depth ); + } + if ( TableJsonIsEnum( f ) ) + { + uint64_t value = TableJsonGetRaw( storage, f->elem_size ); + // a value no variant names has no text spelling, exactly as it has no + // wire identity: the writer REFUSES rather than writing None over it, + // the rule measure and save already apply (docs/SPEC-TABLES.md §5) + if ( (int64_t) value > f->enum_max ) { return false; } + if ( value != 0 && f->variant_id( value ) == 0 ) { return false; } + const char * name = f->enum_name( value ); + if ( !TableJsonNamed( name ) ) { return false; } + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + return true; + } + if ( TableJsonIsFlags( f ) ) + { + uint64_t bits = TableJsonGetRaw( storage, f->elem_size ); + if ( bits == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + bool first = true; + for ( int64_t bit = 0; bit < 64; bit++ ) + { + if ( ( bits & ( uint64_t( 1 ) << bit ) ) == 0 ) { continue; } + if ( bit > f->enum_max ) + { + return false; // a bit no variant names has no text spelling + } + const char * name = f->enum_name( (uint64_t) bit ); + if ( !TableJsonNamed( name ) ) { return false; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + } + out.line( depth ); + out.put( ']' ); + return true; + } + switch ( f->kind ) + { + case 1: + out.text( TableJsonGetRaw( storage, f->elem_size ) != 0 ? "true" : "false" ); + return true; + case 10: + { + float v = 0.0f; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, (double) v, true ); + } + case 11: + { + double v = 0.0; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, v, false ); + } + case 2: case 3: case 4: case 5: + TableJsonWriteSigned( out, TableJsonGetSigned( storage, f->elem_size ) ); + return true; + default: + if ( TableJsonKindWide( f->kind ) ) + { + TableJsonWriteWide( out, storage, f ); + return true; + } + TableJsonWriteUnsigned( out, TableJsonGetRaw( storage, f->elem_size ) ); + return true; + } +} + +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ) +{ + const uint8_t * storage = (const uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonWriteMap( out, (const void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } + if ( f->kind == 17 && !f->is_array ) + { + return TableJsonWritePointer( out, storage, f, depth ); + } + if ( f->kind == 17 ) + { + // an ARRAY OF POINTERS (§2.1): the pointer row per element — the + // pointee's object in place, null, or `&node` for a shared one (§16.7) + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWritePointer( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; + } + if ( f->kind == 12 ) + { + TableJsonWriteString( out, (const char *) storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + TableJsonWriteBase64( out, storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + // one entry per SLOT, keyed by the variant that owns it, so inserting + // a variant next season moves nothing in the text either. Slot i holds + // the key i + 1: nothing is stored for None, so nothing is written for it. + out.put( '{' ); + bool first = true; + for ( int64_t slot = 0; slot < f->array_bound; slot++ ) + { + if ( !TableJsonKeyedSlotValid( f, slot ) ) { continue; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + const char * key = f->key_name( TableJsonKeyedSlotKey( slot ) ); + TableJsonWriteString( out, key, (int32_t) strlen( key ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteScalar( out, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + if ( first ) { out.raw( "}", 1 ); return true; } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->is_array ) + { + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWriteScalar( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + out.line( depth ); + out.put( ']' ); + return true; + } + return TableJsonWriteScalar( out, storage, f, depth ); +} + +// One instance's fields, in DECLARATION ORDER, defaults included — a text is +// for people and tools, and a text that elides is a text a reader has to know +// the schema to complete. `any` says whether the object is already open on +// entry — a shared node's `&node` opens it before the fields (§16.7) — and +// whether it is open on return. +inline bool TableJsonWriteFields( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth, bool & any ) +{ + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( f->guard[0] != 0 && !TableJsonGuardHolds( base, info, f->guard ) ) { continue; } + // an ABSENT optional writes no key: presence of the key IS the + // presence (§16.2), so an absent field is an absent key and nothing + // else would read back as absent + if ( f->optional && + TableJsonGetRaw( (const uint8_t *) base + f->present_offset, 1 ) == 0 ) + { + continue; + } + if ( !any ) { out.put( '{' ); } + else { out.put( ',' ); } + any = true; + out.line( depth + 1 ); + TableJsonWriteString( out, f->json, (int32_t) strlen( f->json ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, base, f, depth + 1 ) ) { return false; } + } + return true; +} + +// One instance as one object. The writer carries the reader's depth cap +// (§16.2): a pointer chain nests as deep as it is long (§16.7), and a text the +// writer produced past the cap would be a text the reader refuses. +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { return false; } + bool any = false; + if ( !TableJsonWriteFields( out, base, info, depth, any ) ) { return false; } + if ( !any ) + { + out.raw( "{}", 2 ); + return true; + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- reading ---- + +struct TableJsonIn +{ + const char * text; + int64_t size; + int64_t pos; + TableReport * report; + bool bad; // the text is not JSON: the walk stops and keeps what it placed + void * graph; // the pointered read's builder and label map (§16.7); NULL for a fixed table +}; + +inline void TableJsonSpace( TableJsonIn & in ) +{ + while ( in.pos < in.size ) + { + char c = in.text[in.pos]; + if ( c == ' ' || c == '\t' || c == '\n' || c == '\r' ) { in.pos++; continue; } + // comments are not JSON, and a walk that guessed at one would be + // reading a dialect nobody wrote down + if ( c == '/' ) { in.bad = true; } + return; + } +} + +inline char TableJsonPeek( TableJsonIn & in ) +{ + TableJsonSpace( in ); + return in.pos < in.size ? in.text[in.pos] : 0; +} + +// the shape of the value sitting at the cursor, without consuming it +inline char TableJsonValueShape( TableJsonIn & in ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return 'o'; + case '[': return 'a'; + case '"': return 's'; + case 't': case 'f': return 'b'; + case 'n': return 'z'; + case 0: return 0; + default: return 'n'; + } +} + +inline bool TableJsonLiteral( TableJsonIn & in, const char * word ) +{ + int64_t length = (int64_t) strlen( word ); + if ( in.pos + length > in.size || memcmp( in.text + in.pos, word, (size_t) length ) != 0 ) + { + in.bad = true; + return false; + } + in.pos += length; + return true; +} + +// one \uXXXX escape body; -1 when the four hex digits are not there +inline int TableJsonHex4( TableJsonIn & in ) +{ + if ( in.pos + 4 > in.size ) { return -1; } + int value = 0; + for ( int i = 0; i < 4; i++ ) + { + char c = in.text[in.pos + i]; + int digit; + if ( c >= '0' && c <= '9' ) { digit = c - '0'; } + else if ( c >= 'a' && c <= 'f' ) { digit = c - 'a' + 10; } + else if ( c >= 'A' && c <= 'F' ) { digit = c - 'A' + 10; } + else { return -1; } + value = ( value << 4 ) | digit; + } + in.pos += 4; + return value; +} + +inline int32_t TableJsonEncodeUtf8( uint32_t code, char * unit ) +{ + if ( code < 0x80 ) { unit[0] = (char) code; return 1; } + if ( code < 0x800 ) + { + unit[0] = (char) ( 0xc0 | ( code >> 6 ) ); + unit[1] = (char) ( 0x80 | ( code & 0x3f ) ); + return 2; + } + if ( code < 0x10000 ) + { + unit[0] = (char) ( 0xe0 | ( code >> 12 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( code & 0x3f ) ); + return 3; + } + unit[0] = (char) ( 0xf0 | ( code >> 18 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 12 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[3] = (char) ( 0x80 | ( code & 0x3f ) ); + return 4; +} + +// Scan one JSON string into a caller buffer. Bytes are appended ONE CODE +// POINT AT A TIME — an escape's encoding, or a UTF-8 sequence read whole — +// so a string longer than the field is clamped AT A CODE POINT BOUNDARY and +// never cut through a multi-byte character. Clamping is counted, never +// fatal, exactly as it is on the wire (§4). A NULL destination scans past a +// string without keeping it. +inline bool TableJsonScanString( TableJsonIn & in, char * out, int32_t capacity, int32_t * length ) +{ + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + int32_t placed = 0; + bool clamped = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos]; + if ( c == '"' ) { in.pos++; break; } + char unit[4]; + int32_t unit_length = 0; + if ( c == '\\' ) + { + in.pos++; + if ( in.pos >= in.size ) { in.bad = true; return false; } + char escape = in.text[in.pos++]; + switch ( escape ) + { + case '"': unit[0] = '"'; unit_length = 1; break; + case '\\': unit[0] = '\\'; unit_length = 1; break; + case '/': unit[0] = '/'; unit_length = 1; break; + case 'b': unit[0] = '\b'; unit_length = 1; break; + case 'f': unit[0] = '\f'; unit_length = 1; break; + case 'n': unit[0] = '\n'; unit_length = 1; break; + case 'r': unit[0] = '\r'; unit_length = 1; break; + case 't': unit[0] = '\t'; unit_length = 1; break; + case 'u': + { + int high = TableJsonHex4( in ); + if ( high < 0 ) { in.bad = true; return false; } + uint32_t code = (uint32_t) high; + if ( high >= 0xd800 && high <= 0xdbff && in.pos + 2 <= in.size && + in.text[in.pos] == '\\' && in.text[in.pos + 1] == 'u' ) + { + int64_t mark = in.pos; + in.pos += 2; + int low = TableJsonHex4( in ); + if ( low >= 0xdc00 && low <= 0xdfff ) + { + code = 0x10000 + ( ( (uint32_t) high - 0xd800 ) << 10 ) + ( (uint32_t) low - 0xdc00 ); + } + else + { + in.pos = mark; // a lone lead surrogate rides as itself + } + } + // a surrogate half that never found its partner has no + // UTF-8 encoding: encoding it anyway would manufacture + // CESU-8 — invalid UTF-8 — out of input that was valid + // JSON, so it reads as the replacement character + if ( code >= 0xd800 && code <= 0xdfff ) { code = 0xfffd; } + unit_length = TableJsonEncodeUtf8( code, unit ); + break; + } + default: in.bad = true; return false; + } + } + else if ( (unsigned char) c < 0x20 ) + { + in.bad = true; // a raw control character is not a JSON string body + return false; + } + else + { + // a UTF-8 sequence read WHOLE, so the clamp below can only land + // between code points. Only bytes that ACTUALLY look like + // continuations are taken: the wire imposes no encoding (§3), so + // a string may legitimately hold a stray lead byte, and one at + // the end of a text must not swallow the closing quote. + unsigned char lead = (unsigned char) c; + int32_t want = 1; + if ( ( lead & 0xe0 ) == 0xc0 ) { want = 2; } + else if ( ( lead & 0xf0 ) == 0xe0 ) { want = 3; } + else if ( ( lead & 0xf8 ) == 0xf0 ) { want = 4; } + unit[0] = c; + in.pos++; + unit_length = 1; + while ( unit_length < want && in.pos < in.size && + ( (unsigned char) in.text[in.pos] & 0xc0 ) == 0x80 ) + { + unit[unit_length++] = in.text[in.pos++]; + } + } + if ( out == NULL ) + { + placed += unit_length; // measured and not kept: a byte buffer's read sizes its node this way (§2.5) + } + else if ( placed + unit_length <= capacity ) + { + memcpy( out + placed, unit, (size_t) unit_length ); + placed += unit_length; + } + else + { + clamped = true; + } + } + if ( clamped ) { in.report->clamped++; } + if ( length != NULL ) { *length = placed; } + return true; +} + +// the numeric token at the cursor, copied out whole; false = not a number +// Scan one number, to JSON's OWN grammar (RFC 8259 §6) and not to a run of +// number-ish characters: +// +// number = [ "-" ] int [ frac ] [ exp ] +// int = "0" / ( digit1-9 *digit ) +// frac = "." 1*digit +// exp = ( "e" / "E" ) [ "-" / "+" ] 1*digit +// +// Scanning the production is what makes a typo in an authoring file a +// DIAGNOSTIC rather than a value: "1-2" scans as 1 and leaves "-2" where the +// object expects a comma, so the text is malformed — which is what §16.2 +// already promises. A permissive scan would hand "1-2" to a digit loop and +// report a clamp, and a config pipeline would never hear about it. Leading +// "+", leading zeros, ".5" and "3." are not JSON either. +inline bool TableJsonWalkNumber( TableJsonIn & in, bool * integral ) +{ + TableJsonSpace( in ); + bool whole = true; + if ( in.pos < in.size && in.text[in.pos] == '-' ) { in.pos++; } + // int: a lone zero, or a non-zero digit and any digits after it + if ( in.pos >= in.size ) { return false; } + if ( in.text[in.pos] == '0' ) + { + in.pos++; + } + else if ( in.text[in.pos] >= '1' && in.text[in.pos] <= '9' ) + { + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + } + else + { + return false; + } + // frac + if ( in.pos < in.size && in.text[in.pos] == '.' ) + { + in.pos++; + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + // exp + if ( in.pos < in.size && ( in.text[in.pos] == 'e' || in.text[in.pos] == 'E' ) ) + { + in.pos++; + if ( in.pos < in.size && ( in.text[in.pos] == '-' || in.text[in.pos] == '+' ) ) { in.pos++; } + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + *integral = whole; + return true; +} + +// the same production, with the token kept for conversion +inline bool TableJsonScanNumber( TableJsonIn & in, char * token, int32_t capacity, int32_t * length, bool * integral ) +{ + TableJsonSpace( in ); + int64_t start = in.pos; + if ( !TableJsonWalkNumber( in, integral ) ) { return false; } + int64_t count = in.pos - start; + if ( count <= 0 || count >= capacity ) { return false; } + memcpy( token, in.text + start, (size_t) count ); + token[count] = 0; + *length = (int32_t) count; + return true; +} + +// the token's exact double, through the runtime's own converter — which +// speaks the LOCALE's decimal point, so the token crosses back over that +// character on its way in +inline double TableJsonTokenDouble( const char * token, int32_t length, bool single ) +{ + char work[kTableJsonMaxNumber]; + memcpy( work, token, (size_t) length ); + work[length] = 0; + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int32_t i = 0; i < length; i++ ) + { + if ( work[i] == '.' ) { work[i] = point; } + } + } + if ( single ) { return (double) strtof( work, NULL ); } + return strtod( work, NULL ); +} + +// the token's exact integer, parsed digit by digit so no width and no +// locale can move it. Saturation is reported as a clamp, the wire's rule for +// a value outside what the reader can hold (§4). +inline int64_t TableJsonTokenInteger( const char * token, int32_t length, bool is_signed, bool * saturated ) +{ + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) + { + negative = token[i] == '-'; + i++; + } + uint64_t magnitude = 0; + bool over = false; + for ( ; i < length; i++ ) + { + uint64_t digit = (uint64_t) ( token[i] - '0' ); + if ( magnitude > ( UINT64_MAX - digit ) / 10 ) { over = true; break; } + magnitude = magnitude * 10 + digit; + } + if ( !is_signed ) + { + // -0 IS zero, and clamping it would report an event that did not + // happen; only a real negative magnitude is out of range here + if ( negative ) { *saturated = magnitude != 0; return 0; } + if ( over ) { *saturated = true; return (int64_t) UINT64_MAX; } + *saturated = false; + return (int64_t) magnitude; + } + if ( negative ) + { + if ( over || magnitude > ( uint64_t( 1 ) << 63 ) ) { *saturated = true; return INT64_MIN; } + *saturated = false; + if ( magnitude == ( uint64_t( 1 ) << 63 ) ) { return INT64_MIN; } + return -(int64_t) magnitude; + } + if ( over || magnitude > (uint64_t) INT64_MAX ) { *saturated = true; return INT64_MAX; } + *saturated = false; + return (int64_t) magnitude; +} + +// A number token into a wide kind's raw storage (docs/SPEC-TABLES.md §16.2). A +// 128-bit integer takes any token whose VALUE is integral; a fixed field any +// token whose value is EXACTLY representable in its Q I.F — a finer fraction +// is the wrong shape for the field, counted as a kind mismatch and never +// rounded, the rule SPEC.md §4.6 gives a fixed default. A magnitude past 128 +// bits saturates and counts as a clamp, as an int64 field saturates at +// INT64_MAX; the declared range clamps after it, on the RAW scale, as it does +// for every bounded scalar. +// +// The token is normalized to its digits with the decimal point after "point" +// of them. An integer part past 40 digits is above 2^128 whatever the digits +// are, and a value below 10^-40 is finer than 2^-127, the finest fraction any +// F can spell — so outside that band the answer is known without the +// arithmetic, and a token spelling 1e999999999 costs nothing to refuse. +inline bool TableJsonReadWide( TableJsonIn & in, const char * token, int32_t length, void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + int frac = f->frac_bits; + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { negative = token[i] == '-'; i++; } + const char * int_digits = token + i; + int32_t int_len = 0; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { int_len++; i++; } + const char * frac_digits = token + i; + int32_t frac_len = 0; + if ( i < length && token[i] == '.' ) + { + i++; + frac_digits = token + i; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { frac_len++; i++; } + } + int64_t exp = 0; + if ( i < length && ( token[i] == 'e' || token[i] == 'E' ) ) + { + i++; + bool exp_negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { exp_negative = token[i] == '-'; i++; } + while ( i < length && token[i] >= '0' && token[i] <= '9' ) + { + if ( exp < 100000 ) { exp = exp * 10 + ( token[i] - '0' ); } + i++; + } + if ( exp_negative ) { exp = -exp; } + } + // the digits, with the point after "point" of them; leading and trailing + // zeros stripped. digit( k ) reads the k-th of the int and frac runs. + int32_t start = 0, end = int_len + frac_len; + int64_t point = int_len + exp; + while ( start < end && ( start < int_len ? int_digits[start] : frac_digits[start - int_len] ) == '0' ) { start++; point--; } + while ( end > start && ( end - 1 < int_len ? int_digits[end - 1] : frac_digits[end - 1 - int_len] ) == '0' ) { end--; } + + TableJsonWide raw = { 0, 0 }; + bool saturated = false; + TableJsonWide signed_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) >> 1 }; + TableJsonWide signed_min = { 0, uint64_t( 1 ) << 63 }; + TableJsonWide unsigned_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) }; + if ( start == end ) + { + // zero, and -0 IS zero + } + else if ( point > 40 ) + { + saturated = true; + if ( !negative ) { raw = is_signed ? signed_max : unsigned_max; } + else if ( is_signed ) { raw = signed_min; } + } + else if ( point < -40 ) + { + in.report->kind_mismatch++; // finer than any F can spell + return true; + } + else + { + // the fraction FIRST, so an inexact value is the wrong shape whatever + // its magnitude: its digits, with the zeros a negative point puts in + // front, doubled F times; each doubling's carry is the next bit, and + // the value is exact iff nothing is left after the last one + char fd[kTableJsonMaxNumber + 48]; + int32_t fn = 0; + for ( int64_t z = point; z < 0; z++ ) { fd[fn++] = 0; } + for ( int32_t k = (int32_t) ( point > 0 ? point : 0 ) + start; k < end; k++ ) + { + fd[fn++] = (char) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ); + } + TableJsonWide fraction = { 0, 0 }; + for ( int b = 0; b < frac; b++ ) + { + int carry = 0; + for ( int32_t k = fn - 1; k >= 0; k-- ) + { + int d = fd[k] * 2 + carry; + fd[k] = (char) ( d % 10 ); + carry = d / 10; + } + fraction = TableJsonWideShl( fraction, 1 ); + fraction.lo |= (uint64_t) carry; + } + for ( int32_t k = 0; k < fn; k++ ) + { + if ( fd[k] != 0 ) + { + in.report->kind_mismatch++; + return true; + } + } + // then the whole part, saturating past 128 bits + TableJsonWide whole = { 0, 0 }; + for ( int64_t k = start; k < start + point && !saturated; k++ ) + { + uint32_t digit = k < end ? (uint32_t) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ) : 0; + if ( TableJsonWideMulAdd( &whole, 10, digit ) != 0 ) { saturated = true; } + } + if ( !saturated && frac > 0 && !TableJsonWideZero( TableJsonWideShr( whole, 128 - frac ) ) ) { saturated = true; } + if ( !saturated ) + { + raw = TableJsonWideShl( whole, frac ); + raw.lo |= fraction.lo; + raw.hi |= fraction.hi; + } + if ( is_signed ) + { + if ( !saturated && !negative && TableJsonWideNegative( raw ) ) { saturated = true; } + if ( !saturated && negative && TableJsonWideCompare( raw, signed_min, false ) > 0 ) { saturated = true; } + if ( saturated ) { raw = negative ? signed_min : signed_max; } + else if ( negative ) { raw = TableJsonWideNeg( raw ); } + } + else + { + if ( saturated ) { raw = unsigned_max; } + if ( negative && !TableJsonWideZero( raw ) ) { raw.lo = 0; raw.hi = 0; saturated = true; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->wide != NULL ) + { + TableJsonWide lo = { f->wide->lo[0], f->wide->lo[1] }; + TableJsonWide hi = { f->wide->hi[0], f->wide->hi[1] }; + if ( TableJsonWideCompare( raw, lo, is_signed ) < 0 ) { raw = lo; in.report->clamped++; } + else if ( TableJsonWideCompare( raw, hi, is_signed ) > 0 ) { raw = hi; in.report->clamped++; } + } + TableJsonWideStore( storage, f->elem_size, raw ); + return true; +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ); + +inline bool TableJsonSkipContainer( TableJsonIn & in, char close, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; // the opening bracket + bool first = true; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == close ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + if ( close == '}' ) + { + // the key is kept, because a skipped OBJECT may still be a + // pointer's: an `&node` opening it names a node the storage could + // not hold, and the numbering has to survive the drop (§16.7). + // Anywhere but first, the prefix is the reserved key out of place + // — in a pointered unit; a fixed unit skips the value whole. + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( key[0] == '&' && in.graph != NULL ) + { + if ( !first ) { in.report->malformed = true; in.bad = true; return false; } + if ( !TableJsonSkippedAmpersand( in, key, depth ) ) { return false; } + first = false; + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } + } + first = false; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return TableJsonSkipContainer( in, '}', depth ); + case '[': return TableJsonSkipContainer( in, ']', depth ); + case '"': return TableJsonScanString( in, NULL, 0, NULL ); + case 't': return TableJsonLiteral( in, "true" ); + case 'f': return TableJsonLiteral( in, "false" ); + case 'n': return TableJsonLiteral( in, "null" ); + case 0: in.bad = true; return false; + default: + { + // consumed, never converted: skipping needs no buffer, and this + // is the one walk a hostile text drives to the depth cap. It is + // the SAME production the value path scans, so an unknown key + // cannot smuggle past a number a named key would refuse. + bool integral = false; + if ( !TableJsonWalkNumber( in, &integral ) ) { in.bad = true; return false; } + return true; + } + } +} + +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration reads through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ); + +// place one scalar at one storage address +inline bool TableJsonReadScalar( TableJsonIn & in, void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; {} is None, and + // two keys is a text this walk will not guess at + const TableUnionInfo * arms = f->arms(); + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, 0 ); + if ( TableJsonPeek( in ) == '}' ) { in.pos++; return true; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t tag = 0; + for ( int64_t t = 1; t <= f->enum_max; t++ ) + { + if ( strcmp( f->enum_name( (uint64_t) t ), key ) == 0 ) { tag = t; break; } + } + if ( tag == 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + void * payload = (uint8_t *) storage + arms->arms[tag].offset; + const TableFieldInfo * arm = arms->arms[tag].field; + bool placed = true; + if ( arm != NULL ) + { + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2). A value of + // the wrong shape for that row is a KIND MISMATCH: the union + // reads None, the event is counted, and the enclosing object + // continues — the rule a FIELD's value lives under, one key + // down. A pointer arm's null is a null pointer, not a shape + // error, exactly as a pointer field's is (§16.7). + char got = TableJsonValueShape( in ); + if ( arm->kind == 17 && !arm->is_array && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + memset( payload, 0, (size_t) arms->arms[tag].size ); + } + else if ( got != TableJsonShape( arm ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( arm->kind == 17 && !arm->is_array ) + { + // A POINTER ARM'S VALUE IS THE POINTEE IN PLACE, or a + // node reference to one (§16.7) — the read a pointer + // FIELD takes, which is not the scalar walk + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadPointer( in, payload, arm, depth + 1 ) ) { return false; } + } + else + { + // SELECTION ZERO-ESTABLISHES THE ARM (SPEC §5): an arm + // takes no specified default, so zero is the establish + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadField( in, storage, arm, depth + 1 ) ) { return false; } + } + } + else if ( arms->arms[tag].table != NULL ) + { + if ( TableJsonValueShape( in ) != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else + { + arms->arms[tag].table->reset( payload ); + if ( !TableJsonReadTable( in, payload, arms->arms[tag].table, depth + 1 ) ) { return false; } + } + } + else + { + // A PAYLOAD-FREE ARM'S VALUE IS null (§2.6): the arm name + // selects it and there is nothing to place + if ( TableJsonValueShape( in ) != 'z' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( !TableJsonLiteral( in, "null" ) ) + { + return false; + } + } + if ( placed ) + { + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, (uint64_t) tag ); + } + } + char c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; // a second key: a one-of with two arms is not a value + return false; + } + if ( f->kind == 13 ) + { + f->table->reset( storage ); + return TableJsonReadTable( in, storage, f->table, depth + 1 ); + } + if ( TableJsonIsEnum( f ) ) + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + for ( int64_t v = 0; v <= f->enum_max; v++ ) + { + if ( strcmp( f->enum_name( (uint64_t) v ), name ) == 0 ) + { + TableJsonSetRaw( storage, f->elem_size, (uint64_t) v ); + return true; + } + } + // a name this build cannot name reads as None and counts as unknown, + // exactly as an unknown variant id does on the wire (§4) + TableJsonSetRaw( storage, f->elem_size, 0 ); + in.report->unknown++; + return true; + } + if ( TableJsonIsFlags( f ) ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + uint64_t bits = 0; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( c != '"' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + bool found = false; + for ( int64_t bit = 0; bit <= f->enum_max; bit++ ) + { + if ( strcmp( f->enum_name( (uint64_t) bit ), name ) == 0 ) + { + bits |= uint64_t( 1 ) << bit; + found = true; + break; + } + } + if ( !found ) { in.report->unknown++; } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + TableJsonSetRaw( storage, f->elem_size, bits ); + return true; + } + if ( f->kind == 1 ) + { + char c = TableJsonPeek( in ); + if ( c == 't' ) { if ( !TableJsonLiteral( in, "true" ) ) { return false; } TableJsonSetRaw( storage, f->elem_size, 1 ); return true; } + if ( !TableJsonLiteral( in, "false" ) ) { return false; } + TableJsonSetRaw( storage, f->elem_size, 0 ); + return true; + } + char token[kTableJsonMaxNumber]; + int32_t length = 0; + bool integral = false; + if ( !TableJsonScanNumber( in, token, kTableJsonMaxNumber, &length, &integral ) ) + { + in.bad = true; + return false; + } + if ( TableJsonKindWide( f->kind ) ) + { + return TableJsonReadWide( in, token, length, storage, f ); + } + if ( f->kind == 10 || f->kind == 11 ) + { + bool single = f->kind == 10; + double value = TableJsonTokenDouble( token, length, single ); + // A magnitude the field's format cannot hold is the WRONG SHAPE for + // the kind, and it never reaches storage: 1e400 is not a float64 and + // 1e300 is not a float32. Storing the infinity the conversion + // produced would leave an instance this walk called CLEAN that + // ToJsonMeasure then refuses forever (a non-finite float has no JSON + // spelling), and §16.1's one invariant is that a text which reads + // clean writes back. + if ( !TableJsonFinite( value ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( f->has_range ) + { + if ( value < f->range_min ) { value = f->range_min; in.report->clamped++; } + else if ( value > f->range_max ) { value = f->range_max; in.report->clamped++; } + } + if ( single ) + { + float narrow = (float) value; + if ( !TableJsonFinite( (double) narrow ) ) + { + in.report->kind_mismatch++; + return true; + } + memcpy( storage, &narrow, sizeof( narrow ) ); + } + else + { + memcpy( storage, &value, sizeof( value ) ); + } + return true; + } + // JSON HAS ONE NUMBER TYPE. 2.0 IS the integer 2 and 1e3 IS 1000, and a + // library that round-trips numbers through a double emits them that way — + // this walker's own float writer emits 1e+21. So an integer field takes + // any number whose VALUE is integral, however it was spelled; only a + // genuinely fractional value is the wrong shape for it. + bool is_signed = f->kind >= 2 && f->kind <= 5; + bool saturated = false; + int64_t value = 0; + if ( integral ) + { + value = TableJsonTokenInteger( token, length, is_signed, &saturated ); + } + else + { + double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( is_signed ) + { + if ( d >= 9223372036854775808.0 ) { value = INT64_MAX; saturated = true; } + else if ( d < -9223372036854775808.0 ) { value = INT64_MIN; saturated = true; } + else if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) d; } + } + else + { + if ( d < 0.0 ) + { + // a negative for an unsigned field clamps to zero, as the + // exact digit path already does + if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + value = 0; + saturated = true; + } + else if ( d >= 18446744073709551616.0 ) { value = (int64_t) UINT64_MAX; saturated = true; } + else if ( d != (double) (uint64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) (uint64_t) d; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->has_range ) + { + if ( (double) value < f->range_min ) { value = (int64_t) f->range_min; in.report->clamped++; } + else if ( (double) value > f->range_max ) { value = (int64_t) f->range_max; in.report->clamped++; } + } + // the field's own storage width is the last bound: a value past it + // clamps rather than wrapping, which is what the wire does too + if ( f->elem_size < 8 ) + { + if ( is_signed ) + { + int64_t high = ( int64_t( 1 ) << ( f->elem_size * 8 - 1 ) ) - 1; + int64_t low = -high - 1; + if ( value > high ) { value = high; in.report->clamped++; } + else if ( value < low ) { value = low; in.report->clamped++; } + } + else + { + uint64_t high = ( uint64_t( 1 ) << ( f->elem_size * 8 ) ) - 1; + if ( value < 0 ) { value = 0; in.report->clamped++; } + else if ( (uint64_t) value > high ) { value = (int64_t) high; in.report->clamped++; } + } + } + // at eight bytes the storage IS the parser's width, and an unsigned value + // past INT64_MAX rides here as a negative int64 by design — the token + // parser already turned a NEGATIVE token for an unsigned field into a + // clamped zero, so there is nothing left to bound. + TableJsonSetRaw( storage, f->elem_size, (uint64_t) value ); + return true; +} + +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ) +{ + uint8_t * storage = (uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonReadMap( in, (void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + + if ( f->kind == 12 ) + { + int32_t length = 0; + if ( !TableJsonScanString( in, (char *) storage, f->array_bound, &length ) ) { return false; } + storage[length] = 0; + TableJsonSetCount( base, f, length ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + // base64 decodes STRAIGHT INTO the field's storage, six bits at a + // time — no window, no temporary, so a bytes(N) of any declared + // extent reads the same way. A base64 body carries no escapes, so a + // backslash in one is simply not an alphabet character. + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + memset( storage, 0, (size_t) f->array_bound ); + TableJsonSetCount( base, f, 0 ); + const char * alphabet = TableJsonBase64Alphabet(); + int32_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + bool clamped = false; + bool malformed = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + const char * at = c != 0 ? strchr( alphabet, c ) : NULL; + if ( at == NULL ) { malformed = true; continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( at - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < f->array_bound ) + { + storage[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); + } + else + { + clamped = true; + } + } + } + if ( malformed ) + { + // a body that is not base64 is the wrong shape for the kind: the + // field keeps its default and the event is counted + in.report->kind_mismatch++; + return true; + } + if ( clamped ) { in.report->clamped++; } + TableJsonSetCount( base, f, placed ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + // every slot back to its declared defaults first, so a key the text + // omits keeps them and a repeated field key cannot leave an earlier + // occurrence's slots standing + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + void * slot = storage + (int64_t) i * f->elem_size; + if ( f->kind == 13 ) { f->table->reset( slot ); } + else { memset( slot, 0, (size_t) f->elem_size ); } + } + char shape = TableJsonElementShape( f ); + // A KEYED OBJECT'S KEYS ARE KEYS: a variant named twice is a duplicate + // key like any other, last-wins and counted (§16.2). Tracked the way + // a table's own field keys are — a bounded, allocation-free bitmask; + // a vocabulary wider than this still reads, its repeats simply stop + // being counted. + uint64_t seen[8] = {}; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t slot = -1; + for ( int64_t v = 0; v < f->array_bound; v++ ) + { + // nothing is stored for None, so "None" finds no slot and is + // an unknown key like any other name this reader cannot place + if ( !TableJsonKeyedSlotValid( f, v ) ) { continue; } + if ( strcmp( f->key_name( TableJsonKeyedSlotKey( v ) ), key ) == 0 ) { slot = v; break; } + } + if ( slot >= 0 && slot < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( slot & 63 ); + if ( ( seen[slot >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[slot >> 6] |= bit; + } + if ( slot < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; + } + if ( f->is_array ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + // LAST WINS has to be true of a repeated ARRAY key too, and it is + // wire-visible: a fixed array writes every slot, so a second, shorter + // occurrence overlaying a prefix would leave the first occurrence's + // tail standing. The field goes back to its declared defaults before + // this occurrence's elements are placed — the re-establishment a nested + // table and a union arm already get. A table element's defaults are + // its own (the reset hook); every other element kind's storage + // default is zero, which is what the generated array declares. + if ( f->kind == 13 ) + { + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + f->table->reset( storage + (int64_t) i * f->elem_size ); + } + } + else + { + memset( storage, 0, (size_t) f->array_bound * (size_t) f->elem_size ); + } + TableJsonSetCount( base, f, 0 ); + int32_t placed = 0; + char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( placed >= f->array_bound ) + { + // more elements than the reader's bound: the bounded prefix + // is kept and the excess counts, the wire's rule (§4) + in.report->clamped++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( f->kind == 17 ) + { + // an element of an ARRAY OF POINTERS (§2.1): null is a null slot, an + // object is the pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( storage + (int64_t) placed * f->elem_size, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed++; + } + else + { + if ( !TableJsonReadScalar( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + // a fixed array's tail keeps the defaults the prefill left there, + // exactly as a short wire count does + TableJsonSetCount( base, f, placed ); + return true; + } + return TableJsonReadScalar( in, storage, f, depth ); +} + +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ); + +// ONE table object: keys are field keys, unknown ones are skipped and +// counted, a repeated key is last-wins and counted. The instance is already +// at its declared defaults when this is entered, so a key the text never +// mentions keeps the default an absent field takes on the wire (§4). +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + return TableJsonReadTableKeys( in, base, info, depth, NULL ); +} + +// The keys of an object whose brace is already consumed. A pointer's object +// opens the same way a table's does, but its FIRST key may be `&node` (§16.7) +// and the adapter that reads it has to scan the key to know — so it hands the +// key it scanned in as `first_key`, with the colon consumed, and this places +// it before scanning the rest. +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ) +{ + // duplicate tracking, bounded and allocation-free: a table with more + // fields than this still reads, its repeats simply stop being counted + uint64_t seen[8] = {}; + for ( ;; ) + { + char key[kTableJsonMaxKey]; + char c = 0; + if ( first_key != NULL ) + { + memcpy( key, first_key, strlen( first_key ) + 1 ); // scanned into a buffer this size by the caller + first_key = NULL; + } + else + { + c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + } + int32_t index = -1; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + if ( strcmp( info->fields[i].json, key ) == 0 ) { index = i; break; } + } + if ( key[0] == '&' ) + { + // THE AMPERSAND PREFIX IS RESERVED TO THE FORM (docs/SPEC-TABLES.md + // §16.7). No declaration may take a key beginning with it, so this + // is never a field this build lacks — it is the sharing construct + // somewhere it cannot stand: `&node` is the FIRST key of a pointer's + // object and nothing else, and the adapter that reads a pointer + // has consumed it before these keys are read. MALFORMED, refused + // and counted; never counted as unknown, never skipped. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( index < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + const TableFieldInfo * f = &info->fields[index]; + if ( index < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( index & 63 ); + if ( ( seen[index >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[index >> 6] |= bit; + } + // PRESENCE OF THE KEY IS THE PRESENCE (§16.2): reaching this line + // is the key being present, so an optional is set present + // whatever its value — with one exception the page names: a JSON + // null, which reads as ABSENT rather than as a value. + char got = TableJsonValueShape( in ); + if ( f->kind == 17 && !f->is_array ) + { + // a pointer: null is a null pointer, an object is the pointee + // in place or an `&node` reference to one (§16.7), a string is + // a BYTE BUFFER's bytes (§2.5), and anything else is the wrong + // shape for the kind + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) base + f->offset, f->elem_size, 0 ); + } + else if ( got != TableJsonShape( f ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) base + f->offset, f, depth ) ) + { + return false; + } + } + else if ( f->optional && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + // absent, and back at its defaults: a repeated key whose last + // occurrence is null must not leave an earlier value standing + if ( f->table != NULL ) { f->table->reset( (uint8_t *) base + f->offset ); } + else { memset( (uint8_t *) base + f->offset, 0, (size_t) f->elem_size ); } + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 0 ); + } + else + { + if ( got != TableJsonShape( f ) ) + { + // the wrong JSON type for the kind: skipped, never coerced + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, base, f, depth ) ) + { + return false; + } + if ( f->optional ) + { + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 1 ); + } + } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +// ---- the two entry points the per-table wrappers name ---- + +inline bool TableJsonRead( void * value, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = NULL; + info->reset( value ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, value, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +inline int64_t TableJsonWrite( const void * value, const TableTypeInfo * info, char * buffer, int64_t capacity ) +{ + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = NULL; + if ( !TableJsonWriteValue( out, value, info, 0 ) ) { return -1; } + // THE CANONICAL TEXT ENDS WITH EXACTLY ONE NEWLINE (docs/SPEC-TABLES.md + // §16.1). Every writer emits it — this walk, the C# walk and + // "schema unpack" — and every reader accepts a text with or without one, + // because the trailing whitespace a read already skips is what makes the + // two the same text. It is a byte of the FORM rather than a file + // convention: a text that is written to a file, pasted into a diff and + // handed back through a pipe has to be one text in all three places, and a + // buffer whose last byte is a closing brace is the one shape that is not. + out.put( '\n' ); + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json walk: end ---- + +// ---- json graph walk: begin ---- +// +// THE VARIABLE CLASS's half of the text form (docs/SPEC-TABLES.md §16.7). The +// walk above places every kind but one; this defines the three adapters it +// calls for that one, and the two entry points a pointered table's wrappers +// name. The text is the fixed class's — a pointee is an object in place — and a +// node named more than once carries `&node`: defined once, with its fields, +// and referenced after by `{ "&node": N }` alone. + +// ---- the identity map ---- +// +// ONE map shape serves both directions. Writing keys it by a node's ADDRESS and +// counts the slots that name the node, so the second pass knows at a node's +// first occurrence whether it will be named again; reading keys it by the +// text's own label and answers the node it defined. Open addressing, a +// multiply-shift hash and quadrupling growth — TablePackMap's shape (§6.2), on +// the same terms: proportional to nodes, never to bytes, on the authoring +// side, and released before the call returns. + +struct TableJsonGraphEntry +{ + uint64_t key; // a node's address (write) or a label (read); 0 is an empty slot + int64_t count; // write: how many slots name this node + int64_t label; // write: the `&node` label assigned at its first write, 0 until then + uint8_t open; // the descent is still open: a reference here is a cycle (write), a self-reference (read) + uint32_t node; // read: the node's arena offset; 0 for a definition the reader dropped + const TableTypeInfo * type; // read: the node's table; NULL for a dropped one +}; + +struct TableJsonGraphMap +{ + TableJsonGraphEntry * entries; + int64_t capacity; // a power of two, or zero while empty + int64_t count; + TableAllocator allocator; // the caller's pair (§6.5): the builder's on read, the one handed to ToJson on write +}; + +inline void TableJsonGraphMapInit( TableJsonGraphMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TableJsonGraphMapShutdown( TableJsonGraphMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TableJsonGraphMapInit( map, map.allocator ); +} + +inline int64_t TableJsonGraphMapSlot( const TableJsonGraphMap & map, uint64_t key ) +{ + uint64_t hash = key * 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != 0 && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TableJsonGraphEntry * TableJsonGraphMapFind( TableJsonGraphMap & map, uint64_t key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +inline bool TableJsonGraphMapGrow( TableJsonGraphMap & map ) +{ + TableJsonGraphMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 64; + grown.count = 0; + grown.entries = (TableJsonGraphEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TableJsonGraphEntry ) ); // zeroed, by the pair's contract + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == 0 ) { continue; } + grown.entries[ TableJsonGraphMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// the entry for a key, made if it was not there; `taken` says which. NULL is the +// allocator refusing, and the walk refuses with it. +inline TableJsonGraphEntry * TableJsonGraphMapReach( TableJsonGraphMap & map, uint64_t key, bool & taken ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TableJsonGraphMapGrow( map ) ) { return NULL; } + } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + taken = entry->key != key; + if ( taken ) + { + entry->key = key; + map.count++; + } + return entry; +} + +// ---- reading: into a builder ---- + +struct TableJsonGraphIn +{ + TableWorker * worker; // where every node comes from + TableJsonGraphMap labels; // a label -> the node it defined +}; + +// `&node`'s value, the LABEL: a positive integer spelled as one — digits, no sign, no +// fraction, no exponent, no leading zero (§16.7). Anything else is malformed. +inline bool TableJsonScanLabel( TableJsonIn & in, uint64_t & label ) +{ + TableJsonSpace( in ); + if ( in.pos >= in.size || in.text[in.pos] < '1' || in.text[in.pos] > '9' ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + uint64_t value = 0; + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) + { + uint64_t digit = (uint64_t) ( in.text[in.pos] - '0' ); + if ( value > ( UINT64_MAX - digit ) / 10 ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + value = value * 10 + digit; + in.pos++; + } + label = value; + return true; +} + +// A BYTE BUFFER's text (docs/SPEC-TABLES.md §2.5, §16.2): a string. For a +// *string the string's bytes become the blob; for a *bytes the string is base64 +// and its decoded bytes do. The blob is allocated at EXACTLY the decoded +// length — the string is scanned once without keeping it to learn the length, +// and once into the node — so a blob of any size reads with no window and no +// bound to clamp against. A *bytes body that is not base64 is the wrong shape +// for the kind: the reference stays null and the event is counted. +inline bool TableJsonReadBlob( TableJsonIn & in, void * slot, const TableFieldInfo * f ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + TableRef * ref = (TableRef *) slot; + ref->value = 0; + if ( strcmp( f->type_name, "string" ) == 0 ) + { + const int64_t mark = in.pos; + int32_t length = 0; + if ( !TableJsonScanString( in, NULL, 0, &length ) ) { return false; } + in.pos = mark; + char * data = TableStringEmplace( *graph->worker, *ref, NULL, (int64_t) length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int32_t placed = 0; + return TableJsonScanString( in, data, length, &placed ); + } + // base64: the alphabet characters decide the length, six bits apiece + const char * alphabet = TableJsonBase64Alphabet(); + const int64_t mark = in.pos + 1; + int64_t symbols = 0; + bool malformed = false; + in.pos++; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + if ( c == 0 || strchr( alphabet, c ) == NULL ) { malformed = true; continue; } + symbols++; + } + if ( malformed ) + { + in.report->kind_mismatch++; + return true; + } + const int64_t length = ( symbols * 6 ) / 8; + uint8_t * data = TableBytesEmplace( *graph->worker, *ref, length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int64_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + for ( int64_t at = mark; ; at++ ) + { + char c = in.text[at]; + if ( c == '"' ) { break; } + const char * symbol = c != '=' ? strchr( alphabet, c ) : NULL; + if ( symbol == NULL ) { continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( symbol - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < length ) { data[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); } + } + } + return true; +} + +// A pointer's object. Its FIRST key decides what it is: `&node` naming a label not +// yet defined, with fields after it, is a DEFINITION; `&node` naming one already +// defined, alone, is a REFERENCE; any other key is a node named once, its +// object in place. The node comes from the +// builder's arena, and the slot holds its arena offset (§6.3). A pointer whose +// target is a BYTE BUFFER — no table — takes a string instead (§2.5). +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( f->table == NULL ) { return TableJsonReadBlob( in, slot, f ); } + // the pointee nests one level down, exactly as a by-value table does, and + // takes the same cap: a chain nests as deep as it is long (§16.7) + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + char c = TableJsonPeek( in ); + if ( c == '}' ) + { + // an empty object: a node at its defaults, named once + in.pos++; + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + return true; + } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( strcmp( key, "&node" ) != 0 ) + { + // a node named once: the pointee's object in place, and this key is + // its first field — unless it is the reserved prefix under a spelling + // this form does not have, which ReadTableKeys refuses + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return TableJsonReadTableKeys( in, node, f->table, depth + 1, key ); + } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->labels, label, taken ); + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + // ONE SPELLING, and what follows the label says which half it is: fields + // after a label the text has not defined DEFINE it, and a label alone that + // the text has defined REFERS to it. The other two are malformed — a label + // alone that the text never defined, which would otherwise read as a default + // node under a silent report, and a field after a label already defined, + // which would be a second definition. That is what keeps a typo loud. + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + bool bare = c == '}'; + if ( bare == taken ) { in.report->malformed = true; in.bad = true; return false; } + if ( bare ) + { + // A REFERENCE. A label is defined when its object CLOSES, so a + // reference met inside its own definition — at any depth of by-value + // nesting — names a node whose descent is still open: the cycle the + // wire refuses (§3.1), refused here where it is written. A definition + // the reader dropped names no node, so the slot stays null with + // nothing more counted — the drop was counted where it happened. A + // node of another table than the slot declares is a kind mismatch, as + // on the wire. + in.pos++; + if ( entry->open != 0 ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + if ( entry->type == NULL ) + { + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + if ( entry->type != f->table ) + { + memcpy( slot, &ref, sizeof( ref ) ); + in.report->kind_mismatch++; + return true; + } + ref.value = (int64_t) entry->node; + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + // A DEFINITION: the node is allocated, the label is its, and the keys after + // `&node` are its fields. The entry is OPEN until the object closes, so a + // reference to the label from inside the node's own fields is refused as + // the cycle it is; the node and its table are filled in at the close. + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + entry->open = 1; + if ( !TableJsonReadTableKeys( in, node, f->table, depth + 1, NULL ) ) { return false; } + entry = TableJsonGraphMapFind( graph->labels, label ); // the map may have grown under the descent + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + memcpy( &ref, slot, sizeof( ref ) ); + entry->node = (uint32_t) ref.value; + entry->type = f->table; + entry->open = 0; + return true; +} + +// An `&`-prefixed key opening an object the walk is SKIPPING — a value past an +// array's bound, an unknown key's value, a value of the wrong shape. A +// definition in there still takes its label, so the numbering survives whatever +// the storage could not hold (§16.7): the label is registered with no node, and a +// reference to it reads null. Any other prefixed key is the reserved prefix +// out of place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL || strcmp( key, "&node" ) != 0 ) { in.report->malformed = true; in.bad = true; return false; } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + if ( TableJsonGraphMapReach( graph->labels, label, taken ) == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return true; // a fresh entry is node 0, type NULL: a definition with no node +} + +// ---- writing: from a region's const root ---- + +struct TableJsonGraphOut +{ + TableJsonGraphMap nodes; // a node's address -> how many slots name it, and its `&node` once assigned + bool counting; // PASS ONE: count the references, refuse a cycle, emit nothing + int64_t next_label; +}; + +// The node a slot names: null as `null`, a node named once as its object in +// place, and a node named more than once under the construct. Which of the +// last two it is was learned in pass one; pass two spells it. +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphOut * graph = (TableJsonGraphOut *) out.graph; + if ( graph == NULL ) { return false; } + const void * node = f->resolve( slot ); + if ( node == NULL ) + { + out.raw( "null", 4 ); + return true; + } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->nodes, (uint64_t) (uintptr_t) node, taken ); + if ( entry == NULL ) { return false; } + if ( f->table == NULL ) + { + // A BYTE BUFFER (§2.5, §16.7): its text is a string, which has no + // first key to carry `&node`, so a blob named from more than one + // slot has no spelling this form can carry and the graph is refused — + // as a shared node with nothing to write is. A blob named once is its + // bytes in place: base64 for a *bytes, the string itself for a *string. + if ( graph->counting ) { entry->count++; return true; } + if ( entry->count > 1 ) { return false; } + const TableBlob * blob = (const TableBlob *) node; + if ( blob->length > (uint32_t) 0x7fffffff ) { return false; } + if ( strcmp( f->type_name, "string" ) == 0 ) { TableJsonWriteString( out, (const char *) ( blob + 1 ), (int32_t) blob->length ); } + else { TableJsonWriteBase64( out, (const uint8_t *) ( blob + 1 ), (int32_t) blob->length ); } + return true; + } + if ( graph->counting ) + { + // PASS ONE: one visit per node, every slot that names it counted, and + // a reference to a node whose descent is still open is a cycle — + // refused here as the wire refuses it (§3.1) + entry->count++; + if ( !taken ) { return entry->open == 0; } + entry->open = 1; + if ( !TableJsonWriteValue( out, node, f->table, depth ) ) { return false; } + entry = TableJsonGraphMapFind( graph->nodes, (uint64_t) (uintptr_t) node ); // the map may have grown under the descent + if ( entry == NULL ) { return false; } + entry->open = 0; + return true; + } + // PASS TWO: a node named once is its object in place; a node named more + // than once is DEFINED at its first occurrence — `&node` first, then its + // fields — and REFERENCED by `&node` alone after that, spelled the same way at + // every site. Labels run from 1 in first-write order and are the text's own, + // so a stray number in a hand-edited text is most often one never defined. + if ( entry->count <= 1 ) + { + return TableJsonWriteValue( out, node, f->table, depth ); + } + if ( depth > kTableJsonMaxDepth ) { return false; } + if ( entry->label != 0 ) + { + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + out.line( depth ); + out.put( '}' ); + return true; + } + entry->label = ++graph->next_label; + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + bool any = true; + int64_t before = out.offset; + if ( !TableJsonWriteFields( out, node, f->table, depth, any ) ) { return false; } + // a definition carries at least one field, because a label alone is a + // reference: a shared node with nothing to write has no definition this + // form can spell, and the writer refuses it as it refuses any value it + // cannot spell (§16.3) + if ( out.offset == before ) { return false; } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- the two entry points a pointered table's wrappers name ---- + +// The text into the builder's root. Every node the text names is allocated in +// the builder's arena through the field's own Emplace; the label map is the +// walk's, released before this returns. The root itself takes no label — nothing +// may name it (§16.7) — so an `&node` at the root is refused like any other key +// of the prefix. +inline bool TableJsonReadGraph( TableWorker & worker, void * root, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + if ( worker.arena == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } + TableJsonGraphIn graph; + graph.worker = &worker; + TableJsonGraphMapInit( graph.labels, worker.arena->allocator ); + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = &graph; + info->reset( root ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, root, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + TableJsonGraphMapShutdown( graph.labels ); + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +// The text of a region's const root: measured when the buffer is NULL, written +// when it is not, over one code path. Two passes over one walk — the first +// counts how many slots name each node and refuses a cycle, the second writes +// — so a node's first occurrence knows whether it will be named again. The +// ROOT's entry is open for the whole first pass, so a reference back at it is +// the cycle it is (§3.1), and it takes no label. +inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * info, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + if ( root == NULL ) { return -1; } + TableJsonGraphOut graph; + TableJsonGraphMapInit( graph.nodes, allocator ); + graph.counting = true; + graph.next_label = 0; + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph.nodes, (uint64_t) (uintptr_t) root, taken ); + if ( entry == NULL ) { TableJsonGraphMapShutdown( graph.nodes ); return -1; } + entry->open = 1; + TableJsonOut count; + count.buffer = NULL; + count.capacity = 0; + count.offset = 0; + count.overflow = false; + count.graph = &graph; + bool ok = TableJsonWriteValue( count, root, info, 0 ); + graph.counting = false; + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = &graph; + if ( ok ) { ok = TableJsonWriteValue( out, root, info, 0 ); } + TableJsonGraphMapShutdown( graph.nodes ); + if ( !ok ) { return -1; } + out.put( '\n' ); // the canonical text ends with exactly one newline (§16.1) + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json graph walk: end ---- + +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + +// ---- json map walk: begin ---- + +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} + +// the entry's two rows: fields[0] IS the key and fields[1] IS the value, which +// is what makes a user's own table of pairs the same bytes (§2.8) +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } + +inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } +inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } + +// AN INTEGER KEY IS THE INTEGER'S DECIMAL SPELLING, QUOTED, because a JSON +// object's keys are strings. Written digit by digit so no locale can move it. +inline void TableJsonWriteMapIntegerKey( TableJsonOut & out, const void * storage, const TableFieldInfo * key ) +{ + uint64_t magnitude = 0; + bool negative = false; + if ( TableJsonMapKeySigned( key ) ) + { + int64_t value = 0; + switch ( key->kind ) + { + case 2: value = (int64_t) *(const int8_t *) storage; break; + case 3: value = (int64_t) *(const int16_t *) storage; break; + case 4: value = (int64_t) *(const int32_t *) storage; break; + default: value = *(const int64_t *) storage; break; + } + negative = value < 0; + magnitude = negative ? ( ~(uint64_t) value ) + 1 : (uint64_t) value; + } + else + { + switch ( key->kind ) + { + case 6: magnitude = (uint64_t) *(const uint8_t *) storage; break; + case 7: magnitude = (uint64_t) *(const uint16_t *) storage; break; + case 8: magnitude = (uint64_t) *(const uint32_t *) storage; break; + default: magnitude = *(const uint64_t *) storage; break; + } + } + char digits[24]; + int32_t at = (int32_t) sizeof( digits ); + do { digits[--at] = (char) ( '0' + ( magnitude % 10 ) ); magnitude /= 10; } while ( magnitude != 0 ); + if ( negative ) { digits[--at] = '-'; } + TableJsonWriteString( out, digits + at, (int32_t) sizeof( digits ) - at ); +} + +inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const TableFieldInfo * key ) +{ + const uint8_t * storage = (const uint8_t *) entry + key->offset; + if ( TableJsonMapKeyIsString( key ) ) + { + // A STRING KEY IS THE STRING (§2.8): every JSON key of a map object is + // a KEY OF THE MAP and none is a field key, so the `&` prefix §16.7 + // reserves for field keys is ordinary data here. + TableJsonWriteString( out, (const char *) storage, *(const int32_t *) ( (const uint8_t *) entry + key->count_offset ) ); + return; + } + TableJsonWriteMapIntegerKey( out, (const void *) storage, key ); +} + +// ToJson WRITES ENTRIES IN ASCENDING KEY ORDER, so unpack then pack is +// byte-stable and a diff of two texts is a diff of two maps (§2.8, §17.2). +// A region holds them in that order already, so this is the array in place. +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "{}", 2 ); return true; } + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); + out.put( '{' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); + TableJsonWriteMapKey( out, entry, key ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// AN INTEGER KEY IS READ BY §16.2's INTEGER RULE AND BY NOTHING ELSE, so +// "2.0" and "1e3" are the integers 2 and 1000 and "-0" is zero. The token is +// walked as a JSON number over its own bytes; a token that rule calls +// malformed makes the KEY malformed, and a genuinely fractional value, or one +// outside the key kind's range, is kind_mismatch for that entry. +inline bool TableJsonMapKeyValue( const char * token, int32_t length, const TableFieldInfo * key, + int64_t & value, bool & fits ) +{ + fits = false; + TableReport scratch; + TableJsonIn probe = { token, (int64_t) length, 0, &scratch, false, NULL }; + bool integral = false; + if ( !TableJsonWalkNumber( probe, &integral ) ) { return false; } + if ( probe.pos != (int64_t) length ) { return false; } // trailing bytes: not a number + if ( !integral ) + { + const double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) { return true; } // a value no key kind holds + const double whole = d < 0 ? -d : d; + if ( whole != (double) (int64_t) whole ) { return true; } // genuinely fractional + } + bool saturated = false; + const bool is_signed = TableJsonMapKeySigned( key ); + value = integral ? TableJsonTokenInteger( token, length, is_signed, &saturated ) + : (int64_t) TableJsonTokenDouble( token, length, false ); + if ( saturated ) { return true; } // outside every width: kind_mismatch, never clamped + switch ( key->kind ) + { + case 2: fits = value >= -128 && value <= 127; break; + case 3: fits = value >= -32768 && value <= 32767; break; + case 4: fits = value >= -2147483647 - 1 && value <= 2147483647; break; + case 5: fits = true; break; + case 6: fits = value >= 0 && value <= 255; break; + case 7: fits = value >= 0 && value <= 65535; break; + case 8: fits = value >= 0 && (uint64_t) value <= 4294967295ull; break; + default: fits = integral; break; // uint64: the token's own magnitude + } + return true; +} + +// FromJson READS KEYS IN WHATEVER ORDER THE TEXT GIVES THEM. A repeated key is +// last-wins and counted duplicate, the object rule (§16.2) applied inside the +// map. An empty object is an empty map, and null is kind_mismatch. +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const char shape = TableJsonShape( value ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char token[kTableJsonMaxKey]; + int32_t token_length = 0; + if ( !TableJsonScanString( in, token, kTableJsonMaxKey - 1, &token_length ) ) { return false; } + token[token_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t key_value = 0; + bool place = true; + if ( !TableJsonMapKeyIsString( key ) ) + { + bool fits = false; + if ( !TableJsonMapKeyValue( token, token_length, key, key_value, fits ) ) + { + // A MALFORMED KEY STOPS THE READ where §16.1's rule stops it, + // with the instance holding what was placed before the stop. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( !fits ) { in.report->kind_mismatch++; place = false; } + } + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; + if ( place && entry == NULL ) + { + // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the + // wire's rule, because a clamped key is a merged entry (§2.8). + in.report->clamped++; + } + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) + { + in.report->duplicate++; // last-wins, the object rule inside the map + } + const char got = TableJsonValueShape( in ); + if ( entry == NULL ) + { + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( value->kind == 17 && !value->is_array ) + { + // A POINTER VALUE IS SHARED EXACTLY AS A POINTER FIELD IS (§2.8): + // null is a null slot, an object is the pointee in place or an + // &node reference to one (§16.7), anything else is the wrong shape — + // the same three the field-key loop gives a pointer field, because + // an entry's value IS a field line. + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) entry + value->offset, value->elem_size, 0 ); + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) entry + value->offset, value, depth + 1 ) ) + { + return false; + } + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, entry, value, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json map walk: end ---- + +// ---- json list walk: begin ---- + +// an unbounded array is the out-of-line array that is not a map (§8.1) +inline bool TableJsonIsList( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && !TableJsonIsMap( f ); +} + +// ToJson WRITES THE ELEMENTS IN INDEX ORDER, which is the only order there is, +// so unpack then pack is byte-stable without a rule of its own (§2.9, §17.2). +// A region holds the array in place, so this steps it at the descriptor's pitch. +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + const uint8_t * elements = TableJsonExtentElements( slot ); + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const uint8_t * element = elements + (int64_t) i * f->elem_size; + if ( f->kind == 17 ) + { + // a []*T's elements take the pointer row (§16.7): the pointee's + // object in place, null, or `&node` for a shared one + if ( !TableJsonWritePointer( out, element, f, depth + 1 ) ) { return false; } + } + else if ( !TableJsonWriteScalar( out, element, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; +} + +// FromJson READS EVERY ELEMENT THE TEXT CARRIES, appending each through the +// descriptor's place resolver: `[]` is an empty list, and null is +// kind_mismatch, the array row's own rule (§16.2). LAST WINS holds for a +// repeated key: the list goes back to EMPTY before this occurrence's elements +// land, the builder's storage being reclaimed at reset (§2.9). +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) slot, 8, 0 ); + TableJsonSetRaw( (uint8_t *) slot + 8, 4, 0 ); + const char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + void * element = f->place( *graph->worker, slot, NULL, 0, 0 ); + if ( element == NULL ) + { + // NOT ADDED: the arena could not carve another segment, or the + // count met the int32 cap. The text cannot be placed whole, and + // the read stops where §16.1's rule stops it. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( f->kind == 17 ) + { + // an element of a []*T (§2.9): null is a null slot, an object is the + // pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) element, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, element, f, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + // the wrong shape for the element kind: the slot keeps its + // defaults and the event counts, the array row's rule (§16.2) + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, element, f, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json list walk: end ---- + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +bool PlacementFromJson( Placement & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, PlacementTableType(), text, bytes, report ); +} + +int64_t PlacementToJsonMeasure( const Placement & value ) +{ + return TableJsonWrite( &value, PlacementTableType(), NULL, 0 ); +} + +int64_t PlacementToJson( const Placement & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, PlacementTableType(), buffer, capacity ); +} + +bool LogEntryFromJson( LogEntry & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, LogEntryTableType(), text, bytes, report ); +} + +int64_t LogEntryToJsonMeasure( const LogEntry & value ) +{ + return TableJsonWrite( &value, LogEntryTableType(), NULL, 0 ); +} + +int64_t LogEntryToJson( const LogEntry & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, LogEntryTableType(), buffer, capacity ); +} + +bool SaveFromJson( SaveBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Save * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, SaveTableType(), text, bytes, report ); +} + +int64_t SaveToJsonMeasure( const Save * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, SaveTableType(), NULL, 0, allocator ); +} + +int64_t SaveToJson( const Save * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, SaveTableType(), buffer, capacity, allocator ); +} + +bool PointFromJson( Point & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, PointTableType(), text, bytes, report ); +} + +int64_t PointToJsonMeasure( const Point & value ) +{ + return TableJsonWrite( &value, PointTableType(), NULL, 0 ); +} + +int64_t PointToJson( const Point & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, PointTableType(), buffer, capacity ); +} + +bool MixedFromJson( MixedBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Mixed * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, MixedTableType(), text, bytes, report ); +} + +int64_t MixedToJsonMeasure( const Mixed * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, MixedTableType(), NULL, 0, allocator ); +} + +int64_t MixedToJson( const Mixed * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, MixedTableType(), buffer, capacity, allocator ); +} + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/SaveTable.h b/testdata/golden/tables/lists/SaveTable.h new file mode 100644 index 000000000..3566e3b49 --- /dev/null +++ b/testdata/golden/tables/lists/SaveTable.h @@ -0,0 +1,8534 @@ +// Code generated by the schema compiler from Save.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — protocol id 0xa5fbe602c119cdd9 (packets only: tables version by field id, not by protocol id) +// The TABLE wire (evolution-tolerant, docs/SPEC-TABLES.md): no serialize +// dependency — includable from any TU. + +#pragma once + +#include +#include // the prefill's scalar-array fills +#include // offsetof, for the reflection descriptors + +// ---- the hooks (docs/USAGE.md, "the C++ table runtime's hooks") ---- +// +// schema_assert — the runtime's own assert, and the refusal a debugger reads. +// NDEBUG removes it, exactly as it removes assert. A caller who already routes +// serialize's asserts writes `#define schema_assert serialize_assert` before +// including this header and both halves land in one handler. +#ifndef schema_assert +#include +#define schema_assert assert +#endif // #ifndef schema_assert + +// schema_fatal — what stands after the assert on a path that cannot continue. +// NDEBUG does not remove it. Supply it and is never included. +#ifndef schema_fatal +#include // abort +#define schema_fatal abort +#endif // #ifndef schema_fatal + +// schema_allocate / schema_release — what "no allocator handed in" means for +// this program. schema_allocate hands back ZEROED bytes and NULL on failure: +// an arena segment is copied whole, padding included, so anything left +// uninitialized here would reach a packed region. Supply both and +// is never included; hand a TableAllocator to a builder to route one +// structure's allocations somewhere else again. +#ifndef schema_allocate +#include // calloc, free +#define schema_allocate( bytes ) calloc( (size_t) 1, (size_t) ( bytes ) ) +#define schema_release( pointer ) free( pointer ) +#endif // #ifndef schema_allocate +#include // a node's lifetime starts in arena storage (placement new) +#include // one atomic per slab: the arena is lock-free by ownership + +#include "Save.h" + +#ifndef LISTDEMO_SCHEMA_TABLE_PRIMITIVES +#define LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +// THE CODEC DOES NOT DEPEND ON THE COMPILER'S INLINING BUDGET. A table of a +// realistic field count emits one large body per type, and the cursor a body +// writes through lives in the caller's `TableWriter`: across a call boundary +// that cursor round-trips through memory, and a `uint8_t *` store may alias the +// writer itself, so every put reloads it. When a budget runs out mid-body the +// codec silently degrades to that shape. Forcing the primitives and the +// fixed-class bodies inline is what keeps the cursor in registers and lets +// adjacent constant framing bytes merge into one store. +#if defined( _MSC_VER ) +#define LISTDEMO_TABLE_INLINE __forceinline +#elif defined( __GNUC__ ) || defined( __clang__ ) +#define LISTDEMO_TABLE_INLINE inline __attribute__(( always_inline )) +#else +#define LISTDEMO_TABLE_INLINE inline +#endif + +namespace listdemo { + +// WHY A READ WAS REFUSED, by name (docs/SPEC-TABLES.md §3.3, §11). A REFUSAL +// is not one of §4's events: nothing is decoded, no counter moves and no +// damage is reported, so five zero counters and a false flag are what a clean +// read prints too and only the verdict tells them apart. The reason says which +// refusal it was. +// +// This is the MESSAGE PATH's vocabulary and not the cooked form's (§7.4): a +// caller meeting one of these has been refused a MESSAGE on a connection, +// which is a different recovery with a different owner than a file a header +// match turned down. +enum TableMessageReason +{ + newer_form, // a FORM BYTE this reader does not carry (§3) + no_vocabulary, // no table for this connection: the message arrived before the announcement, or after a refused one + second_announcement, // a second announcement on a connection: it sets nothing, amends nothing, and the connection closes + vocabulary_too_large, // an announcement above the receiver's declared bound, refused before an entry is touched + message_form_as_file // a form 2 wire where a FILE was expected: its table is somewhere else +}; + +// The table-wire read report — the permissive contract's ledger. Silence +// (all zero) means the data matched this reader's schema exactly. +struct TableReport +{ + int32_t unknown = 0; // unknown field ids skipped (newer data) + int32_t kind_mismatch = 0; // known id, changed type — skipped, never misdecoded + int32_t clamped = 0; // out-of-range values clamped to declared bounds + // a key the TEXT form saw twice: last wins, and the repeat is counted + // (docs/SPEC-TABLES.md §16.2). The wire never raises it — a body carrying an + // id twice is legal input whose last occurrence wins, silently (§3). + int32_t duplicate = 0; + bool malformed = false; // framing damage; decode stopped, partial result kept + // THE REFUSAL VERDICT, which is not one of §4's events and moves no counter + // (docs/SPEC-TABLES.md §3): a FORM BYTE this reader does not carry. Five + // zero counters and a false flag are what a clean read prints too, so the + // verdict is what tells the two apart. + bool refused = false; + // WHICH refusal, and it is read only when refused is set: a read that + // was not refused has no reason, and this member is the one the caller + // must not look at then (docs/SPEC-TABLES.md §3.3). + TableMessageReason reason = newer_form; +}; + + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; +// ---- reflection (tables only, docs/SPEC-TABLES.md) ---- +// +// Static field descriptors for every type in the table closure: name, wire +// id/kind, storage offset, bounds, ranges, enum names and branch guards — +// enough to walk, print, diff, edit or bind any table value at runtime with +// no RTTI and no schema files. TableType() returns X's descriptor. + +struct TableTypeInfo; + +// One arm of a union field: where its payload sits inside the union's storage +// and what its payload looks like. The arm's NAME and its table-wire id come +// from the field's enum_name/variant_id functions at the same tag, so nothing +// is spelled twice (docs/SPEC-TABLES.md §8). +struct TableFieldInfo; + +struct TableUnionArmInfo +{ + uint32_t offset; // offsetof the arm's payload within the union storage + const TableTypeInfo * table; // the arm payload's descriptor, or NULL + // AN ARM IS A FIELD LINE (docs/SPEC-TABLES.md §2.6): an arm that names no + // declared type or table carries the FIELD descriptor a field of that + // type would carry instead — offsets taken within the union storage — so + // a generic walk meets an arm's kind, width, bounds and companions where + // it meets a field's. Exactly one of the two is non-NULL on a set arm. + const TableFieldInfo * field; + uint32_t size; // the arm's whole storage, which selection zero-establishes +}; + +// A union field's shape: the tag, and the arms indexed by it. Arms run +// [0, enum_max]; index 0 is the EMPTY arm and carries no payload. +struct TableUnionInfo +{ + uint32_t tag_offset; // offsetof the tag within the union storage + uint32_t tag_size; // sizeof the tag + const TableUnionArmInfo * arms; +}; + +// The exact raw range of a wide-kind field (docs/SPEC-TABLES.md §8.2): two 128-bit +// values as 64-bit lanes, low lane first, two's complement for the signed kinds. +struct TableWideRange +{ + uint64_t lo[2]; + uint64_t hi[2]; +}; + +// the arena's allocation front, defined with the variable-length runtime +// below; a descriptor names it only through a pointer parameter. +struct TableWorker; + +struct TableFieldInfo +{ + const char * name; // schema field name, e.g. "health" + const char * json; // the TEXT form's key: the json = "key" attribute, else name (§16.3) + const char * type_name; // schema type name, e.g. "float32", "Grade" + uint64_t id; // table-wire field id: fnv1a64 of the name, of the was alias after a rename (§5) + uint8_t kind; // table-wire kind; for arrays/strings/bytes, the ELEMENT kind + bool is_array; // fixed or counted array (bytes included) + bool is_pointer; // a *T pointer field: storage is an 8-byte TableRef; the target is a table + // THE TWO THE TEXT FORM NEEDS (docs/SPEC-TABLES.md §16.7), and they + // are here for the same reason is_pointer is: the walk is ONE walk + // over descriptors and cannot spell a target's own At or + // Emplace. `resolve` reads a slot in a REGION and answers the + // node it names, or NULL; `emplace` allocates one in a BUILDER's + // arena and points the slot at it. NULL on every field that is not + // a pointer, and emitted only in a unit that declares one. + const void * (*resolve)( const void * slot ); + void * (*emplace)( TableWorker & worker, void * slot ); + bool counted; // a _count/_length int32 companion exists (counted arrays, strings, bytes) + bool optional; // a ?T field: a _present bool companion decides whether it rides + int32_t array_bound; // array capacity / string max length; 0 for plain scalars + uint32_t offset; // offsetof the storage member + uint32_t elem_size; // sizeof the member (element size for arrays) + uint32_t count_offset; // offsetof the _count/_length companion, or 0xffffffff + uint32_t present_offset; // offsetof the _present companion, or 0xffffffff + const TableTypeInfo * table; // nested table's descriptor, or NULL + bool has_range; // a declared [min, max] (int or float) + double range_min; // NOTE: int64 ranges beyond 2^53 lose precision here + double range_max; + // the WIDE kinds (18-29, docs/SPEC-TABLES.md §3, §8.2): frac_bits is a fixed + // field's F — its storage holds units × 2^F — and wide is the declared + // range on that RAW scale, exact, as two 128-bit two's-complement values + // in 64-bit lanes (low lane first). NULL where the declaration bounds + // nothing (a bare uint128) and for every other kind; frac_bits is 0 for + // every kind that is not fixed-point. range_min/range_max still carry + // the declared bounds as doubles — whole units for a fixed field — for + // a walker that only shows them. + uint8_t frac_bits; + const TableWideRange * wide; + int64_t enum_max; // enums: highest valid value (None = 0 always valid); + // unions: the arm count (tag range [0, enum_max]); + // flags: the highest declared BIT INDEX; else -1 + // the vocabulary's names, indexed the same way enum_max bounds: an enum's + // value -> name, a union's tag -> arm name, a FLAGS field's bit index -> + // variant name. NULL for every other kind. + const char * (*enum_name)( uint64_t value ); + // the TABLE-WIRE id of one variant (docs/SPEC-TABLES.md §5): for an enum, the + // hash of the variant's name; for a union, the hash of the arm's name. + // 0 is the reserved id — an enum's None, a union's empty. NULL for every + // other kind — a FLAGS field's variants have no per-variant wire id (§4), + // so a NULL here beside a non-NULL enum_name is what says "flags". + // Walk [0, enum_max] to enumerate a vocabulary and its ids. + uint64_t (*variant_id)( uint64_t value ); + // an ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): the array has one slot per + // variant of key_type_name, indexed by the variant's value, and its slots + // ride under variant ids rather than positions. key_name and key_id are + // the key's vocabulary — walk [0, array_bound) to print slots by name. + // NULL on every other field. + const char * key_type_name; + const char * (*key_name)( uint64_t value ); + uint64_t (*key_id)( uint64_t value ); + // union fields: the tag and its arms, behind a function so the whole + // descriptor stays CONSTANT-INITIALISED (a captureless lambda converts to + // a function pointer at compile time; the arms themselves are a static + // inside it). NULL for every other kind. + const TableUnionInfo * (*arms)(); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded +}; + +struct TableTypeInfo +{ + const char * name; // schema type name + uint32_t size; // sizeof the storage struct + int32_t num_fields; + const TableFieldInfo * fields; + // put one instance back at its declared defaults, in place. A generic + // walker that fills a value has to be able to establish the defaults an + // absent field takes, and it holds no type to spell — this is the one + // thing the descriptors could not express without it. Placement-new + // value-init, exactly what the wire's read path does, and no temporary. + void (*reset)( void * storage ); + // the DERIVED mode (docs/SPEC-TABLES.md): false = fixed-size, a plain + // relocatable struct; true = variable-length, built through a Builder + // and read through a region root. Nobody declares it; the compiler + // works it out. + bool variable; +}; + +struct TableWriter +{ + uint8_t * buffer; + int64_t capacity; + int64_t offset = 0; + bool overflow = false; + + // the parameters do not repeat the member names: a parameter that hides a + // member is a warning the estate's compilers disagree about (gcc's + // -Wshadow and cl's C4458 refuse it, clang's -Wshadow does not), and this + // is a header a consumer compiles under its OWN flags + TableWriter( uint8_t * to_buffer, int64_t to_capacity ) : buffer( to_buffer ), capacity( to_capacity ) {} + + LISTDEMO_TABLE_INLINE void raw( const void * data, int64_t bytes ) + { + if ( offset + bytes > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) bytes ); + offset += bytes; + } + LISTDEMO_TABLE_INLINE void put8( uint8_t v ) { raw( &v, 1 ); } + LISTDEMO_TABLE_INLINE void put16( uint16_t v ) { uint8_t b[2] = { uint8_t( v ), uint8_t( v >> 8 ) }; raw( b, 2 ); } + LISTDEMO_TABLE_INLINE void put32( uint32_t v ) { uint8_t b[4] = { uint8_t( v ), uint8_t( v >> 8 ), uint8_t( v >> 16 ), uint8_t( v >> 24 ) }; raw( b, 4 ); } + LISTDEMO_TABLE_INLINE void put64( uint64_t v ) { put32( uint32_t( v ) ); put32( uint32_t( v >> 32 ) ); } + // a 128-bit value as two lanes, the low half first (docs/SPEC-TABLES.md §3) + LISTDEMO_TABLE_INLINE void put128( uint64_t lo, uint64_t hi ) { put64( lo ); put64( hi ); } + // EVERY LENGTH, COUNT, INDEX AND ID REFERENCE IS ONE CANONICAL UNSIGNED + // LEB128 (docs/SPEC-TABLES.md §3): seven value bits a byte, the lowest + // group first, the high bit set on every byte but the last. One value has + // one spelling, so two conforming writers agree byte for byte. + LISTDEMO_TABLE_INLINE void putleb( uint64_t v ) + { + while ( v >= 0x80 ) { put8( uint8_t( v ) | 0x80 ); v >>= 7; } + put8( uint8_t( v ) ); + } +}; + +// TableLebBytes is one value's spelling length, which a MEASURE needs before +// the bytes exist — the length of a body has to be known before it is written, +// because a length whose own width moves cannot be patched in place. +inline int64_t TableLebBytes( uint64_t v ) +{ + int64_t n = 1; + while ( v >= 0x80 ) { v >>= 7; n++; } + return n; +} + +// THE ID TABLE, WRITER SIDE (docs/SPEC-TABLES.md §3). It holds every id the +// body used, once each, in FIRST-USE order over the whole wire, and the body +// names them by position: reference k is the kth entry, counted from 1, and +// reference 0 names NO ID. +// +// Its capacity is a COMPILE-TIME fact of the unit — the distinct names its +// table closure can spell — so a save allocates nothing: the table is a local +// of Measure and of Save. The bucket chain makes ref constant time and makes +// truncate constant time too, which is what an ELIDED field needs: a field +// that turns out not to ride costs nothing in the id table either, so the walk +// interns its id, builds the payload that decides, and undoes the entry when +// nothing rides. +struct TableIds +{ + static const int32_t kCapacity = 60; + static const int32_t kBuckets = 128; + + uint64_t ids[ kCapacity ]; + int32_t chain[ kCapacity ]; + int32_t head[ kBuckets ]; + int32_t count; + bool overflow; + // THE MESSAGE FORM'S SLOTS (docs/SPEC-TABLES.md §3.3). A form 2 wire + // names ids through the CONNECTION's table, which is the unit's whole + // vocabulary in a compiler-settled order — so every reference is known at + // compile time and rides at the header as a literal beside the id. This + // flag is what selects it: false interns the id in first-use order and + // writes a trailer, true answers the slot and writes none, and the walk + // that decides is one walk. + bool vocabulary; + + TableIds() : count( 0 ), overflow( false ), vocabulary( false ) + { + for ( int32_t i = 0; i < kBuckets; i++ ) { head[i] = -1; } + } + + static LISTDEMO_TABLE_INLINE uint32_t bucket_of( uint64_t id ) + { + return uint32_t( ( id * 0x9E3779B97F4A7C15ull ) >> 57 ) & uint32_t( kBuckets - 1 ); + } + + // the reference an id takes: its message-form SLOT under the connection's + // table, or the file's own first-use entry + LISTDEMO_TABLE_INLINE uint64_t ref( uint64_t id, uint64_t slot ) + { + if ( vocabulary ) { return slot; } + return intern( id ); + } + + // the FILE form's half, appending the id on first use + uint64_t intern( uint64_t id ) + { + const uint32_t b = bucket_of( id ); + for ( int32_t i = head[b]; i >= 0; i = chain[i] ) + { + if ( ids[i] == id ) { return uint64_t( i ) + 1; } + } + if ( count >= kCapacity ) { overflow = true; return 1; } + ids[count] = id; chain[count] = head[b]; head[b] = count; count++; + return uint64_t( count ); + } + + // undo every entry appended since mark. An entry removed is the most + // recent one in its bucket, so it sits at that bucket's head. + void truncate( int32_t mark ) + { + // a SLOT costs no entry, so an elided field has nothing to undo + if ( vocabulary ) { return; } + while ( count > mark ) + { + count--; + head[ bucket_of( ids[count] ) ] = chain[count]; + } + } +}; + +// TableIdsBytes is the trailer's own size: the entries, each a fixed +// little-endian u64, and the ENTRY COUNT, the one fixed-width number on the +// wire (docs/SPEC-TABLES.md §3). +inline int64_t TableIdsBytes( const TableIds & ids ) { return int64_t( ids.count ) * 8 + 8; } + +// TableIdsWrite puts the trailer where the walk ended: a writer never patches, +// because first-use order is known only when the walk ends. +inline void TableIdsWrite( TableWriter & w, const TableIds & ids ) +{ + for ( int32_t i = 0; i < ids.count; i++ ) { w.put64( ids.ids[i] ); } + w.put64( uint64_t( ids.count ) ); +} + +// THE ID TABLE, READER SIDE (docs/SPEC-TABLES.md §3). A reader locates it from +// the END of the wire and resolves it ONCE, at open: the entries are eight +// bytes each and a body names them by position, so every field dispatches +// through an index rather than through a search over hashes. +struct TableIdTable +{ + const uint8_t * entries = NULL; + int64_t count = 0; + + // the id a reference names. ref is 1-based and bounds-checked by the + // caller: a reference ABOVE the entry count is framing damage on the body + // that carries it, and 0 names no id at all. + uint64_t at( uint64_t ref ) const + { + const uint8_t * e = entries + ( ref - 1 ) * 8; + uint64_t lo = uint64_t( e[0] ) | uint64_t( e[1] ) << 8 | uint64_t( e[2] ) << 16 | uint64_t( e[3] ) << 24; + uint64_t hi = uint64_t( e[4] ) | uint64_t( e[5] ) << 8 | uint64_t( e[6] ) << 16 | uint64_t( e[7] ) << 24; + return lo | ( hi << 32 ); + } +}; + +struct TableReader +{ + const uint8_t * buffer; + int64_t size; + int64_t offset = 0; + TableReport * report; + const TableIdTable * ids = NULL; + // ONLY THE ROOT BODY CARRIES THE NODE TABLE (docs/SPEC-TABLES.md §3.1), so + // a body has to know which it is: the reserved id inside a NESTED body is + // malformed, because a second numbering cannot exist. Every reader made + // for a payload is nested; the two the wire surfaces make for a root say so. + bool nested = true; + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report ) + : buffer( from_buffer ), size( from_size ), report( to_report ) {} + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report, const TableIdTable * to_ids ) + : buffer( from_buffer ), size( from_size ), report( to_report ), ids( to_ids ) {} + + LISTDEMO_TABLE_INLINE bool has( int64_t bytes ) const { return offset + bytes <= size; } + // A LENGTH IS A 64-BIT NUMBER AND A BUFFER IS NOT (docs/SPEC-TABLES.md + // §3): every length, count and index on this wire has sixty-four bits of + // capability, so one past what remains must be compared UNSIGNED. Casting + // it to int64 first turns 0xFFFFFFFFFFFFFFFF into -1, and a negative + // length looks like room. + LISTDEMO_TABLE_INLINE bool room( uint64_t bytes ) const { return bytes <= (uint64_t) ( size - offset ); } + LISTDEMO_TABLE_INLINE uint8_t get8() { return buffer[offset++]; } + LISTDEMO_TABLE_INLINE uint16_t get16() { uint16_t v = uint16_t( buffer[offset] ) | uint16_t( buffer[offset+1] ) << 8; offset += 2; return v; } + LISTDEMO_TABLE_INLINE uint32_t get32() { uint32_t v = uint32_t( buffer[offset] ) | uint32_t( buffer[offset+1] ) << 8 | uint32_t( buffer[offset+2] ) << 16 | uint32_t( buffer[offset+3] ) << 24; offset += 4; return v; } + LISTDEMO_TABLE_INLINE uint64_t get64() { uint64_t lo = get32(); uint64_t hi = get32(); return lo | ( hi << 32 ); } + LISTDEMO_TABLE_INLINE void get128( uint64_t & lo, uint64_t & hi ) { lo = get64(); hi = get64(); } + + // ONE CANONICAL UNSIGNED LEB128 (docs/SPEC-TABLES.md §3), and a + // non-minimal spelling is MALFORMED: 0x80 0x00 and 0x00 both spell zero, + // and only the second is legal input. An encoding past ten bytes, or a + // tenth byte with a bit above the 64th value bit, is malformed on the same + // rule. false = framing damage on the body carrying it. + bool getleb( uint64_t & value ) + { + // A NUMBER THIS READER REFUSES LEAVES THE CURSOR WHERE IT WAS. The + // caller's next question is often "did this body end exactly at its + // L", and a rejected number that had moved the cursor would answer + // that question with the damage already stepped over. + const int64_t at = offset; + value = 0; + uint32_t shift = 0; + for ( int32_t i = 0; i < 10; i++ ) + { + if ( !has( 1 ) ) { offset = at; return false; } + const uint8_t b = get8(); + if ( i == 9 && b > 1 ) { offset = at; return false; } + value |= uint64_t( b & 0x7F ) << shift; + if ( ( b & 0x80 ) == 0 ) + { + if ( i > 0 && b == 0 ) { offset = at; return false; } // a redundant continuation + return true; + } + shift += 7; + } + offset = at; + return false; + } + + // resolve one id reference against the file's table. false = a reference + // ABOVE the entry count, or a 0 where an id is required, both of which + // are framing damage on the body that carries it. + bool getid( uint64_t & id ) + { + uint64_t ref = 0; + if ( !getleb( ref ) ) { return false; } + if ( ref == 0 || ids == NULL || ref > (uint64_t) ids->count ) { return false; } + id = ids->at( ref ); + return true; + } + + // skip one payload by kind; false = framing damage. FOUR RULES COVER THE + // SET (docs/SPEC-TABLES.md §3), and a kind outside it is not skippable — + // which is why the set is closed and why kind 31 exists. + bool skip( uint8_t kind ) + { + switch ( kind ) + { + // the fixed-width kinds, each by its width: 18-29 are the 128-bit integers and + // the fixed-point family at every storage width (docs/SPEC-TABLES.md §3) + case 1: case 2: case 6: case 20: case 25: return has( 1 ) ? ( offset += 1, true ) : false; + case 3: case 7: case 21: case 26: return has( 2 ) ? ( offset += 2, true ) : false; + case 4: case 8: case 10: case 22: case 27: return has( 4 ) ? ( offset += 4, true ) : false; + case 5: case 9: case 11: case 23: case 28: return has( 8 ) ? ( offset += 8, true ) : false; + case 18: case 19: case 24: case 29: return has( 16 ) ? ( offset += 16, true ) : false; + case 17: case 30: // a NODE INDEX (§3.1) and an ENUM's variant reference: one LEB128 and stop + { + uint64_t ignored = 0; + return getleb( ignored ); + } + case 12: case 13: case 14: case 16: case 31: case 32: // 31 is the ESCAPE, 32 the payload-free kind + { + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + case 15: // union: the arm id reference, then its kind, its L and its payload (reference 0 = empty) + { + uint64_t arm = 0; + if ( !getleb( arm ) ) return false; + if ( arm == 0 ) return true; + if ( !has( 1 ) ) return false; + offset += 1; // the arm's kind byte + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + } + return false; + } +}; + +// The RESERVED node-table id, the one id the language holds back +// (docs/SPEC-TABLES.md §3.1, §5). It rides in every unit, pointered or not, +// because every body has to know that a NESTED body claiming one is damaged. +static const uint64_t kTableNodeTableFieldId = 0xFFFFFFFFFFFFFFFFull; + +// TableWireForm is the FORM BYTE, and it is the whole header +// (docs/SPEC-TABLES.md §3). A reader that meets a byte it does not know +// refuses the wire by name and never reports damage. +const uint8_t kTableWireForm = 1; + +// TableOpen reads the form byte and the trailer, in that order, and hands back +// the ROOT BODY. It answers one of three verdicts, because five zero counters +// and a false flag are what a clean read prints too: +// +// TableOpenOk the form is known and the table read whole +// TableOpenRefused a FORM BYTE this reader does not carry: nothing is +// decoded, nothing is counted, and no damage is reported +// TableOpenDamaged a table that cannot be read whole — fewer than eight +// bytes, a count whose entries run past the front of the +// file, a count that leaves no room for the form byte, or +// ONE ID IN TWO ENTRIES. The whole wire is malformed, +// nothing is decoded, and one event is counted. +// TableOpenBodyStopped the form and the table were good and the ROOT BODY +// could not be walked to its own terminator. What it +// decoded before that is kept, as everywhere on this wire. +enum TableOpenVerdict { TableOpenOk, TableOpenRefused, TableOpenDamaged, TableOpenBodyStopped }; + +inline TableOpenVerdict TableOpen( const uint8_t * buffer, int64_t bytes, TableIdTable & table, int64_t & body_bytes ) +{ + if ( bytes < 1 ) { return TableOpenDamaged; } + if ( buffer[0] != kTableWireForm ) { return TableOpenRefused; } + if ( bytes < 9 ) { return TableOpenDamaged; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + uint64_t count = lo | ( hi << 32 ); + if ( count > (uint64_t) ( bytes / 8 ) ) { return TableOpenDamaged; } + const int64_t span = (int64_t) count * 8 + 8; + if ( span + 1 > bytes ) { return TableOpenDamaged; } + table.entries = buffer + bytes - span; + table.count = (int64_t) count; + // THE ENTRIES ARE DISTINCT: a table that carries one id twice is malformed + // for the whole wire, because no wire this schema writes carries a repeat + // and it would leave one more shape of table for a hostile writer to aim + // at (docs/SPEC-TABLES.md §3). + for ( int64_t i = 1; i < table.count; i++ ) + { + const uint64_t id = table.at( uint64_t( i ) + 1 ); + for ( int64_t j = 0; j < i; j++ ) + { + if ( table.at( uint64_t( j ) + 1 ) == id ) { return TableOpenDamaged; } + } + } + body_bytes = bytes - span - 1; + return TableOpenOk; +} + +// TableBodyExtent walks a body's framing to the zero reference that ends it, +// so a reader can tell a body that ENDED EARLY — leaving bytes no field claims +// — from one that is merely damaged. ANY BYTE BETWEEN THE ROOT'S TERMINATOR +// AND THE TABLE'S FIRST ENTRY IS MALFORMED, because no field claims it and the +// two ends of the file have met (docs/SPEC-TABLES.md §3). +inline bool TableBodyEndsEarly( const uint8_t * body, int64_t bytes, const TableIdTable & table ) +{ + TableReport ignored; + TableReader r( body, bytes, &ignored, &table ); + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { return false; } + if ( ref == 0 ) { return r.offset != bytes; } + if ( ref > (uint64_t) table.count ) { return false; } + if ( !r.has( 1 ) ) { return false; } + if ( !r.skip( r.get8() ) ) { return false; } + } +} + +// THE MESSAGE FORM (docs/SPEC-TABLES.md §3.3): a FILE carries its own id +// table and a MESSAGE STREAM announces one and then carries none. +// +// A form 2 wire is TWO PARTS, the form byte and the root body: the body ends +// at its own zero reference as it does in a file, there is no trailer, and the +// message's last byte is the body's terminator. Its references resolve against +// the CONNECTION's table, which is the unit's whole vocabulary in the order +// the compiler settled. +const uint8_t kTableWireMessageForm = 2; + +// The RESERVED build-version id, the second id the language holds back (§5, +// §11), beside the node table's. It is the announcement's one required field, +// and a reserved id in any body but the one whose transport it is, is +// malformed (§3.1). +static const uint64_t kTableBuildVersionFieldId = 0xFFFFFFFFFFFFFFFEull; + +// The reserved NODE-TABLE id's own slot in this unit's vocabulary (§3.3). A +// pointered message names the node table through it, exactly as every other +// field header names its id through a slot. +static const uint64_t kTableNodeTableFieldSlot = 38; + +// THE UNIT'S ANNOUNCEMENT, byte for byte: 61 entries and 508 bytes. It is an +// ordinary form 1 FILE — the form byte, a body carrying the BUILD VERSION +// under the reserved id at kind 9, and the trailer that IS the connection's +// table, slot 1 the reserved id and slots 2 and up the vocabulary under one +// numbering. +// +// The vocabulary is the unit's whole closure in the COOK PROJECTION's order +// (§20.2) — each record in the order the projection renders it and each +// record's fields in the order the projection renders them, then each enum's +// variants and each union's arms — followed by the tail the projection does +// not name: the reserved node-table id, the three blob type ids as bytes, +// string and wstring, and every table's own name id in the projection's sorted +// record order. The tail is UNCONDITIONAL, so an ordinary edit only ever grows +// it at its end and never moves a slot a generated field header carries as a +// literal. +static const int64_t kTableAnnounceBytes = 508; +static const uint8_t kTableAnnounce[ kTableAnnounceBytes ] = { + 0x01, 0x01, 0x09, 0xc7, 0x71, 0x45, 0xca, 0xda, 0x0e, 0x7c, 0x8d, 0x00, + 0xfe, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x30, 0xb1, 0x3a, 0xff, + 0x4a, 0xd9, 0xb1, 0x40, 0x20, 0xea, 0x4d, 0x40, 0x8e, 0xa7, 0x19, 0xaa, + 0x26, 0xa9, 0x02, 0x0c, 0x9b, 0x01, 0x48, 0x78, 0xe9, 0xea, 0x71, 0x6f, + 0x0f, 0x01, 0x82, 0xbf, 0x6f, 0x2c, 0x41, 0x4f, 0xbf, 0x84, 0x78, 0x3e, + 0xf3, 0xa4, 0x48, 0x44, 0x19, 0xab, 0xd7, 0x56, 0x05, 0x4a, 0xa3, 0x30, + 0x67, 0x55, 0x5b, 0x85, 0xc9, 0xe2, 0x4e, 0x30, 0x69, 0x6a, 0xb4, 0x81, + 0xfb, 0x67, 0x4d, 0x1a, 0xcf, 0x7b, 0x27, 0x21, 0x74, 0xa2, 0x79, 0x44, + 0x8e, 0xe2, 0xe5, 0xb1, 0x84, 0x76, 0xbc, 0x2e, 0xef, 0x83, 0x76, 0x1e, + 0xc5, 0x99, 0xf7, 0x82, 0x76, 0x4e, 0x0a, 0xd9, 0xa8, 0x2e, 0x86, 0x70, + 0x84, 0xed, 0xf2, 0x4a, 0xbb, 0xf0, 0x0c, 0x9b, 0xcc, 0xfb, 0x2d, 0x73, + 0x68, 0xb7, 0xf0, 0xae, 0x4c, 0x0c, 0xf6, 0x52, 0xbf, 0xe9, 0xd1, 0x2f, + 0x93, 0xcd, 0xda, 0xdb, 0x22, 0x72, 0x34, 0x7d, 0xf6, 0x0b, 0x72, 0x17, + 0x07, 0x17, 0x02, 0x86, 0x4c, 0xf5, 0x63, 0xaf, 0x54, 0x15, 0x02, 0x86, + 0x4c, 0xf4, 0x63, 0xaf, 0x3a, 0x70, 0x6e, 0x3e, 0x93, 0x43, 0xe5, 0x9d, + 0x3d, 0x62, 0xcb, 0x8f, 0xec, 0xfc, 0xf7, 0x39, 0x09, 0x06, 0x02, 0x86, + 0x4c, 0xeb, 0x63, 0xaf, 0x09, 0x4b, 0x4d, 0x57, 0xaa, 0x33, 0x47, 0xd2, + 0x31, 0x54, 0xaf, 0x1d, 0x19, 0x73, 0x50, 0x12, 0xb2, 0x0f, 0x40, 0x27, + 0x0b, 0x6b, 0x98, 0x01, 0x38, 0x81, 0x0a, 0xf1, 0x1f, 0x06, 0xa7, 0xa3, + 0x0f, 0x62, 0xad, 0x07, 0x77, 0x47, 0x82, 0x5f, 0x42, 0x4f, 0x4f, 0x30, + 0x0d, 0x39, 0x84, 0x1c, 0x86, 0x1b, 0x63, 0x8e, 0xba, 0xad, 0xbc, 0xc4, + 0xec, 0x10, 0x5b, 0x36, 0x19, 0x4a, 0xc9, 0x3d, 0xea, 0x0c, 0xe8, 0x30, + 0x94, 0xfd, 0xe4, 0x7c, 0xec, 0x22, 0x02, 0x86, 0x4c, 0xfc, 0x63, 0xaf, + 0x05, 0x28, 0x02, 0x86, 0x4c, 0xff, 0x63, 0xaf, 0x52, 0x26, 0x02, 0x86, + 0x4c, 0xfe, 0x63, 0xaf, 0xb1, 0x45, 0xc3, 0x44, 0x35, 0xab, 0xfe, 0x73, + 0xc0, 0x7f, 0xb3, 0x8a, 0xbe, 0x08, 0x63, 0x7f, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xe4, 0x4f, 0x1c, 0x4f, 0x47, 0xc0, 0x2e, 0x2f, + 0x58, 0xfc, 0xaf, 0xfa, 0xd8, 0xe0, 0x4b, 0x70, 0xc7, 0xd4, 0x7b, 0x26, + 0xb0, 0x9d, 0x29, 0x5f, 0xcc, 0x14, 0x15, 0x7f, 0xcb, 0xc2, 0x58, 0xd8, + 0x84, 0x9e, 0x3a, 0x55, 0x8b, 0x37, 0xe2, 0x06, 0x2a, 0x89, 0xf5, 0x12, + 0x09, 0xc1, 0xca, 0x0a, 0x44, 0xa2, 0x31, 0xc1, 0xad, 0xa7, 0xee, 0xee, + 0xe8, 0xcf, 0xbf, 0x43, 0x73, 0x18, 0x43, 0xd0, 0x42, 0xad, 0xf6, 0xf8, + 0x59, 0x86, 0x63, 0x91, 0xb7, 0xce, 0x00, 0x7c, 0xd1, 0xc5, 0x34, 0x20, + 0x06, 0x68, 0x47, 0x98, 0xd1, 0xa1, 0xcf, 0x52, 0x5f, 0x82, 0x58, 0xac, + 0x36, 0x15, 0x78, 0x5e, 0xb8, 0x8b, 0x59, 0x6f, 0xc9, 0xc6, 0x86, 0xbb, + 0xc3, 0x64, 0x89, 0x50, 0xd2, 0x8d, 0xa7, 0xf1, 0x80, 0xea, 0x3a, 0xb9, + 0xf1, 0x21, 0xf7, 0x41, 0x11, 0xed, 0xd9, 0xce, 0x96, 0x92, 0x43, 0x8a, + 0xfb, 0x06, 0xc9, 0xfe, 0x19, 0xe1, 0x13, 0xa0, 0xa7, 0x0a, 0xc7, 0x54, + 0x12, 0xd6, 0x40, 0xdc, 0x08, 0xf0, 0xf5, 0xc0, 0x24, 0x5f, 0xf8, 0x33, + 0xc8, 0xfb, 0x85, 0x9a, 0xaf, 0xe0, 0xc9, 0x0c, 0x91, 0x0a, 0x55, 0x60, + 0xf7, 0xa2, 0x07, 0xec, 0x8b, 0x6d, 0x02, 0x86, 0x43, 0xf3, 0xc2, 0x2e, + 0x87, 0x27, 0xcc, 0x86, 0xf0, 0xe0, 0x26, 0x8f, 0x3d, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, +}; + +// AnnounceMeasure is the announcement's byte count, which is a constant of the +// unit and not a walk. +inline int64_t AnnounceMeasure() { return kTableAnnounceBytes; } + +// Announce writes the announcement into the caller's buffer and answers the +// bytes written — exactly AnnounceMeasure's answer — or -1 when the buffer is +// too small. It allocates nothing and walks nothing. +inline int64_t Announce( uint8_t * buffer, int64_t capacity ) +{ + if ( buffer == NULL || capacity < kTableAnnounceBytes ) { return -1; } + memcpy( buffer, kTableAnnounce, (size_t) kTableAnnounceBytes ); + return kTableAnnounceBytes; +} + +// TableVocabulary is ONE DIRECTION of ONE CONNECTION's id table (§3.3): the +// entries an announcement carried, whole, under one numbering with slot 1 the +// reserved build-version id. +// +// A peer holds TWO of these for a connection, the one it writes with and the +// one it reads with, and neither is the other's. A restart opens a fresh +// connection with empty tables and nothing is cached across connections, so +// its whole life is one connection's. It BORROWS the announcement's bytes rather than +// copying them, so a receiver holds one table a direction and its memory is +// the bound below and nothing else. +struct TableVocabulary +{ + // THE CONFORMING DEFAULT BOUND (§3.3): 32 KiB a direction, eight times the + // 500-id unit that is already a large one. A connection's table is bounded + // by nothing the wire carries, so the receiver declares the maximum and an + // announcement above it is refused by name before an entry is touched. + static const int64_t kDefaultMaxEntries = 4096; + + TableIdTable table; + uint64_t build_version = 0; + bool announced = false; + int64_t max_entries = kDefaultMaxEntries; +}; + +// AnnounceRead reads an announcement into one direction's table (§3.3). +// +// THE BOUND IS CHECKED BEFORE ANYTHING IS ALLOCATED: the entry count is a +// fixed little-endian u64 at the end, so a receiver reads it, compares it and +// refuses without touching an entry. After that it is §3's ordinary FILE read, +// because the announcement IS a file, with EXACTLY ONE STRICT CHECK over its +// body: the reserved build-version field present, exactly once, under kind 9, +// eight bytes wide. Everything else is an ordinary field under §4's tolerance, +// so an unknown one is skipped and counted and the announcement can GAIN a +// field in a later minor without a lockstep redeploy. +// +// The FIRST announcement sets the table and it is the only one that can. A +// SECOND is refused by name: it does not replace the table, it does not amend +// it and it changes nothing. A refused announcement sets NO TABLE. +inline bool AnnounceRead( TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + if ( vocabulary.announced ) + { + to->refused = true; + to->reason = second_announcement; + return false; + } + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireForm ) + { + to->refused = true; + to->reason = buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + return false; + } + if ( bytes < 9 ) { to->malformed = true; return false; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + if ( ( lo | ( hi << 32 ) ) > (uint64_t) vocabulary.max_entries ) + { + to->refused = true; + to->reason = vocabulary_too_large; + return false; + } + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else { to->refused = true; to->reason = newer_form; } + return false; + } + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) { to->malformed = true; return false; } + // the body, under §4's tolerance and this form's one strict check + TableReader r( buffer + 1, body_bytes, to, &table ); + uint64_t version = 0; + int32_t seen = 0; + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { to->malformed = true; return false; } + if ( ref == 0 ) { break; } + if ( ref > (uint64_t) table.count || !r.has( 1 ) ) { to->malformed = true; return false; } + const uint64_t id = table.at( ref ); + const uint8_t kind = r.get8(); + if ( id != kTableBuildVersionFieldId ) + { + to->unknown++; + if ( !r.skip( kind ) ) { to->malformed = true; return false; } + continue; + } + if ( kind != 9 || !r.has( 8 ) ) { to->refused = true; to->reason = no_vocabulary; return false; } + version = r.get64(); + seen++; + } + if ( seen != 1 ) { to->refused = true; to->reason = no_vocabulary; return false; } + vocabulary.table = table; + vocabulary.build_version = version; + vocabulary.announced = true; + return true; +} + +inline float table_bits_to_float( uint32_t bits ) { float f; memcpy( &f, &bits, 4 ); return f; } +inline uint32_t table_float_to_bits( float f ) { uint32_t b; memcpy( &b, &f, 4 ); return b; } +inline double table_bits_to_double( uint64_t bits ) { double d; memcpy( &d, &bits, 8 ); return d; } +inline uint64_t table_double_to_bits( double d ) { uint64_t b; memcpy( &b, &d, 8 ); return b; } + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +#ifndef LISTDEMO_SCHEMA_TABLE_ARENA +#define LISTDEMO_SCHEMA_TABLE_ARENA + +namespace listdemo { + +// ---- variable-length tables: tuning constants (docs/SPEC-TABLES.md) ---- +// +// The segment size and the count multiply to exactly 2^32: the u32 reference +// is the arena's hard ceiling, and these constants saturate it rather than +// leaving address space unreachable. Slab handout costs one atomic per slab, +// so per-node allocation costs no synchronization at all. + +static const uint32_t kTableSegmentBits = 22; // 4 MiB segments +static const uint32_t kTableSegmentSize = 1u << kTableSegmentBits; +static const uint32_t kTableSegmentMask = kTableSegmentSize - 1u; +static const uint32_t kTableMaxSegments = 1u << ( 32 - kTableSegmentBits ); // 1024 -> 4 GiB +static const uint32_t kTableSlabBytes = 64u * 1024u; // one atomic per slab +static const uint32_t kTableAlign = 8; // every node starts 8-aligned +static const uint32_t kTableAllocFailed = 0xFFFFFFFFu; + +// ---- THE CALLER'S ALLOCATOR (docs/SPEC-TABLES.md §6.5) ---- +// +// Every allocation the variable-length runtime makes goes through one of +// these — the arena's segments, the pack walk's identity map, the numbering's +// entry array, the packed region, and the tool path's node directory. There is +// no other call to the C library on this path, so a counting allocator sees +// every byte and a game's own heap can own all of it. +// +// It is the shape TableBlockAllocator already has (§19.1): two function +// pointers and a context the caller carries. What it adds is a CONTRACT ON +// alloc — the bytes come back ZEROED. Lock copies whole nodes, PADDING +// INCLUDED, so anything left uninitialized reaches a packed region; the default +// pair reaches that through calloc, which costs nothing measurable because a +// fresh segment is untouched pages either way. +struct TableAllocator +{ + void * ( *alloc )( void * context, int64_t bytes ); // ZEROED bytes, NULL on failure + void ( *free )( void * context, void * pointer ); + void * context; +}; + +// The default pair, and it is the one every entry point takes when the caller +// names none. It calls schema_allocate / schema_release, so a program with its +// own C-library replacement can move the floor without writing a struct at all. +inline void * table_default_alloc( void * context, int64_t bytes ) { (void) context; return schema_allocate( bytes ); } +inline void table_default_free( void * context, void * pointer ) { (void) context; schema_release( pointer ); } + +inline TableAllocator TableDefaultAllocator() +{ + TableAllocator allocator; + allocator.alloc = table_default_alloc; + allocator.free = table_default_free; + allocator.context = NULL; + return allocator; +} + +// ---- TableRef: a relocatable reference (never a machine pointer) ---- +// +// Two encodings, one slot, and the FORM says which is in force: +// +// in the arena — the node's arena offset (segment index in the high bits) +// in a region — the SELF-RELATIVE byte delta from this slot's own address, +// so a deref is one add, needs no base pointer, and a whole +// region relocates by memcpy with zero fix-up +// +// 0 is null in both, and a slot can never name the node that contains it, so +// zero names nothing real in either form. +// +// A REGION DELTA HAS NO REQUIRED SIGN (§6.3). A region is packed depth-first, +// so a node's FIRST reference points forward; every LATER reference to that +// same node points BACK at the one body it already has, which is exactly what +// makes one node one node in a region. Sharing and a back-reference are the +// same fact, and nothing validates a reference by its sign. +// +// IT IS EIGHT BYTES, SIGNED, so ONE REGION REACHES EVERYTHING (§6.3, §7): a +// four-byte slot bounded a region at 2 GiB, and the scale a cook exists for is +// *"100mbs or many gigabytes of data in Assets.bin"*. +struct TableRef +{ + int64_t value = 0; + bool null() const { return value == 0; } +}; + +// TableSlot is what Alloc hands back: usable as the node pointer (write +// fields through it) AND as the reference to store in a pointer field. +template struct TableSlot +{ + T * ptr = NULL; + TableRef ref; + T * operator->() const { return ptr; } + T & operator*() const { return *ptr; } + operator T *() const { return ptr; } + operator TableRef() const { return ref; } + bool null() const { return ptr == NULL; } +}; + +inline uint32_t TableAlignUp( uint32_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( kTableAlign - 1 ); } +inline int64_t TableAlignUp64( int64_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( int64_t( kTableAlign ) - 1 ); } + +// ---- a BYTE BUFFER's node (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// A *bytes or *string slot is a TableRef like every pointer slot, and it names +// a BLOB NODE: this eight-byte header and then the bytes, at offset eight so +// the data is eight-aligned. A *string blob carries one more zero byte after +// its data, so a region hands back a C string with no copy. The node's extent +// is the header plus its bytes, rounded to the arena's alignment like every +// node's; on the wire it is a record whose body is the bytes (§3.1). +struct TableBlob +{ + uint32_t length; + uint32_t zero; +}; + +static const int64_t kTableBlobHeader = 8; // length (u32), then four zero bytes +static const int64_t kTableBlobMaxLength = 0xFFFFFFFF; // a record's length is a u32 (§3.1) + +// the node's storage: the header, the bytes, a string's terminator, rounded +// to the arena's alignment like every node +inline int64_t TableBlobStorage( int64_t length, bool terminated ) +{ + return TableAlignUp64( kTableBlobHeader + length + ( terminated ? 1 : 0 ) ); +} + +// What a read answers: a pointer INTO the region and the length, NULL and +// zero for a null slot. Off a locked region, a loaded one or an opened cook +// the pointer is one add from the slot, and nothing is copied. +struct TableBytesView +{ + const uint8_t * data; + int64_t length; +}; + +struct TableStringView +{ + const char * data; // zero-terminated + int64_t length; +}; + +// What AllocBytes and AllocString hand back: the bytes to write through, the +// length asked for, and the reference to store in the slot — the three +// answers TableSlot gives for a table node. +struct TableBytesSlot +{ + uint8_t * data = NULL; + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +struct TableStringSlot +{ + char * data = NULL; // room for length bytes and the terminator, already zero + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +// ---- the arena: segmented, slab-handed, lock-free by ownership ---- +// +// Allocation is thread-local inside a worker's slab — no atomics on the node +// path. A worker takes its next slab with ONE compare-exchange, and a new +// segment is published with one more. Nothing ever moves: a segment, once +// allocated, lives untouched until the arena is torn down, so a T* obtained +// from Alloc stays valid while other workers allocate, and an offset stays +// correct while the arena grows. +// +// The model this DELIBERATELY refuses: one buffer under a lock, grown by +// realloc. A realloc moves the buffer under workers mid-write; offsets fix +// identity but not the raw references already resolved from them, and the +// resulting corruption is invisible until much later. Segments never move, so +// that bug class cannot be written here. +// +// Slack: at most one slab tail per worker plus one slab per segment (a slab +// that will not fit is skipped rather than split), i.e. under 2% of a segment +// plus threads x 64 KiB. That is the price of never synchronizing per node. +struct TableArena +{ + std::atomic segments[ kTableMaxSegments ]; + std::atomic cursor; // (segment << kTableSegmentBits) | bytes handed out + bool locked = false; // MONOTONIC: Lock() is one-way, there is no unlock + // THE ARENA CARRIES ITS OWN, so everything downstream of a builder — + // segments, pack map, numbering, region, node directory — allocates through + // the one pair the caller named, with nothing to thread by hand. + TableAllocator allocator; +}; + +inline void TableArenaInit( TableArena & arena, TableAllocator allocator ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + arena.segments[i].store( NULL, std::memory_order_relaxed ); + } + arena.cursor.store( 0, std::memory_order_relaxed ); + arena.locked = false; + arena.allocator = allocator; +} + +inline void TableArenaShutdown( TableArena & arena ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + uint8_t * segment = arena.segments[i].exchange( NULL, std::memory_order_acq_rel ); + if ( segment != NULL ) { arena.allocator.free( arena.allocator.context, segment ); } + } + arena.cursor.store( 0, std::memory_order_relaxed ); +} + +// one L1 load plus an add: the segment table is 8 KiB and stays hot +inline uint8_t * TableArenaAt( const TableArena & arena, uint32_t offset ) +{ + return arena.segments[ offset >> kTableSegmentBits ].load( std::memory_order_relaxed ) + ( offset & kTableSegmentMask ); +} + +// TableArenaGrabSlab hands one worker its next private slab. Returns +// kTableAllocFailed when the arena's address space or the allocator is +// exhausted — a loud refusal, never a silent smaller slab. +inline uint32_t TableArenaGrabSlab( TableArena & arena ) +{ + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t segment = cursor >> kTableSegmentBits; + uint32_t used = cursor & kTableSegmentMask; + // strictly less: a slab is never split across segments, and the tail + // is the documented slack + if ( used + kTableSlabBytes < kTableSegmentSize ) + { + if ( arena.segments[segment].load( std::memory_order_acquire ) == NULL ) + { + // THE SEGMENT COMES BACK ZEROED, which is the allocator's + // contract and not an extra pass here: Lock copies whole nodes, + // PADDING INCLUDED, so anything uninitialized reaches a packed + // region. Value-initializing a node with placement new zeroes + // its MEMBERS and not its padding, so the zeroing has to happen + // at the segment or not at all. It costs nothing measurable: a + // fresh segment is untouched pages either way, and the default + // pair's calloc has the kernel hand them over zeroed. + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, (int64_t) kTableSegmentSize ); + if ( memory == NULL ) { return kTableAllocFailed; } + uint8_t * expected = NULL; + if ( !arena.segments[segment].compare_exchange_strong( expected, memory, std::memory_order_acq_rel ) ) + { + // another worker published this segment first + arena.allocator.free( arena.allocator.context, memory ); + } + } + if ( arena.cursor.compare_exchange_weak( cursor, cursor + kTableSlabBytes, std::memory_order_acq_rel ) ) + { + return ( segment << kTableSegmentBits ) | used; + } + continue; + } + uint32_t next_segment = segment + 1; + if ( next_segment >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + arena.cursor.compare_exchange_weak( cursor, next_segment << kTableSegmentBits, std::memory_order_acq_rel ); + } +} + +// TableArenaGrabSpan reserves a SPAN of the arena's address space for one node +// larger than a slab — a BYTE BUFFER of any size (docs/SPEC-TABLES.md §2.5) — +// and allocates it as one contiguous block. It takes whole segment indices +// from the cursor, starting at the index after the cursor's so nothing else +// is ever handed out inside the span, and publishes the block under the first +// of them; the indices the span covers past that one stay NULL, which is +// enough, because only a node's START is ever resolved through the segment +// table and a blob's bytes follow its header inside the one allocation. The +// unused tail of the segment the cursor was in is slack, like a slab tail. +// Returns kTableAllocFailed when the address space or the allocator is +// exhausted — a loud refusal, never a smaller blob. +inline uint32_t TableArenaGrabSpan( TableArena & arena, int64_t bytes ) +{ + if ( bytes <= 0 || bytes > ( (int64_t) kTableMaxSegments - 2 ) * (int64_t) kTableSegmentSize ) { return kTableAllocFailed; } + const uint32_t spanned = (uint32_t) ( ( bytes + kTableSegmentSize - 1 ) >> kTableSegmentBits ); + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t start = ( cursor >> kTableSegmentBits ) + 1; + if ( start + spanned >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + uint32_t next = ( start + spanned ) << kTableSegmentBits; + if ( !arena.cursor.compare_exchange_weak( cursor, next, std::memory_order_acq_rel ) ) { continue; } + // the span is this worker's now: nothing else can publish under its + // first index, so a plain store suffices, and the block comes back + // ZEROED like every segment — the blob's bytes and its tail are zeros + // until written + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, bytes ); + if ( memory == NULL ) { return kTableAllocFailed; } + arena.segments[start].store( memory, std::memory_order_release ); + return start << kTableSegmentBits; + } +} + +// ---- TableWorker: one thread's allocation front ---- +// +// The threading contract, stated plainly: +// * Alloc on YOUR OWN worker is safe concurrently with any other worker's. +// No locks, no atomics per node. +// * Writing fields of a node ANOTHER worker allocated is your own +// synchronization problem — this runtime does not arbitrate it. +// * Lock and Save are single-threaded: call them after the workers have +// joined. +struct TableWorker +{ + TableArena * arena = NULL; + uint32_t next = 0; + uint32_t end = 0; + + template TableSlot Alloc() + { + static_assert( alignof( T ) <= kTableAlign, "a table node's alignment must fit the arena's" ); + TableSlot slot; + if ( arena == NULL || arena->locked ) { return slot; } + uint32_t bytes = TableAlignUp( (uint32_t) sizeof( T ) ); + if ( bytes > kTableSlabBytes ) { return slot; } // a node larger than a slab: refused, never split + if ( end == 0 || next + bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return slot; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + uint32_t at = next; + next += bytes; + // A NODE IS BORN IN TWO HALVES: start its lifetime in the raw + // storage, then write the declared defaults ONE MEMBER AT A TIME. + // + // It is "T", not "T{}". Value-initialising the whole aggregate says + // the same thing and costs cl O(BYTES) TO COMPILE — it expands element + // by element in its front end — while both halves here cost + // O(declarations). The slab cap below refuses a large node at RUN + // TIME and bounds nothing at compile time: the cost is paid by + // whatever T a caller instantiates this with. + // Padding is not the difference: value-initialisation zeroes MEMBERS + // and not padding either way, which is why the segment is calloc'd. + // + // TableReset is an OVERLOAD SET, one per closure member, reached from + // this template by argument-dependent lookup on T's own namespace — + // Alloc is a template and cannot spell Reset. + // + // The reset is here because ONE DEFINITION SAYS WHAT THE DECLARED + // DEFAULTS ARE, and it is Reset. Default-initialisation lands on + // the same values today, because a member with a non-zero default + // carries a member initializer that says so — but that is the class + // definition agreeing with Reset, not the arena reading it, and #320's + // fix was itself a pass that MOVED initialisation between the two. + // The arena reads the definition. + slot.ptr = new ( TableArenaAt( *arena, at ) ) T; + TableReset( *slot.ptr ); + slot.ref.value = at; + return slot; + } + + // Alloc a BYTE BUFFER's node of exactly length bytes (docs/SPEC-TABLES.md + // §2.5): the blob header and its bytes, zeroed, in this thread's slab when + // it fits and in a span of the arena's own when it does not. NULL is the + // arena locked, a length below zero or past a record's u32, or the + // allocator refusing. The offset comes back for the reference. + TableBlob * AllocBlob( int64_t length, bool terminated, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( length < 0 || length > kTableBlobMaxLength ) { return NULL; } + const int64_t bytes = TableBlobStorage( length, terminated ); + if ( bytes > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, bytes ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + } + else + { + if ( end == 0 || next + (uint32_t) bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) bytes; + } + TableBlob * blob = (TableBlob *) TableArenaAt( *arena, at ); + blob->length = (uint32_t) length; // the bytes after it are the segment's zeros + blob->zero = 0; + return blob; + } + + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries + // no type id, takes no index and has no Reset, so it goes through the same + // slab and span the blob path uses rather than through Alloc. + uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( bytes <= 0 || align > (int64_t) kTableAlign ) { return NULL; } + const int64_t rounded = TableAlignUp64( bytes ); + if ( rounded > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, rounded ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + return TableArenaAt( *arena, at ); + } + if ( end == 0 || next + (uint32_t) rounded > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) rounded; + return TableArenaAt( *arena, at ); // the segment came back zeroed + } + // a *bytes node: the bytes to write through, and the reference to store + TableBytesSlot AllocBytes( int64_t length ) + { + TableBytesSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, false, at ); + if ( blob == NULL ) { return slot; } + slot.data = (uint8_t *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } + + // a *string node: room for length bytes and the zero byte after them + TableStringSlot AllocString( int64_t length ) + { + TableStringSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, true, at ); + if ( blob == NULL ) { return slot; } + slot.data = (char *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } +}; + +// ---- TablePackMap: the pack walk's identity map (docs/SPEC-TABLES.md §3.1, §6.2) ---- +// +// ONE ENTRY PER REACHABLE NODE, and that map IS identity: a node must know +// where it landed to be named a second time, so Lock packs a shared node ONCE +// and every later reference resolves to the one body it already has. That is +// the same first-visit numbering the wire uses, so the pack order and the node +// order are one order. +// +// COLOURING AN ENTRY WHILE ITS DESCENT IS OPEN COSTS ONE BIT, and it is what +// makes a data cycle free to refuse: a reference to an entry still open is a +// cycle, and Lock returns failure rather than recursing away. The ROOT's entry +// is open for the whole walk. +// +// The map is proportional to NODES, never to bytes, and it lives on the +// AUTHORING side, where §6.5 licenses allocation. Nothing on the reading path +// ever builds one. +struct TablePackEntry +{ + const void * key; // the node's address in the graph being packed + int64_t offset; // where that node landed in the region + uint8_t open; // its descent is still open: a reference here is a cycle +}; + +struct TablePackMap +{ + TablePackEntry * entries = NULL; + int64_t capacity = 0; // a power of two, or zero while empty + int64_t count = 0; + TableAllocator allocator; // the caller's, carried from the walk that built it +}; + +inline void TablePackMapInit( TablePackMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TablePackMapShutdown( TablePackMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TablePackMapInit( map, map.allocator ); +} + +// The two walks behind Lock re-derive the SAME map from the same graph — the +// numbering is never carried between them (§3.1) — so the second starts from +// an empty map and keeps the capacity the first paid for. +inline void TablePackMapReset( TablePackMap & map ) +{ + if ( map.entries != NULL ) { memset( map.entries, 0, (size_t) map.capacity * sizeof( TablePackEntry ) ); } + map.count = 0; +} + +// open addressing, linear probing, a multiply-shift hash over the address: a +// node key is a pointer and its low bits are alignment, so the low bits alone +// would collide on every node of one type +inline int64_t TablePackMapSlot( const TablePackMap & map, const void * key ) +{ + uint64_t hash = (uint64_t) (uintptr_t) key; + hash *= 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != NULL && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TablePackEntry * TablePackMapFind( TablePackMap & map, const void * key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TablePackEntry * entry = &map.entries[ TablePackMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +// QUADRUPLING, not doubling, and the reason is measured: growth rehashes every +// entry, and on a graph of 131,071 nodes the doubling schedule spent 45% of +// Lock in rehashing alone. Quadrupling from 1024 buys 1.35x on that graph and +// keeps the map NODE-proportional (§6.2) — under 128 bytes a node at its +// worst, right after a grow, and about 64 on average. +inline bool TablePackMapGrow( TablePackMap & map ) +{ + TablePackMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 1024; + grown.entries = (TablePackEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TablePackEntry ) ); + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == NULL ) { continue; } + grown.entries[ TablePackMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// REACH a node: one probe answers both questions the walk has. A true "taken" +// says this is a FIRST visit, and the entry is now the node's, coloured open +// at "offset"; otherwise the entry is the one the node already has, and its +// open bit says cycle or sharing. NULL is an allocation failure, and it is a +// refusal like any other: Lock fails rather than packing a graph it cannot +// track. +// +// It is one call and not a find followed by an insert because the walk asks +// this question twice per node — once to measure, once to pack — and every +// probe is a miss into a table larger than L2. +inline TablePackEntry * TablePackMapReach( TablePackMap & map, const void * key, int64_t offset, bool & taken, int64_t & slot ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TablePackMapGrow( map ) ) { return NULL; } + } + slot = TablePackMapSlot( map, key ); + TablePackEntry * entry = &map.entries[slot]; + taken = entry->key != key; // an empty slot is a first visit; the key is never NULL + if ( taken ) + { + entry->key = key; + entry->offset = offset; + entry->open = 1; + map.count++; + } + return entry; +} + +// The descent finished: the node keeps its entry — identity outlives the +// descent — and stops being a cycle. The "hint" is the slot Reach returned, and it +// is checked against the key rather than trusted, so a rehash between the two +// costs a second probe instead of correctness. +inline void TablePackMapClose( TablePackMap & map, const void * key, int64_t hint ) +{ + if ( hint >= 0 && hint < map.capacity && map.entries[hint].key == key ) + { + map.entries[hint].open = 0; + return; + } + TablePackEntry * entry = TablePackMapFind( map, key ); + if ( entry != NULL ) { entry->open = 0; } +} + +// ---- resolution contexts: which encoding a walk is reading ---- + +struct TableArenaCtx { const TableArena * arena; }; +struct TableRegionCtx {}; + +// ---- a BYTE BUFFER's resolution (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// The same two encodings a table pointer has, resolved the same way: a +// self-relative delta in a region — one add, no base — and an arena offset +// while the builder is mutable. The blob is reached through its header, and a +// view is the header plus eight and the header's first word. Nothing here +// allocates and nothing copies: off a locked region, a loaded one or an +// opened cook the view points INTO the region. +inline const TableBlob * TableBlobAt( const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableRegionCtx &, const TableRef & ref ) { return TableBlobAt( ref ); } +inline const TableBlob * TableBlobAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} + +inline TableBytesView TableBytesViewOf( const TableBlob * blob ) +{ + TableBytesView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const uint8_t *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} +inline TableStringView TableStringViewOf( const TableBlob * blob ) +{ + TableStringView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const char *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} + +// the const form's hot path: one add, no base +inline TableBytesView TableBytesAt( const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ref ) ); } +inline TableStringView TableStringAt( const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ref ) ); } +// and the context forms a walk uses: a region context, an arena context, or +// the arena itself while the builder is mutable +template inline TableBytesView TableBytesAt( const Ctx & ctx, const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ctx, ref ) ); } +template inline TableStringView TableStringAt( const Ctx & ctx, const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ctx, ref ) ); } + +// allocate a blob in the arena and point the slot at it; the slot holds the +// arena offset, as every slot does while the builder is mutable +inline uint8_t * TableBytesEmplace( TableWorker & worker, TableRef & slot, int64_t length ) +{ + TableBytesSlot allocated = worker.AllocBytes( length ); + slot = allocated.ref; + return allocated.data; +} +// the text is copied in when one is given; a NULL text leaves the zeros for +// the caller to fill +inline char * TableStringEmplace( TableWorker & worker, TableRef & slot, const char * text, int64_t length ) +{ + TableStringSlot allocated = worker.AllocString( length ); + slot = allocated.ref; + if ( allocated.data != NULL && text != NULL && length > 0 ) { memcpy( allocated.data, text, (size_t) length ); } + return allocated.data; +} + +// ---- the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table, and a +// pointer field rides as an INDEX into it under kind 17. The encoding is +// flat: no pointer edge is a nesting level, so a chain's length is not a depth, +// and two references to one node are one node. +// +// THE FIELD RIDES ONCE: an L with sixty-four bits of capability frames a +// numbering of any size, so the whole numbering is one contiguous payload and a +// save's node bodies have no aggregate ceiling. + +static const uint64_t kTableNodeIndexNull = 0; // absence and null are one value +static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts the table + +// The not-materialized sentinel (§6.3): a record whose type id this build could +// not name. Distinct from every real offset including the root's 0, so an index +// resolving through it yields NULL and can never fabricate the root. +static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; + +// ---- the numbering, on the SAVE side ---- +// +// One entry per reachable node in FIRST-VISIT order, so entry k is node index +// k + 2. The two thunks are what let one loop write a table of mixed types: the +// numbering walk knows each target's type STATICALLY at the site it numbers it, +// so it stores the instantiation there and the loop never asks what a node is. +struct TableNumbering; + +struct TableNodeEntry +{ + const void * node; + uint64_t type_id; + // the type id's MESSAGE-FORM SLOT (docs/SPEC-TABLES.md §3.3), stored where + // the numbering walk stores the id itself and for the same reason: the + // target's type is known STATICALLY at the site that numbers it, so a + // form 2 save reads the slot out of the entry instead of looking an id up. + // Every pointer target's type id is an entry of the announcement, which is + // what makes the slot a compile-time fact of a POINTERED message too. + uint64_t type_slot; + int64_t ( * measure )( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ); + bool ( * save )( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ); +}; + +struct TableNumbering +{ + TablePackMap seen; // node -> index; the ROOT is index 1, open for the whole walk + TableNodeEntry * entries = NULL; + int64_t count = 0; + int64_t capacity = 0; +}; + +// The numbering allocates through the map's pair rather than carrying a second +// copy of it: one numbering is one walk, and a walk has one allocator. +inline void TableNumberingInit( TableNumbering & n, TableAllocator allocator ) +{ + TablePackMapInit( n.seen, allocator ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +inline void TableNumberingShutdown( TableNumbering & n ) +{ + TableAllocator allocator = n.seen.allocator; + TablePackMapShutdown( n.seen ); + allocator.free( allocator.context, n.entries ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +// The index a numbered node was given, for the save that writes it into a +// pointer slot. False means the two walks disagree about the graph, which is a +// refusal and never a guess. +inline bool TableNumberingIndex( const TableNumbering & n, const void * node, uint64_t & index ) +{ + if ( n.seen.capacity == 0 ) { return false; } + const TablePackEntry & entry = n.seen.entries[ TablePackMapSlot( n.seen, node ) ]; + if ( entry.key != node ) { return false; } + index = (uint64_t) entry.offset; + return true; +} + +inline bool TableNumberingAppend( TableNumbering & n, const TableNodeEntry & entry ) +{ + if ( n.count == n.capacity ) + { + // GROW BY COPY, never by realloc: the allocator hook is a PAIR, and a + // game's heap is not required to have a resize primitive at all. The + // schedule quadruples, so the copying is amortized to a constant per + // entry and the growth is the same growth it always was. + int64_t capacity = n.capacity != 0 ? n.capacity * 4 : 256; + TableAllocator allocator = n.seen.allocator; + TableNodeEntry * grown = (TableNodeEntry *) allocator.alloc( allocator.context, capacity * (int64_t) sizeof( TableNodeEntry ) ); + if ( grown == NULL ) { return false; } + if ( n.entries != NULL ) + { + memcpy( grown, n.entries, (size_t) n.count * sizeof( TableNodeEntry ) ); + allocator.free( allocator.context, n.entries ); + } + n.entries = grown; + n.capacity = capacity; + } + n.entries[n.count++] = entry; + return true; +} + +// The thunks the numbering stores. Each resolves to the closure member's own +// MeasureBody / SaveBodyFields through an overload set in the member's DECLARING +// file, reached by argument-dependent lookup at instantiation — the same bridge +// the arena's TableReset uses, and the reason a numbering may span the files of +// one unit without any file naming another's members. +template +inline int64_t TableNodeMeasureThunk( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ) +{ + return TableNodeMeasure( *(const Ctx *) ctx, numbering, ids, *(const T *) node ); +} + +template +inline bool TableNodeSaveThunk( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ) +{ + return TableNodeSave( *(const Ctx *) ctx, numbering, w, ids, *(const T *) node ); +} + +// ---- a BYTE BUFFER's record (docs/SPEC-TABLES.md §2.5, §3.1) ---- +// +// A blob rides as a node record under one of two RESERVED type ids — the fold +// a table's name takes, over the keywords "bytes" and "string", which no table +// can be named — with the bytes as its body and nothing framed inside. These +// two thunks are what the numbering stores for a blob, as it stores a +// member's codec for a table: the length, and the bytes verbatim. +static const uint64_t kTableBytesTypeId = 0x2f2ec0474f1c4fe4ull; // fnv1a64( "bytes" ) +static const uint64_t kTableStringTypeId = 0x704be0d8faaffc58ull; // fnv1a64( "string" ) + +template +inline int64_t TableBlobMeasureThunk( const void *, const TableNumbering &, TableIds &, const void * node ) +{ + return (int64_t) ( (const TableBlob *) node )->length; +} + +template +inline bool TableBlobSaveThunk( const void *, const TableNumbering &, TableWriter & w, TableIds &, const void * node ) +{ + const TableBlob * blob = (const TableBlob *) node; + w.raw( (const void *) ( blob + 1 ), (int64_t) blob->length ); + return true; +} + +// TableNodeTableMeasure and TableNodeTableSave are the framing, and they are +// ONE fill rule written twice — measure derives it from the graph and save +// derives the same one, which is what makes measure == save hold across a +// pointer graph (§3.1). +// +// The field rides ONCE, under the reserved id, kind 12: the payload opens with +// the count and then carries the records back to back, each a type id +// REFERENCE, a length and a body. The reserved id is interned BEFORE the +// records, and a record's type id before its body, which is the first-use order +// the trailer is written in (§3). +template +inline int64_t TableNodeTablePayload( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + int64_t payload = TableLebBytes( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + payload += TableLebBytes( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return -1; } + payload += TableLebBytes( (uint64_t) body ) + body; + } + return payload; +} + +template +inline int64_t TableNodeTableMeasure( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return 0; } // a root that reaches no nodes writes none of them + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return -1; } + return TableLebBytes( ref ) + 1 + TableLebBytes( (uint64_t) payload ) + payload; +} + +template +inline bool TableNodeTableSave( const Ctx & ctx, TableWriter & w, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return true; } + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return false; } + w.putleb( ref ); + w.put8( 12 ); // kind 12 is the opaque byte payload: a reader that cannot name the id skips by L + w.putleb( (uint64_t) payload ); + w.putleb( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + w.putleb( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return false; } + w.putleb( (uint64_t) body ); + if ( !n.entries[k].save( (const void *) &ctx, n, w, ids, n.entries[k].node ) ) { return false; } + } + return true; +} + +// ---- the numbering, on the LOAD side: a region's NODE DIRECTORY (§6.3) ---- +// +// The wire's numbering made resident: one entry per numbered node, in index +// order, position i describing node index i + 1 — so position 0 is the ROOT at +// offset 0. It is ATTRIBUTION, and attribution is separable: nothing that reads +// a structure touches it, a deref is one add on a self-relative offset, and a +// caller may release it once Load returns. +struct TableNodeDirEntry +{ + uint64_t offset; + uint64_t type_id; +}; + +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; + +// TableNodeMap is what a pointer slot resolves through while a body decodes. +struct TableNodeMap +{ + uint8_t * base = NULL; + const TableNodeDirEntry * entries = NULL; + int64_t count = 0; // the ROOT's entry included, so it is records + 1 + bool good = false; // the node table read whole; a numbering that failed resolves nothing + // WHERE THE NODES LIVE, and therefore what a resolved slot holds: a region + // takes the SELF-RELATIVE delta so a deref is one add, and the tool's + // builder path takes the node's ARENA OFFSET (§6.3). + bool arena = false; + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. + TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; +}; + +// TableNodeResolve places one node index in a pointer slot, and every failure +// is one of §4's events with the pointer left null. The declared TARGET type id +// is checked at every index, the root's included: the root carries no record +// and therefore no wire type id, so the READER'S OWN root type is what the +// claim is checked against. +inline void TableNodeResolve( const TableNodeMap & map, TableRef & slot, uint64_t index, uint64_t target, TableReport * report ) +{ + slot.value = 0; + if ( index == kTableNodeIndexNull || !map.good ) { return; } + if ( index - 1 >= (uint64_t) map.count ) + { + report->malformed = true; // an index above node_count + 1 + return; + } + const TableNodeDirEntry & entry = map.entries[index - 1]; + if ( entry.offset == kTableNodeAbsent ) + { + // a node whose type id this build could not name KEEPS ITS INDEX, and + // every pointer naming it reads null. The unknown was counted once, at + // the node, not once per pointer. + return; + } + if ( entry.type_id != target ) + { + report->kind_mismatch++; + return; + } + slot.value = map.arena ? (int64_t) entry.offset + : (int64_t) ( ( map.base + entry.offset ) - (const uint8_t *) &slot ); +} + +// ---- the record SCAN, and it is the whole of load's bound (§3.1) ---- +// +// Reading follows no reference. The scan walks the root body's top-level fields, +// finds the ONE under the reserved id, and reads records out of its payload in +// order — the field rides once, so nothing is copied to make a body contiguous +// and the generated body decoder never learns the transport exists. +struct TableNodeScan +{ + TableReader fields; // over the ROOT body, skipping past everything else + const uint8_t * payload; // the node-table field's payload + int64_t payload_size; + int64_t payload_offset; + bool opened; // the root body has been walked for the field + uint64_t declared; + int64_t records; + bool present; // the root body carries a node table at all + bool malformed; + const TableIdTable * ids; +}; + +inline TableNodeScan TableNodeScanBegin( const uint8_t * body, int64_t size, TableReport * report, const TableIdTable * ids ) +{ + TableNodeScan s = { TableReader( body, size, report, ids ), NULL, 0, 0, false, 0, 0, false, false, ids }; + return s; +} + +// find the node-table field, or answer false when the root body has none. A +// body carrying an id more than once is legal input and THE LAST OCCURRENCE +// WINS (docs/SPEC-TABLES.md §3), so the walk runs to the terminator and keeps +// the last rather than stopping at the first. +inline bool TableNodeScanOpen( TableNodeScan & s ) +{ + if ( s.opened ) { return false; } + s.opened = true; + for ( ;; ) + { + uint64_t ref = 0; + if ( !s.fields.getleb( ref ) ) { break; } + if ( ref == 0 ) { break; } // the terminator + if ( s.ids == NULL || ref > (uint64_t) s.ids->count ) { break; } + const uint64_t id = s.ids->at( ref ); + if ( !s.fields.has( 1 ) ) { break; } + const uint8_t kind = s.fields.get8(); + if ( id == kTableNodeTableFieldId ) + { + s.present = true; + if ( kind != 12 ) { s.malformed = true; return false; } + uint64_t length = 0; + if ( !s.fields.getleb( length ) || !s.fields.room( length ) ) { s.malformed = true; return false; } + s.payload = s.fields.buffer + s.fields.offset; + s.payload_size = (int64_t) length; + s.fields.offset += (int64_t) length; + continue; + } + if ( !s.fields.skip( kind ) ) { break; } + } + if ( s.payload == NULL ) { return false; } + TableReader head( s.payload, s.payload_size, s.fields.report, s.ids ); + if ( !head.getleb( s.declared ) ) { s.malformed = true; return false; } + s.payload_offset = head.offset; + return true; +} + +// the next record, or false at the end of the table — s.malformed says whether +// the end was the end or the framing giving out +inline bool TableNodeScanNext( TableNodeScan & s, uint64_t & type_id, const uint8_t * & body, int64_t & length ) +{ + if ( !s.opened && !TableNodeScanOpen( s ) ) { return false; } + if ( s.payload == NULL || s.payload_offset >= s.payload_size ) { return false; } + TableReader rec( s.payload, s.payload_size, s.fields.report, s.ids ); + rec.offset = s.payload_offset; + uint64_t ref = 0; + if ( !rec.getleb( ref ) || ref == 0 || s.ids == NULL || ref > (uint64_t) s.ids->count ) + { + s.malformed = true; // a type id reference of 0, or one past the table + return false; + } + type_id = s.ids->at( ref ); + uint64_t declared_length = 0; + if ( !rec.getleb( declared_length ) ) + { + s.malformed = true; // a record whose length is damaged + return false; + } + if ( declared_length > (uint64_t) ( s.payload_size - rec.offset ) ) + { + s.malformed = true; // a record whose length runs past its field + return false; + } + body = s.payload + rec.offset; + length = (int64_t) declared_length; + s.payload_offset = rec.offset + length; + s.records++; + return true; +} + +// The record scan is AUTHORITATIVE: node_count is data from the wire, and a +// count that disagrees with the scan is malformed. Nothing is sized from it +// before the scan has confirmed it. +inline bool TableNodeScanWhole( TableNodeScan & s ) +{ + if ( s.malformed ) { return false; } + if ( !s.present ) { return true; } // no node table at all is not a broken one + return s.declared == (uint64_t) s.records; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_ARENA + +#ifndef LISTDEMO_SCHEMA_TABLE_EXTENT +#define LISTDEMO_SCHEMA_TABLE_EXTENT + +namespace listdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_EXTENT + +#ifndef LISTDEMO_SCHEMA_TABLE_MAP +#define LISTDEMO_SCHEMA_TABLE_MAP + +namespace listdemo { + +// ---- a MAP: a sorted entry array, and the lookup over it (§2.8) ---- +// +// On the wire, in a region and in a cook a map is an array of one generated +// ENTRY table held in ascending key order. What this adds is Find — a binary +// search over that array where it lies — and a builder that inserts, replaces +// and erases by key. Nothing here is stored: a region and a cook carry the +// array and the count, and not one byte about a hash or a probe. + +// entries carved from ONE call to the allocator pair; a new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableMapSegmentEntries = 32; + +// TableDeclRef names a type in an unevaluated context and is never defined — +// what 's declval is for, without the include the generated corpus +// refuses to pay for (the iterator_traits note, §13.9). +template T & TableDeclRef(); + +// THE ORDER IS TOTAL, AND IT IS THE SAME IN NINE LANGUAGES (§2.8). Integers +// compare by VALUE, signed for the signed kinds and unsigned for the unsigned. +// Strings compare by BYTES, unsigned, a shorter string that is a prefix of a +// longer one first: memcmp over the common length, then the lengths. Never a +// locale, never a code point, never a case fold. +inline int TableKeyOrder( uint64_t a, uint64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( int64_t a, int64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( const char * a, int32_t a_length, const char * b, int32_t b_length ) +{ + const int32_t common = a_length < b_length ? a_length : b_length; + if ( common > 0 ) + { + const int order = memcmp( (const void *) a, (const void *) b, (size_t) common ); + if ( order != 0 ) { return order < 0 ? -1 : 1; } + } + return a_length < b_length ? -1 : ( a_length > b_length ? 1 : 0 ); +} + +// the length of a NUL-terminated key at a call site, bounded by the storage it +// has to fit: a key one byte longer than the bound is refused, never truncated +inline int32_t TableKeyLength( const char * key, int32_t bound ) +{ + if ( key == NULL ) { return 0; } + for ( int32_t i = 0; i <= bound; i++ ) { if ( key[i] == 0 ) { return i; } } + return bound + 1; // longer than the bound: the caller refuses it +} + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.8, §7.2) ---- +// +// An int64 self-relative reference to the entry array and an int32 count, then +// padding to eight. The reference is a TableRef like a pointer's: in the arena +// it names the builder's HEAD, in a region it is the delta from the slot to +// the first entry, and 0 is the empty map in both. +template struct TableMap +{ + TableRef entries; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Entry * Entries() const + { + return entries.value != 0 ? (const Entry *) ( (const uint8_t *) &entries + entries.value ) : NULL; + } + int32_t size() const { return count; } + + // FIND: floor( log2 n ) + 1 key compares, in place, no allocation. NULL + // when absent, and on a map[K]*T the RESOLVED pointer, which is what a + // pointer field's accessor answers. + template const Entry * FindEntry( Key key ) const + { + const Entry * base = Entries(); + int32_t low = 0, high = count; + while ( low < high ) + { + const int32_t mid = low + ( high - low ) / 2; + const int order = TableEntryOrder( base[mid], key ); + if ( order == 0 ) { return base + mid; } + if ( order < 0 ) { low = mid + 1; } else { high = mid; } + } + return NULL; + } + // the return type is DEDUCED, so it is worked out when a call site + // instantiates Find and not when the holder's record declares the slot — + // which is what lets the entry's own overloads be declared after it + template auto Find( Key key ) const + { + return TableEntryFound( FindEntry( key ) ); + } + + // ---- iteration: ASCENDING key order, the key beside the value ---- + // + // A proxy BY VALUE, the keyed array's shape (§2.4): for ( auto [ key, + // value ] : map ). It carries no iterator_traits, for the reason + // TableKeyed's does not (§13.9). + struct ConstEntry + { + decltype( TableEntryKey( TableDeclRef() ) ) key; + decltype( TableEntryFound( (const Entry *) NULL ) ) value; + }; + + struct ConstIterator + { + const Entry * at; + ConstEntry operator*() const { return ConstEntry{ TableEntryKey( *at ), TableEntryFound( at ) }; } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Entries() }; } + ConstIterator end() const { return ConstIterator{ Entries() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.8, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first entry is inserted. Each +// segment is a fixed number of entries carved from one call to the allocator +// pair. An entry's address is stable for the arena's life, so a value handed +// back by an insert stays valid while other entries arrive. +struct TableMapHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an insert appends into + int32_t live; + int32_t dead; +}; + +template struct TableMapSegment +{ + TableRef next; + int32_t used; // entries carved from this segment + int32_t padding; + uint32_t dead[ ( kTableMapSegmentEntries + 31 ) / 32 ]; // Erase marks one bit, never the entry + Entry entries[ kTableMapSegmentEntries ]; +}; + +inline bool TableMapSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// ---- the ORDERED CURSOR the four writing walks read (§2.8) ---- +// +// Measure, Save, Lock and Cook each write a map's entries in ascending key +// order with no key twice, deriving the order from the builder's entries as +// each walk derives the numbering (§3.1). Nothing passes between them, so +// measure == save over a map is a real check on two sorts agreeing. +// +// A REGION is already sorted, so its cursor is the array in place and +// allocates nothing. The BUILDER's is the sort: an array of entry pointers +// allocated through the pair and released before the walk returns, because +// sorting the segments themselves would move entries whose addresses a caller +// holds. +template struct TableMapCursor +{ + const Entry * const * order = NULL; // the builder's form: sorted pointers + const Entry * entries = NULL; // the region's form: the array in place + int32_t count = 0; + TableAllocator allocator; + bool ok = false; + const Entry * operator[]( int32_t index ) const + { + return order != NULL ? order[index] : entries + index; + } +}; + +// heapsort: O( n log n ) once per map, no recursion, no allocation past the +// pointer array the caller already paid for +template inline void TableMapSort( const Entry ** order, int32_t count ) +{ + for ( int32_t start = count / 2 - 1; start >= 0; start-- ) + { + int32_t root = start; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= count ) { break; } + if ( child + 1 < count && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * swap = order[root]; order[root] = order[child]; order[child] = swap; + root = child; + } + } + for ( int32_t end = count - 1; end > 0; end-- ) + { + const Entry * swap = order[0]; order[0] = order[end]; order[end] = swap; + int32_t root = 0; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= end ) { break; } + if ( child + 1 < end && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * hold = order[root]; order[root] = order[child]; order[child] = hold; + root = child; + } + } +} + +// the REGION form: the array is already sorted, so the cursor is the array +template +inline TableMapCursor TableMapOrder( const TableRegionCtx &, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.entries = map.Entries(); + cursor.count = map.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: gather the LIVE entries out of the segment chain in +// insertion order, then sort. A dead entry costs nothing on any wire (§2.8). +template +inline TableMapCursor TableMapOrder( const TableArena & arena, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.allocator = arena.allocator; + cursor.count = map.count; + if ( map.entries.value == 0 || map.count <= 0 ) { cursor.ok = map.count == 0; cursor.count = 0; return cursor; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + if ( head->live != map.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + const Entry ** order = (const Entry **) arena.allocator.alloc( arena.allocator.context, (int64_t) map.count * (int64_t) sizeof( const Entry * ) ); + if ( order == NULL ) { return cursor; } + int32_t at = 0; + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 && at < map.count ) + { + const TableMapSegment * segment = (const TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used && at < map.count; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + order[at++] = segment->entries + i; + } + segment_ref = segment->next; + } + if ( at != map.count ) + { + arena.allocator.free( arena.allocator.context, order ); + return cursor; + } + TableMapSort( order, map.count ); + cursor.order = order; + cursor.ok = true; + return cursor; +} + +template +inline TableMapCursor TableMapOrder( const TableArenaCtx & ctx, const TableMap & map ) +{ + return TableMapOrder( *ctx.arena, map ); +} + +template inline void TableMapRelease( TableMapCursor & cursor ) +{ + if ( cursor.order != NULL ) { cursor.allocator.free( cursor.allocator.context, (void *) cursor.order ); } + cursor.order = NULL; +} + +// ---- the builder's five (§2.8) ---- +// +// Insert APPENDS after one LINEAR SCAN of the live entries for the key it may +// replace, Find is that same scan, and Erase is the scan and one bit. The +// builder builds NO INDEX, and that is a rule: the sort happens once, at Lock, +// Save or Cook, and every lookup that matters runs over the sorted region. + +// the head, allocated when the first entry is inserted +template +inline TableMapHead * TableMapReach( TableWorker & worker, TableMap & map ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( map.entries.value != 0 ) { return (TableMapHead *) TableArenaAt( *worker.arena, (uint32_t) map.entries.value ); } + uint32_t at = 0; + TableMapHead * head = (TableMapHead *) worker.AllocRaw( (int64_t) sizeof( TableMapHead ), (int64_t) alignof( TableMapHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + map.entries.value = (int64_t) at; + return head; +} + +// one entry's storage, appended: the current segment when it has room, a new +// one carved from one call to the pair when it does not +template +inline Entry * TableMapAppend( TableWorker & worker, TableMapHead * head, TableMap & map ) +{ + TableMapSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableMapSegmentEntries ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableMapSegment *) worker.AllocRaw( (int64_t) sizeof( TableMapSegment ), (int64_t) alignof( TableMapSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableMapSegment * previous = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Entry * entry = segment->entries + segment->used; + segment->used++; + head->live++; + map.count++; + return entry; +} + +// the LINEAR SCAN: the live entries in insertion order, O( n ) key compares +template +inline Entry * TableMapScan( const TableArena & arena, const TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) == 0 ) { return segment->entries + i; } + } + segment_ref = segment->next; + } + return NULL; +} + +// ERASE marks the entry DEAD, one bit in the segment's slot and not in the +// entry table, and decrements the live count. Its storage is reclaimed at +// RESET and never reused mid-build, because reusing a slot would make "an +// entry's address is stable" false for exactly one case. +template +inline bool TableMapErase( TableArena & arena, TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return false; } + TableMapHead * head = (TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) != 0 ) { continue; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + map.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INSERTION order, live entries only (§2.8) ---- +template struct TableMapEach +{ + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableMapSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableMapSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + auto operator*() const { return TableEntryEach( segment->entries + index ); } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableMapEach TableMapEachOf( const TableArena & arena, const TableMap & map ) +{ + TableMapEach each = { &arena, TableRef() }; + if ( map.entries.value != 0 ) + { + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + each.first = head->first; + } + return each; +} + +// ---- the LOAD side: where a decoded entry lands (§2.8) ---- +// +// THE READER TRUSTS NOTHING and spends one compare per entry. Every load path +// applies the same rules and produces one report (§4), so the region load of +// §6.5 and LoadBuilder never disagree about a wire. These two shapes are what +// makes that true with one generated decoder: a REGION carves the entry array +// out of the holder node's own extent, and the TOOL's path appends into the +// builder's arena, and the decoder above them cannot tell which it has. + +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. + +// TableMapFill is one map field being decoded: where the next entry lands, and +// the entry that last LANDED, which is what the ascending check compares +// against. +template struct TableMapFill +{ + TableMap * map = NULL; + Entry * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; +}; + +template +inline TableMapFill TableMapFillBegin( const TableNodeMap & nodes, TableMap & map, uint32_t n ) +{ + TableMapFill fill; + fill.map = ↦ + map.entries.value = 0; + map.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Entry ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Entry ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Entry *) base; + fill.capacity = (int32_t) n; + map.entries.value = (int64_t) ( base - (const uint8_t *) &map.entries ); + fill.ok = true; + return fill; +} + +// the entry that last LANDED — NULL before the first +template inline Entry * TableMapFillLast( TableMapFill & fill ) +{ + if ( fill.map->count <= 0 ) { return NULL; } + if ( fill.array != NULL ) { return fill.array + ( fill.map->count - 1 ); } + return TableMapLive( *fill.worker->arena, *fill.map, fill.map->count - 1 ); +} + +// the next slot, at the entry type's declared defaults +template inline Entry * TableMapFillNext( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + if ( fill.map->count >= fill.capacity ) { return NULL; } + Entry * entry = fill.array + fill.map->count; + TableReset( *entry ); + fill.map->count++; + return entry; + } + TableMapHead * head = TableMapReach( *fill.worker, *fill.map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( *fill.worker, head, *fill.map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// A MAP WITH HALF ITS KEYS IS NOT A MAP (§2.8): at the first entry whose key +// kind disagrees with the reader's declaration the map resets to EMPTY, one +// kind_mismatch is counted for the map, and its remaining bytes are skipped. +template inline void TableMapFillReset( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + fill.map->entries.value = 0; + fill.map->count = 0; + return; + } + if ( fill.map->entries.value != 0 ) + { + TableMapHead * head = (TableMapHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.map->entries.value ); + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + } + fill.map->count = 0; +} + +// an EMPTY map's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableMapFillEnd( TableMapFill & fill ) +{ + if ( fill.array != NULL && fill.map->count == 0 ) { fill.map->entries.value = 0; } +} + +// the k-th LIVE entry of a builder map, in insertion order — what the tool +// path's ascending check compares against +template +inline Entry * TableMapLive( const TableArena & arena, const TableMap & map, int32_t index ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + int32_t at = 0; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( at == index ) { return segment->entries + i; } + at++; + } + segment_ref = segment->next; + } + return NULL; +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.8, §6.5) ---- +// +// LoadMeasure's term for a map is N x sizeof( Entry ) rounded to +// alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this +// reads no field: it walks the map's own header and, where an entry's value +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. +// A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its +// own L and the body's terminator, and under this form's variable lengths that +// footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a +// map's L can carry, and therefore what a LoadMeasure may be asked for. +static const int64_t kTableMapEntryFloor = 2; + +inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry + at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); + at += (int64_t) n * entry_size; + if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// ---- the TEXT form's placement (docs/SPEC-TABLES.md §2.8, §16) ---- +// +// The text is a plain JSON object keyed by the KEY, and the generic walk fills +// it through the ENTRY'S OWN descriptor — so all it needs from here is one +// entry at one key, handed back at its defaults. It is the builder's Insert +// with the ENTRY returned rather than its value, because the walk writes the +// value through a field row and not through a typed pointer. +template +inline Entry * TableMapPlace( TableWorker & worker, TableMap & map, Key key ) +{ + if ( worker.arena == NULL ) { return NULL; } + Entry * found = TableMapScan( *worker.arena, map, key ); + if ( found != NULL ) + { + TableResetMapValue( *found ); // a repeated key is LAST-WINS, whole + return found; + } + TableMapHead * head = TableMapReach( worker, map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( worker, head, map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// ---- the OPTIONAL RUNTIME INDEX (§2.8) ---- +// +// Open addressing with LINEAR PROBING over the sorted array, built AT LOAD for +// a map large enough that log n compares over a cold array cost more than one +// hash and a probe. IT IS NEVER STORED: the caller measures it, owns its +// storage, builds it in one pass and releases it whenever. +// +// ITS HASH AND ITS LOAD FACTOR ARE NOT A CROSS-PORT CONTRACT, and that is a +// rule. What a port is held to is the CONTRACT of the lookup: the same value +// the sorted array's Find returns for the same key, and no allocation past the +// storage the caller handed in. +struct TableMapIndex +{ + int32_t * slots = NULL; // entry indices, +1; 0 is an empty slot + int32_t capacity = 0; + bool good = false; +}; + +// this runtime's own, and no port reproduces it: fnv1a64 over the key's bytes +inline uint64_t TableMapHash( const void * bytes, int32_t length ) +{ + uint64_t hash = 0xCBF29CE484222325ull; + const uint8_t * at = (const uint8_t *) bytes; + for ( int32_t i = 0; i < length; i++ ) { hash ^= (uint64_t) at[i]; hash *= 0x100000001B3ull; } + return hash; +} +inline uint64_t TableMapHash( uint64_t key ) { return TableMapHash( (const void *) &key, (int32_t) sizeof( key ) ); } + +// this runtime's own load factor, and no port reproduces it either: the next +// power of two at or above twice the count, so a probe run stays short +inline int32_t TableMapIndexSlots( int32_t count ) +{ + int32_t slots = 8; + while ( slots < count * 2 ) { slots *= 2; } + return slots; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_MAP + +#ifndef LISTDEMO_SCHEMA_TABLE_LIST +#define LISTDEMO_SCHEMA_TABLE_LIST + +namespace listdemo { + +// ---- an UNBOUNDED ARRAY: a counted array whose count the data decides (§2.9) ---- +// +// On the wire, in a region and in a cook a list is the kind 14 body a [..N]T +// writes, its elements by-value records inside the holder's node extent. What +// this adds is the slot, a builder that appends into segments that never +// move, and a const surface that indexes and iterates in place. There is no +// sort, no key and no lookup: the order is INSERTION order, and it is +// identity the way position is identity in a fixed array. + +// elements carved from ONE call to the allocator pair. A new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableListSegmentElements = 32; + +// THE ELEMENT STORAGE: T itself, and a TableRef slot for a []*T, whose +// elements are references exactly as a pointer field's slot is (§2.1) +template struct TableListStorage { typedef T Element; }; +template struct TableListStorage { typedef TableRef Element; }; + +// WHAT THE CONST FORM ANSWERS: the element by reference, and on a []*T the +// RESOLVED pointer, one add on the self-relative delta, NULL for a null slot, +// exactly as At answers it (§6.2, §6.3) +template struct TableListConst +{ + typedef const T & Result; + static Result At( const T * element ) { return *element; } +}; +template struct TableListConst +{ + typedef const T * Result; + static Result At( const TableRef * element ) + { + return element->value != 0 ? (const T *) ( (const uint8_t *) element + element->value ) : NULL; + } +}; + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.9, §7.2) ---- +// +// An int64 self-relative reference to the element array and an int32 count, +// then padding to eight. The reference is a TableRef like a pointer's: in the +// arena it names the builder's HEAD, in a region it is the delta from the slot +// to the first element, and 0 is the empty list in both. It is the map's slot +// exactly, because it is the same two facts. +template struct TableList +{ + typedef typename TableListStorage::Element Element; + + TableRef elements; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Element * Elements() const + { + return elements.value != 0 ? (const Element *) ( (const uint8_t *) &elements + elements.value ) : NULL; + } + int32_t size() const { return count; } + + // INDEXING IS BOUNDS-CHECKED IN EVERY BUILD (§2.4, §2.9): the extent is a + // number that CAME FROM A FILE, so an index past it is not a mistake a + // release build gets to make cheaply. There is no undefined-behavior path + // here in any configuration. The assert carries the message where a + // debugger can read it and NDEBUG removes that. The fatal is what stands + // after it. Both go through the hooks: define schema_assert and + // schema_fatal and this refusal lands in your own handler. + void RefuseIndex( int32_t index ) const + { + if ( (uint32_t) index >= (uint32_t) count ) + { + schema_assert( false && "an unbounded array is indexed inside its count, which came from a file" ); + schema_fatal(); + } + } + typename TableListConst::Result operator[]( int32_t index ) const + { + RefuseIndex( index ); + return TableListConst::At( Elements() + index ); + } + + // ---- iteration: INDEX order, the element and no key ---- + // + // It carries no iterator_traits, for the reason TableKeyed's does not + // (§13.9). + struct ConstIterator + { + const Element * at; + typename TableListConst::Result operator*() const { return TableListConst::At( at ); } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Elements() }; } + ConstIterator end() const { return ConstIterator{ Elements() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.9, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first element is added. Each +// segment is a fixed number of elements carved from one call to the allocator +// pair. An element's address is stable for the arena's life, so a T * handed +// back by Add stays valid while other elements arrive. +struct TableListHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an Add appends into + int32_t live; + int32_t dead; +}; + +template struct TableListSegment +{ + TableRef next; + int32_t used; // elements carved from this segment + int32_t padding; + uint32_t dead[ ( kTableListSegmentElements + 31 ) / 32 ]; // Erase marks one bit, never the element + Element elements[ kTableListSegmentElements ]; +}; + +inline bool TableListSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// the head, allocated when the first element is added +template +inline TableListHead * TableListReach( TableWorker & worker, TableList & list ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( list.elements.value != 0 ) { return (TableListHead *) TableArenaAt( *worker.arena, (uint32_t) list.elements.value ); } + uint32_t at = 0; + TableListHead * head = (TableListHead *) worker.AllocRaw( (int64_t) sizeof( TableListHead ), (int64_t) alignof( TableListHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + list.elements.value = (int64_t) at; + return head; +} + +// one element's storage, appended: the current segment when it has room, a +// new one carved from one call to the pair when it does not. NULL means NOT +// ADDED: an arena that cannot carve another segment, or a count at the int32 +// cap (§2.2, §2.9). +template +inline typename TableList::Element * TableListAppend( TableWorker & worker, TableListHead * head, TableList & list ) +{ + typedef typename TableList::Element Element; + if ( list.count >= INT32_MAX ) { return NULL; } // the int32 storage cap + TableListSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableListSegmentElements ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableListSegment *) worker.AllocRaw( (int64_t) sizeof( TableListSegment ), (int64_t) alignof( TableListSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableListSegment * previous = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Element * element = segment->elements + segment->used; + segment->used++; + head->live++; + list.count++; + return element; +} + +// ADD, whole: the head, the append, and the element at its declared defaults +// (§2.9). The text form's placement is this same call, because a list has no +// key to place under (§16). +template +inline typename TableList::Element * TableListPlace( TableWorker & worker, TableList & list ) +{ + typedef typename TableList::Element Element; + TableListHead * head = TableListReach( worker, list ); + if ( head == NULL ) { return NULL; } + Element * element = TableListAppend( worker, head, list ); + if ( element == NULL ) { return NULL; } + new ( element ) Element(); // value-init: the declared defaults, and null for a slot + return element; +} + +// ERASE, ADDRESSED BY THE POINTER (§2.9): the element Add handed back is the +// handle, because a list has no key and the address is the one thing the +// builder promises never moves (§6.4). It marks the element DEAD, one bit in +// the segment's slot and not in the element storage, and decrements the live +// count. False when the pointer is not this list's. Its storage is reclaimed +// at RESET and never reused mid-build, the map's rule for the map's reason. +template +inline bool TableListErase( TableArena & arena, TableList & list, const typename TableList::Element * element ) +{ + typedef typename TableList::Element Element; + if ( list.elements.value == 0 || element == NULL ) { return false; } + TableListHead * head = (TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableListSegment * segment = (TableListSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + if ( element >= segment->elements && element < segment->elements + segment->used ) + { + const int32_t i = (int32_t) ( element - segment->elements ); + if ( TableListSegmentDead( segment->dead, i ) ) { return false; } // already erased + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + list.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INDEX order, live elements only (§2.9) ---- +template struct TableListEach +{ + typedef typename TableList::Element Element; + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableListSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableListSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + Element * operator*() const { return segment->elements + index; } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableListEach TableListEachOf( const TableArena & arena, const TableList & list ) +{ + TableListEach each = { &arena, TableRef() }; + if ( list.elements.value != 0 ) + { + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + each.first = head->first; + } + return each; +} + +// ---- the INDEX-ORDER CURSOR the four writing walks read (§2.9) ---- +// +// Measure, Save, Lock and Cook each visit a list's live elements in the order +// they were added, and they allocate nothing to do it: a region's cursor is +// the array in place, and the builder's walks the segment chain. Indexing the +// builder's form is SEQUENTIAL by construction, every walk steps i, i + 1, +// i + 2, so the cursor remembers where the last access landed and moves one +// live slot per step. An access behind the memo restarts from the first +// segment, which no walk here does. +template struct TableListCursor +{ + const Element * elements = NULL; // the region's form: the array in place + const TableArena * arena = NULL; // the builder's form: the segments + TableRef first; + int32_t count = 0; + bool ok = false; + // the memo: the segment and slot the last access landed on, and the live + // index that slot holds + mutable const TableListSegment * segment = NULL; + mutable int32_t within = -1; + mutable int32_t logical = -1; + + const Element * At( int32_t index ) const + { + if ( elements != NULL ) { return elements + index; } + if ( segment == NULL || index < logical ) + { + segment = first.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL; + within = -1; + logical = -1; + } + while ( logical < index ) + { + for ( ;; ) + { + within++; + while ( segment != NULL && within >= segment->used ) + { + segment = segment->next.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + within = 0; + } + if ( segment == NULL ) { return NULL; } // the slot and the head disagree + if ( !TableListSegmentDead( segment->dead, within ) ) { break; } + } + logical++; + } + return segment->elements + within; + } + const Element & operator[]( int32_t index ) const { return *At( index ); } +}; + +// the REGION form: the array is the cursor +template +inline TableListCursor::Element> TableListElements( const TableRegionCtx &, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.elements = list.Elements(); + cursor.count = list.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: the live elements out of the segment chain, in the +// order they were added. A dead element costs nothing on any wire (§2.9). +template +inline TableListCursor::Element> TableListElements( const TableArena & arena, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.arena = &arena; + cursor.count = list.count; + if ( list.elements.value == 0 || list.count <= 0 ) { cursor.ok = list.count == 0; cursor.count = 0; return cursor; } + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + if ( head->live != list.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + cursor.first = head->first; + cursor.ok = true; + return cursor; +} + +template +inline TableListCursor::Element> TableListElements( const TableArenaCtx & ctx, const TableList & list ) +{ + return TableListElements( *ctx.arena, list ); +} + +// ---- the LOAD side: where a decoded element lands (§2.9) ---- +// +// The same two shapes the map's fill takes, because the decoder above them +// cannot tell which it has: a REGION carves the element array out of the +// holder node's own extent, PRE-ORDER, and the TOOL's path appends into the +// builder's arena. +template struct TableListFill +{ + typedef typename TableList::Element Element; + TableList * list = NULL; + Element * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; + bool refused = false; // a count above the int32 cap on the tool's path: LoadBuilder answers NULL +}; + +template +inline TableListFill TableListFillBegin( const TableNodeMap & nodes, TableList & list, uint64_t n ) +{ + typedef typename TableList::Element Element; + TableListFill fill; + fill.list = &list; + list.elements.value = 0; + list.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( n > (uint64_t) INT32_MAX ) + { + // A COUNT ABOVE THE int32 STORAGE CAP (§2.2, §2.9): into a region it was + // refused by LoadMeasure before this ran, and into a builder it is the + // refusal LoadBuilder answers NULL for, moving no counter + fill.refused = nodes.carve->worker != NULL; + return fill; + } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Element ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Element ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Element *) base; + fill.capacity = (int32_t) n; + list.elements.value = (int64_t) ( base - (const uint8_t *) &list.elements ); + fill.ok = true; + return fill; +} + +// the next slot, at the element's declared defaults. NULL when the arena +// could not carve, which the decoder reports as framing damage +template inline typename TableList::Element * TableListFillNext( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count >= fill.capacity ) { return NULL; } + Element * element = fill.array + fill.list->count; + new ( element ) Element(); + fill.list->count++; + return element; + } + return TableListPlace( *fill.worker, *fill.list ); +} + +// A SLOT WHOSE ELEMENT NEVER LANDED is given back (§2.9, §4): the array keeps +// what it decoded, and an element whose own framing gave out before one byte +// of it decoded was not decoded. The region's form uncounts it, and the builder's +// marks it dead, which is what the storage rule allows mid-build. +template inline void TableListFillDrop( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count > 0 ) { fill.list->count--; } + return; + } + if ( fill.list->elements.value == 0 ) { return; } + TableListHead * head = (TableListHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.list->elements.value ); + if ( head->last.value == 0 ) { return; } + TableListSegment * segment = (TableListSegment *) TableArenaAt( *fill.worker->arena, (uint32_t) head->last.value ); + if ( segment->used <= 0 ) { return; } + const int32_t i = segment->used - 1; + if ( TableListSegmentDead( segment->dead, i ) ) { return; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + fill.list->count--; +} + +// an EMPTY list's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableListFillEnd( TableListFill & fill ) +{ + if ( fill.array != NULL && fill.list->count == 0 ) { fill.list->elements.value = 0; } +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.9, §6.5) ---- +// +// N x sizeof( T ) rounded to alignof( T ), AT EVERY DEPTH. N is framing and +// not a value, so this reads no field: it walks the list's own header and, +// where a table element holds a list or a map of its own, the elements' +// headers under it. Every -1 carries its REASON (§6.5): the int32 cap first, +// because a count past it cannot fit any body, and then the body's own L. +inline bool TableListWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t elem_size, int64_t elem_align, uint8_t elem_kind, int64_t elem_floor, + TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != elem_kind ) { return true; } // another element kind: §4's ordinary kind mismatch, the field reads empty + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; } // an N the list's L cannot carry + at = ( at + elem_align - 1 ) & ~( elem_align - 1 ); + at += (int64_t) n * elem_size; + if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_LIST + +#ifndef LISTDEMO_SCHEMA_BUILD_VERSION +#define LISTDEMO_SCHEMA_BUILD_VERSION + +namespace listdemo { + +// THE BUILD VERSION (docs/SPEC-TABLES.md §20): one digest over every fact the bytes +// this build produces depend on — the type wire's protocol id, every record's +// layout as the compiler's own C ABI model computes it, and the facts that +// decide what a load PUTS in those slots. It is the number a cook's header +// carries and the number Open compares, and the number a block's prologue +// carries and BlockOpen compares: a build version answers "which build?" and +// not "which form?", and what separates the two forms is their MAGIC. +// +// There are TWO ids in the design and they are not interchangeable: the +// PROTOCOL ID is the type wire's and nothing else, and the BUILD VERSION is +// what everything cooked or blocked is keyed by. A table edit moves this and +// never the protocol id; a type edit moves both. +static const uint64_t BuildVersion = 0x8d7c0edaca4571c7ull; + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_BUILD_VERSION + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK +#define LISTDEMO_SCHEMA_TABLE_COOK + +namespace listdemo { + +// ---- the cooked form (docs/SPEC-TABLES.md §7) ---- +// +// A cooked file is a HEADER, a DATA part and an ATTRIBUTION part, in that +// order. Every word of the header is a u64 written in the byte order the cook +// was produced in, and the header is 64 bytes: +// +// 0 magic 0x4b4f4f434d484353, read BYTEWISE before anything else +// 8 build_version the unit's id (docs/SPEC-TABLES.md §20) +// 16 byte_order 1 little, 2 big — the order that WROTE the file +// 24 data_length the region's bytes, rounded up to alignment +// 32 attribution_length the directory's bytes, or 0 +// 40 alignment the region's alignment, never below eight +// 48 reserved zero +// 56 reserved zero +// +// The DATA part is Lock's region written verbatim (§7.2) — the root at its +// base — and it is what a runtime points at. The ATTRIBUTION part is the node +// directory (§6.3), and NOTHING THAT READS THE STRUCTURE TOUCHES IT: it is +// written beside the data for schema cook-check, so a build that ships no +// tooling need not carry it at all. +static const int64_t kTableCookHeaderBytes = 64; + +// THE MAGIC'S VALUE, and a consumer written from the page needs the constant +// rather than a description of one. It is "SCHMCOOK" read as ASCII in the byte +// order a little-endian store produces — the same shape the block form's +// SCHMABLK takes, so a hex dump of a little-endian cook is legible and the two +// accelerators sit in one vocabulary. +// +// IT IS STORED IN THE PRODUCER'S ORDER, which is what makes it the byte-order +// check as well as the form check: a consumer reads back this build's +// constant, or that constant byte-reversed — which identifies a cook of the +// OTHER order — or something that is not a cook. All three answers but the +// first refuse, and a cook and a BLOCK are separated here too, because a +// form's identity belongs in its magic rather than in a second digest. +static const uint64_t TableCookMagic = 0x4b4f4f434d484353ull; + +// THIS BUILD's byte order, as the header's own word carries it. The magic is +// what REFUSES a foreign order; this word is what RECORDS which order wrote +// the file, so a refusal names the order rather than inferring it and a tool +// dumping a cook reads the fact. A file whose magic matched and whose order +// word did not is corrupt, and there is no reading that recovers it. +// +// The BUILD VERSION cannot do either job: §20.1 digests byteorder as a +// GENERATION input, little for every target schema generates for today, so +// two builds of one schema for two orders emit the same id. +#if defined( __BYTE_ORDER__ ) && defined( __ORDER_BIG_ENDIAN__ ) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__ +static const uint64_t TableCookByteOrder = 2; // big +#else +static const uint64_t TableCookByteOrder = 1; // little +#endif + +// The greatest region alignment a cooked file may name. The DATA part begins +// at align_up( 64, alignment ), which is 64 for every unit this language can +// declare — the largest alignment it has is sixteen — so a word past this cap +// describes a file no build of this schema wrote (docs/SPEC-TABLES.md §7.1). +static const uint64_t TableCookMaxAlign = 64; + +// The header read, BYTEWISE. memcpy is the portable spelling of "these eight +// bytes, in this machine's order"; every compiler this repo builds under folds +// it to one load, and it is the only read in the whole of Open that is not a +// comparison. +inline uint64_t table_cook_read64( const uint8_t * p ) +{ + uint64_t v; + memcpy( &v, p, sizeof( v ) ); + return v; +} + +// TableCookOpen: THE WHOLE CHECK, in one place, because §7 states the +// enumeration once and every generated Open is that one enumeration plus +// its own root's two layout facts. +// +// THE CHECK, in order: the magic read bytewise, the byte order it establishes, +// the build version against this build's own, both RESERVED words zero, the +// region alignment the header names, the two part lengths against the length +// the caller passed — a truncated file and a file with trailing bytes are the +// same refusal — the root's own storage inside the data part, and the +// alignment of the base. +// +// AND THAT IS ALL OF IT. On a match the bytes ARE what this build wrote, in +// this build's layout and this build's byte order, so there is nothing to +// validate and nothing to fix up: the caller gets the root. Nothing per node +// happens here, which is what makes open O(1) in the file's size; a walk of +// any shape would forfeit that, and validating an untrusted file is schema +// cook-check's job and a person's decision (§7.4). +// +// EVERY NUMBER BELOW COMES OUT OF THE FILE, so the arithmetic is unsigned and +// each term is BOUNDED BEFORE IT IS ADDED: a forged length near 2^64 must +// refuse, and an addition that wrapped would be the defect the comparison +// after it was supposed to catch. Nothing past length is read on any path, +// including every refusing one. +inline const uint8_t * TableCookOpen( const void * bytes, uint64_t length, uint64_t root_size, uint64_t root_align ) +{ + if ( bytes == NULL ) { return NULL; } + if ( length < (uint64_t) kTableCookHeaderBytes ) { return NULL; } + const uint8_t * raw = (const uint8_t *) bytes; + // the MAGIC, bytewise and first: it is what establishes the byte order + // every other header word is read in, so nothing else may be read before + // it. A byte-reversed constant is a cook of the other order and refuses + // here, which is why the order never reaches a fix-up pass. + if ( table_cook_read64( raw ) != TableCookMagic ) { return NULL; } + if ( table_cook_read64( raw + 16 ) != TableCookByteOrder ) { return NULL; } + if ( table_cook_read64( raw + 8 ) != BuildVersion ) { return NULL; } + // the RESERVED words: a non-zero one means a writer used a form this build + // does not understand, and Open refuses rather than ignoring it. + if ( table_cook_read64( raw + 48 ) != 0 ) { return NULL; } + if ( table_cook_read64( raw + 56 ) != 0 ) { return NULL; } + const uint64_t data_length = table_cook_read64( raw + 24 ); + const uint64_t attribution_length = table_cook_read64( raw + 32 ); + const uint64_t alignment = table_cook_read64( raw + 40 ); + // THE ALIGNMENT WORD IS DATA, and it is the one header field the rest of + // the check does arithmetic WITH rather than only comparison against. A + // region's alignment is a power of two, never below eight (the floor that + // puts the attribution part on an eight-byte boundary without a second + // padding rule) and never past the cap above; a word that is none of those + // rounds nothing and aligns nothing, so it is refused before it is used. + if ( alignment < 8 || alignment > TableCookMaxAlign ) { return NULL; } + if ( ( alignment & ( alignment - 1 ) ) != 0 ) { return NULL; } + // and it must be an alignment THE ROOT CAN SIT AT, since the root is at + // the region's base: both are powers of two, so "at least the root's" + // is one division. + if ( ( alignment % root_align ) != 0 ) { return NULL; } + // The DATA part begins at align_up( 64, alignment ). It is DERIVED and not + // a header field, because a fact a reader computes is a fact two writers + // cannot disagree about. + const uint64_t data_offset = ( (uint64_t) kTableCookHeaderBytes + alignment - 1 ) & ~( alignment - 1 ); + if ( length < data_offset ) { return NULL; } + // the two part lengths against the length the caller passed. The whole + // file is data_offset + data_length + attribution_length, and a length + // that is not EXACTLY that refuses — truncation and trailing bytes are one + // refusal, and both terms are subtracted rather than added so no sum can + // carry. + if ( data_length > length - data_offset ) { return NULL; } + if ( attribution_length != length - data_offset - data_length ) { return NULL; } + // the ROOT sits at the region's base, so the region has to hold it: a + // shorter data part describes a root partly outside the file, which is the + // one way a match-and-point reader could hand back storage it never + // received. + if ( data_length < root_size ) { return NULL; } + const uint8_t * base = raw + data_offset; + // the alignment of the BASE. The header pads the data part to the region's + // alignment, so a base an allocator or mmap gave you is already aligned — + // mmap gives page alignment for free — and a base that is not is a caller's + // buffer this form cannot be read out of. + if ( ( (uintptr_t) base % (uintptr_t) alignment ) != 0 ) { return NULL; } + return base; +} + +// ---- the cooked form, the WRITE side (docs/SPEC-TABLES.md §7.6) ---- +// +// THE BYTE ORDER IS THE TARGET'S, NOT THE HOST'S. A cook is produced in the +// byte order of the build that will read it (§7), so the fixing happens here — +// offline, once, on the writing side — and never at Open. Passing +// TableByteOrder::Big on a little-endian machine produces a big-endian build's +// file, and nothing about the writing host reaches the bytes. +enum class TableByteOrder +{ + Little = 1, // the header's byte_order word, and the order every scalar is written in + Big = 2, +}; + +// One store, width as an argument. Every call site passes a literal width, so +// the loop folds to a store (and a byte swap on the foreign order); a name per +// width would claim four §11 names to save nothing. +inline void table_cook_put( uint8_t * at, uint64_t value, int32_t width, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * i ) ); } + } + else + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * ( width - 1 - i ) ) ); } + } +} + +// A 128-bit store as two lanes: sixteen bytes, the low lane first in the +// little order and the high lane first — each lane big-endian — in the big +// order, exactly as a u64 is one lane of eight (docs/SPEC-TABLES.md §7.2). +inline void table_cook_put128( uint8_t * at, uint64_t lo, uint64_t hi, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) { table_cook_put( at, lo, 8, order ); table_cook_put( at + 8, hi, 8, order ); } + else { table_cook_put( at, hi, 8, order ); table_cook_put( at + 8, lo, 8, order ); } +} + +// A buffer piece: the USED bytes and nothing else. The tail is already zero — +// the whole extent was zeroed before any field was written — so this copies the +// used prefix and leaves the rest, which is what makes a string's unused tail a +// consequence of one memset rather than a rule per buffer. A used length past +// the buffer, or below zero, is a value no reader could have produced and it is +// clamped rather than trusted: this writes inside the caller's buffer on every +// input. +inline void table_cook_bytes( uint8_t * at, const void * source, int64_t used, int64_t capacity ) +{ + if ( used <= 0 ) { return; } + const int64_t n = used < capacity ? used : capacity; + memcpy( at, source, (size_t) n ); +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE +#define LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// ---- the cooked form's WRITE side for a POINTERED root (docs/SPEC-TABLES.md §7.6) ---- +// +// A pointered root's cook is the region of §7.2: every node the numbering +// reached (§3.1), once, at its own type's alignment, in index order, the root +// at offset zero. This is that region while it is being laid out and written — +// the tool's own Layout and Write, in one struct. +// +// The OFFSETS are one per node, the root's zero at position 0 and node index k +// at position k - 1, which is the directory's own order (§6.3); they are the +// one allocation the write makes beyond the numbering, and they go through the +// same pair. A measure needs no offsets and leaves the pointer NULL. +struct TableCookRegion +{ + const TableNumbering * numbering = NULL; // node -> index, from the walk that placed it + int64_t * offsets = NULL; // index - 1 -> the node's region offset; NULL while measuring + int64_t count = 0; // nodes, the root included + int64_t bytes = 0; // the data part's length, rounded to align + int64_t align = 0; // the region's alignment: the nodes' greatest, never below eight + uint8_t * base = NULL; // where the data part is being written; NULL while measuring +}; + +// A reference slot: the SELF-RELATIVE delta from the slot's own address to the +// node's start (§6.3), and zero for null. The node is found by the address the +// numbering keyed it under, which is the same address the walk resolved through +// the same context — so a reference the numbering does not carry is a slot the +// walk never reached (a counted array's slot past its count, an absent +// optional's value) holding a node the region will not hold, and it is refused +// rather than written as a delta to nowhere. +inline bool table_cook_ref( const TableCookRegion & region, uint8_t * at, const void * pointee, TableByteOrder order ) +{ + if ( pointee == NULL ) { table_cook_put( at, 0, 8, order ); return true; } + uint64_t index = 0; + if ( !TableNumberingIndex( *region.numbering, pointee, index ) ) { return false; } + if ( index == 0 || index > (uint64_t) region.count ) { return false; } + const int64_t delta = region.offsets[index - 1] - (int64_t) ( at - region.base ); + table_cook_put( at, (uint64_t) delta, 8, order ); + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// table Placement — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Placement { + float x = 0.0f; + float y = 0.0f; + uint32_t model = 0; +}; + +// table LogEntry — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct LogEntry { + uint32_t tick = 0; +}; + +// table Save — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Save { + TableList placements; // Placement: the element array, empty until an Add + TableList log; // LogEntry: the element array, empty until an Add + TableList scores; // int32: the element array, empty until an Add +}; + +// table Point — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Point { + int32_t x = 0; + int32_t y = 0; +}; + +// HitType: union Hit's tag — None = 0, then each variant in declared order (SPEC §4.8) +enum class HitType : uint8_t { + None = 0, + Point = 1, + Damage = 2, + Max = 2, // the exported extent (SPEC §4.2) +}; + +// union Hit — at most one of the arms; the tag says which. AN ARM IS A FIELD +// LINE (docs/SPEC-TABLES.md §2.6), so an arm's storage is the field's storage +// overlaid — and an arm whose storage needs a companion, a string's length or +// a counted array's count, is one member of an unnamed struct, `value` beside +// `value_length` or `value_count`. Such a union has no packet wire and lives +// here, after its arms. Construction is None: the tag alone is initialized; an +// arm's storage is established when the arm is selected — by HitLoadBody +// before it decodes, or by assigning it: value.point = Point{}. +// Bytes of unselected arms are indeterminate. +struct Hit +{ + HitType type; + + union + { + Point point; + int32_t damage; + }; + + Hit() : type( HitType::None ) {} // the tag only — arms are established at selection +}; + +// table Mixed — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Mixed { + TableList grades; // Grade: the element array, empty until an Add + TableList perms; // Perm: the element array, empty until an Add + TableList hits; // Hit: the element array, empty until an Add + TableList bounds; // int32: the element array, empty until an Add +}; + +// Grade on the TABLE wire: a value rides under its OWN kind 30, carrying the +// REFERENCE to its variant name's id, whatever the declaration-side storage +// width — so a variant may be added anywhere, removed, or reordered and old +// data still reads (docs/SPEC-TABLES.md §3, §5). None is the ZERO REFERENCE, +// the one value that names no id, so no declared variant can be mistaken for it. +#ifndef LISTDEMO_SCHEMA_TABLE_ENUM_GRADE +#define LISTDEMO_SCHEMA_TABLE_ENUM_GRADE +inline bool TableEnumRef( TableIds & ids, Grade value, uint64_t & ref ) +{ + switch ( value ) + { + case Grade::None: ref = 0; return true; + case Grade::A: ref = ids.ref( 0xaf63fc4c860222ecull, 33 ); return true; + case Grade::B: ref = ids.ref( 0xaf63ff4c86022805ull, 34 ); return true; + case Grade::C: ref = ids.ref( 0xaf63fe4c86022652ull, 35 ); return true; + default: return false; // no variant names this value: no wire identity + } +} +inline bool TableEnumNamed( Grade value ) +{ + switch ( value ) + { + case Grade::None: return true; + case Grade::A: return true; + case Grade::B: return true; + case Grade::C: return true; + default: return false; + } +} +inline bool TableEnumId( Grade value, uint64_t & id ) +{ + switch ( value ) + { + case Grade::None: id = 0; return true; + case Grade::A: id = 0xaf63fc4c860222ecull; return true; + case Grade::B: id = 0xaf63ff4c86022805ull; return true; + case Grade::C: id = 0xaf63fe4c86022652ull; return true; + default: return false; // no variant names this value: no wire identity + } +} +inline bool TableEnumValue( uint64_t id, Grade & out ) +{ + switch ( id ) + { + case 0xaf63fc4c860222ecull: out = Grade::A; return true; + case 0xaf63ff4c86022805ull: out = Grade::B; return true; + case 0xaf63fe4c86022652ull: out = Grade::C; return true; + default: return false; // an id this build cannot name + } +} +#endif // LISTDEMO_SCHEMA_TABLE_ENUM_GRADE + +// ---- prefill: the declared defaults, in place (docs/SPEC-TABLES.md) ---- + +inline void PlacementReset( Placement & value ); +inline void LogEntryReset( LogEntry & value ); +inline void SaveReset( Save & value ); +inline void PointReset( Point & value ); +inline void MixedReset( Mixed & value ); + +inline void PlacementReset( Placement & value ) +{ + value.x = 0.0f; + value.y = 0.0f; + value.model = 0; +} + +inline void LogEntryReset( LogEntry & value ) +{ + value.tick = 0; +} + +inline void SaveReset( Save & value ) +{ + value.placements.elements.value = 0; // Placement: empty + value.placements.count = 0; + value.placements.padding = 0; + value.log.elements.value = 0; // LogEntry: empty + value.log.count = 0; + value.log.padding = 0; + value.scores.elements.value = 0; // int32: empty + value.scores.count = 0; + value.scores.padding = 0; +} + +inline void PointReset( Point & value ) +{ + value.x = 0; + value.y = 0; +} + +inline void MixedReset( Mixed & value ) +{ + value.grades.elements.value = 0; // Grade: empty + value.grades.count = 0; + value.grades.padding = 0; + value.perms.elements.value = 0; // Perm: empty + value.perms.count = 0; + value.perms.padding = 0; + value.hits.elements.value = 0; // Hit: empty + value.hits.count = 0; + value.hits.padding = 0; + value.bounds.elements.value = 0; // int32: empty + value.bounds.count = 0; + value.bounds.padding = 0; +} + +// ---- the arena's reset hook (docs/SPEC-TABLES.md §6) ---- +// +// TableWorker::Alloc is a template and cannot name a member's Reset, so +// the arena reaches it through this overload set by argument-dependent +// lookup. It is how a node born in raw arena storage comes to hold the +// declared defaults without value-initialising the whole aggregate. + +inline void TableReset( Placement & value ) { PlacementReset( value ); } +inline void TableReset( LogEntry & value ) { LogEntryReset( value ); } +inline void TableReset( Save & value ) { SaveReset( value ); } +inline void TableReset( Point & value ) { PointReset( value ); } +inline void TableReset( Mixed & value ) { MixedReset( value ); } + +// ---- pointer targets: allocation and resolution (docs/SPEC-TABLES.md §2) ---- +// +// A reference resolves differently in the two forms, and the CONTEXT says +// which: in the arena it is an offset; in a region it is a self-relative +// delta, so the const deref below is one add and needs no base pointer. + +// LogEntry is a pointer target. +inline const LogEntry * LogEntryAt( const TableRef & ref ) // the const form's hot path: one add, no base +{ + return ref.value != 0 ? (const LogEntry *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline LogEntry * LogEntryAt( TableRef & ref ) +{ + return ref.value != 0 ? (LogEntry *) ( (uint8_t *) &ref + ref.value ) : NULL; +} +inline const LogEntry * LogEntryAt( const TableRegionCtx &, const TableRef & ref ) { return LogEntryAt( ref ); } +inline const LogEntry * LogEntryAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const LogEntry *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +// while the builder is mutable, resolve against the arena itself +inline LogEntry * LogEntryAt( TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (LogEntry *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} +inline const LogEntry * LogEntryAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const LogEntry *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} +// allocate one LogEntry in the arena; the slot holds the arena offset +inline LogEntry * LogEntryEmplace( TableWorker & worker, TableRef & slot ) +{ + TableSlot allocated = worker.Alloc(); + slot = allocated.ref; + return allocated.ptr; +} + +// ---- codecs: measure/save/load per closure member ---- + +inline int64_t PlacementMeasureBody( TableIds & ids, const Placement & value ); +LISTDEMO_TABLE_INLINE bool PlacementSaveBody( TableWriter & w, TableIds & ids, const Placement & value ); +LISTDEMO_TABLE_INLINE bool PlacementLoadBody( TableReader & r, Placement & value ); +inline int64_t LogEntryMeasureBody( TableIds & ids, const LogEntry & value ); +LISTDEMO_TABLE_INLINE bool LogEntrySaveBody( TableWriter & w, TableIds & ids, const LogEntry & value ); +LISTDEMO_TABLE_INLINE bool LogEntryLoadBody( TableReader & r, LogEntry & value ); +template inline int64_t SaveMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Save & value ); +template inline bool SaveSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Save & value ); +template inline bool SaveSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Save & value ); +inline bool SaveLoadBody( TableReader & r, const TableNodeMap & nodes, Save & value ); +inline int64_t PointMeasureBody( TableIds & ids, const Point & value ); +LISTDEMO_TABLE_INLINE bool PointSaveBody( TableWriter & w, TableIds & ids, const Point & value ); +LISTDEMO_TABLE_INLINE bool PointLoadBody( TableReader & r, Point & value ); +template inline int64_t MixedMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Mixed & value ); +template inline bool MixedSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Mixed & value ); +template inline bool MixedSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Mixed & value ); +inline bool MixedLoadBody( TableReader & r, const TableNodeMap & nodes, Mixed & value ); + +// ---- pointer-graph walkers: number (measure/save), pack (Lock) ---- + +template inline bool LogEntryNumber( const Ctx & ctx, TableNumbering & numbering, const LogEntry & value ); +template inline int64_t LogEntryPackMeasure( const Ctx & ctx, TablePackMap & seen, const LogEntry & value ); +template inline bool LogEntryPack( const Ctx & ctx, TablePackMap & seen, const LogEntry & src, LogEntry & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool SaveNumber( const Ctx & ctx, TableNumbering & numbering, const Save & value ); +template inline int64_t SavePackMeasure( const Ctx & ctx, TablePackMap & seen, const Save & value ); +template inline bool SavePack( const Ctx & ctx, TablePackMap & seen, const Save & src, Save & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool MixedNumber( const Ctx & ctx, TableNumbering & numbering, const Mixed & value ); +template inline int64_t MixedPackMeasure( const Ctx & ctx, TablePackMap & seen, const Mixed & value ); +template inline bool MixedPack( const Ctx & ctx, TablePackMap & seen, const Mixed & src, Mixed & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +// ---- the numbering's bridge to each member's codec (docs/SPEC-TABLES.md §3.1) ---- + +template inline int64_t TableNodeMeasure( const Ctx &, const TableNumbering &, TableIds & ids, const LogEntry & value ) { return LogEntryMeasureBody( ids, value ); } +template inline bool TableNodeSave( const Ctx &, const TableNumbering &, TableWriter & w, TableIds & ids, const LogEntry & value ) { return LogEntrySaveBody( w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Save & value ) { return SaveMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Save & value ) { return SaveSaveBody( ctx, numbering, w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Mixed & value ) { return MixedMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Mixed & value ) { return MixedSaveBody( ctx, numbering, w, ids, value ); } + +inline int64_t PlacementMeasureBody( TableIds & ids, const Placement & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.x != 0.0f ) { bytes += TableLebBytes( ids.ref( 0xaf63f54c86021707ull, 19 ) ) + 1 + 4; } // x + if ( value.y != 0.0f ) { bytes += TableLebBytes( ids.ref( 0xaf63f44c86021554ull, 20 ) ) + 1 + 4; } // y + if ( value.model != 0 ) { bytes += TableLebBytes( ids.ref( 0x9de543933e6e703aull, 21 ) ) + 1 + 4; } // model + return bytes; +} + +inline int64_t PlacementMeasure( const Placement & value ) +{ + TableIds ids; + const int64_t body = PlacementMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool PlacementSaveBody( TableWriter & w, TableIds & ids, const Placement & value ) +{ + if ( value.x != 0.0f ) + { + w.putleb( ids.ref( 0xaf63f54c86021707ull, 19 ) ); w.put8( 10 ); // x + w.put32( table_float_to_bits( value.x ) ); + } + if ( value.y != 0.0f ) + { + w.putleb( ids.ref( 0xaf63f44c86021554ull, 20 ) ); w.put8( 10 ); // y + w.put32( table_float_to_bits( value.y ) ); + } + if ( value.model != 0 ) + { + w.putleb( ids.ref( 0x9de543933e6e703aull, 21 ) ); w.put8( 8 ); // model + w.put32( uint32_t( value.model ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t PlacementSave( const Placement & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !PlacementSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == PlacementMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool PlacementLoadBody( TableReader & r, Placement & value ) +{ + PlacementReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xaf63f54c86021707ull: // x + { + if ( kind != 10 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + value.x = table_bits_to_float( r.get32() ); + break; + } + case 0xaf63f44c86021554ull: // y + { + if ( kind != 10 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + value.y = table_bits_to_float( r.get32() ); + break; + } + case 0x9de543933e6e703aull: // model + { + if ( kind != 8 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + uint32_t decoded_v = uint32_t( r.get32( ) ); + value.model = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict PlacementLoadVerdict( Placement & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + PlacementReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + PlacementReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !PlacementLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool PlacementLoad( Placement & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return PlacementLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t PlacementMeasureMessage( const Placement & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = PlacementMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t PlacementSaveMessage( const Placement & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !PlacementSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == PlacementMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool PlacementLoadMessage( Placement & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + PlacementReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return PlacementLoadBody( r, value ); +} + +inline int64_t LogEntryMeasureBody( TableIds & ids, const LogEntry & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.tick != 0 ) { bytes += TableLebBytes( ids.ref( 0x1e7683ef2ebc7684ull, 12 ) ) + 1 + 4; } // tick + return bytes; +} + +inline int64_t LogEntryMeasure( const LogEntry & value ) +{ + TableIds ids; + const int64_t body = LogEntryMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool LogEntrySaveBody( TableWriter & w, TableIds & ids, const LogEntry & value ) +{ + if ( value.tick != 0 ) + { + w.putleb( ids.ref( 0x1e7683ef2ebc7684ull, 12 ) ); w.put8( 8 ); // tick + w.put32( uint32_t( value.tick ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t LogEntrySave( const LogEntry & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !LogEntrySaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == LogEntryMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool LogEntryLoadBody( TableReader & r, LogEntry & value ) +{ + LogEntryReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x1e7683ef2ebc7684ull: // tick + { + if ( kind != 8 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + uint32_t decoded_v = uint32_t( r.get32( ) ); + value.tick = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict LogEntryLoadVerdict( LogEntry & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + LogEntryReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + LogEntryReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !LogEntryLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool LogEntryLoad( LogEntry & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return LogEntryLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t LogEntryMeasureMessage( const LogEntry & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = LogEntryMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t LogEntrySaveMessage( const LogEntry & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !LogEntrySaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == LogEntryMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool LogEntryLoadMessage( LogEntry & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + LogEntryReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return LogEntryLoadBody( r, value ); +} + +template +inline int64_t SaveMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Save & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // placements: a kind 14 array of kind 13 elements, INDEX order (§2.9) + TableListCursor cursor_placements = TableListElements( ctx, value.placements ); + if ( !cursor_placements.ok ) { return -1; } // the slot and the head disagree + if ( cursor_placements.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_placements = ids.ref( 0xd24733aa574d4b09ull, 24 ); + int64_t body_placements = 0; + body_placements += 1 + TableLebBytes( (uint64_t) ( cursor_placements.count ) ); // the element kind byte and the count + for ( int32_t elem_i_placements = 0; elem_i_placements < cursor_placements.count; elem_i_placements++ ) + { + const int64_t elem_bytes_placements = PlacementMeasureBody( ids, cursor_placements[elem_i_placements] ); + if ( elem_bytes_placements < 0 ) { return -1; } + body_placements += TableLebBytes( (uint64_t) ( elem_bytes_placements ) ) + ( elem_bytes_placements ); + } + bytes += TableLebBytes( ref_placements ) + 1 + TableLebBytes( (uint64_t) ( body_placements ) ) + ( body_placements ); + } + } + { + // log: a kind 14 array of kind 17 elements, INDEX order (§2.9) + TableListCursor cursor_log = TableListElements( ctx, value.log ); + if ( !cursor_log.ok ) { return -1; } // the slot and the head disagree + if ( cursor_log.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_log = ids.ref( 0x125073191daf5431ull, 25 ); + int64_t body_log = 0; + body_log += 1 + TableLebBytes( (uint64_t) ( cursor_log.count ) ); // the element kind byte and the count + for ( int32_t elem_i_log = 0; elem_i_log < cursor_log.count; elem_i_log++ ) + { + { + const LogEntry * slot_pointee_log = LogEntryAt( ctx, cursor_log[elem_i_log] ); + uint64_t slot_index_log = 0; + if ( slot_pointee_log != NULL && !TableNumberingIndex( numbering, (const void *) slot_pointee_log, slot_index_log ) ) { return -1; } + body_log += TableLebBytes( slot_index_log ); + } + } + bytes += TableLebBytes( ref_log ) + 1 + TableLebBytes( (uint64_t) ( body_log ) ) + ( body_log ); + } + } + { + // scores: a kind 14 array of kind 4 elements, INDEX order (§2.9) + TableListCursor cursor_scores = TableListElements( ctx, value.scores ); + if ( !cursor_scores.ok ) { return -1; } // the slot and the head disagree + if ( cursor_scores.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_scores = ids.ref( 0x01986b0b27400fb2ull, 26 ); + int64_t body_scores = 0; + body_scores += 1 + TableLebBytes( (uint64_t) ( cursor_scores.count ) ); // the element kind byte and the count + body_scores += (int64_t) ( cursor_scores.count ) * 4; + bytes += TableLebBytes( ref_scores ) + 1 + TableLebBytes( (uint64_t) ( body_scores ) ) + ( body_scores ); + } + } + return bytes; +} + +template +inline bool SaveSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Save & value ) +{ + { + TableListCursor cursor_placements = TableListElements( ctx, value.placements ); // placements + if ( !cursor_placements.ok ) { return false; } + if ( cursor_placements.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_placements = ids.ref( 0xd24733aa574d4b09ull, 24 ); + int64_t body_placements = 0; + body_placements += 1 + TableLebBytes( (uint64_t) ( cursor_placements.count ) ); // the element kind byte and the count + for ( int32_t elem_i_placements = 0; elem_i_placements < cursor_placements.count; elem_i_placements++ ) + { + const int64_t elem_bytes_placements = PlacementMeasureBody( ids, cursor_placements[elem_i_placements] ); + if ( elem_bytes_placements < 0 ) { return false; } + body_placements += TableLebBytes( (uint64_t) ( elem_bytes_placements ) ) + ( elem_bytes_placements ); + } + w.putleb( ref_placements ); w.put8( 14 ); w.putleb( (uint64_t) body_placements ); // placements + w.put8( 13 ); w.putleb( (uint64_t) ( cursor_placements.count ) ); + for ( int32_t elem_i_placements = 0; elem_i_placements < cursor_placements.count; elem_i_placements++ ) + { + { + const int64_t elem_len_placements = PlacementMeasureBody( ids, cursor_placements[elem_i_placements] ); + if ( elem_len_placements < 0 ) return false; + w.putleb( (uint64_t) elem_len_placements ); + if ( !PlacementSaveBody( w, ids, cursor_placements[elem_i_placements] ) ) return false; + } + } + } + } + { + TableListCursor cursor_log = TableListElements( ctx, value.log ); // log + if ( !cursor_log.ok ) { return false; } + if ( cursor_log.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_log = ids.ref( 0x125073191daf5431ull, 25 ); + int64_t body_log = 0; + body_log += 1 + TableLebBytes( (uint64_t) ( cursor_log.count ) ); // the element kind byte and the count + for ( int32_t elem_i_log = 0; elem_i_log < cursor_log.count; elem_i_log++ ) + { + { + const LogEntry * slot_pointee_log = LogEntryAt( ctx, cursor_log[elem_i_log] ); + uint64_t slot_index_log = 0; + if ( slot_pointee_log != NULL && !TableNumberingIndex( numbering, (const void *) slot_pointee_log, slot_index_log ) ) { return false; } + body_log += TableLebBytes( slot_index_log ); + } + } + w.putleb( ref_log ); w.put8( 14 ); w.putleb( (uint64_t) body_log ); // log + w.put8( 17 ); w.putleb( (uint64_t) ( cursor_log.count ) ); + for ( int32_t elem_i_log = 0; elem_i_log < cursor_log.count; elem_i_log++ ) + { + { + const LogEntry * slot_pointee_log = LogEntryAt( ctx, cursor_log[elem_i_log] ); + uint64_t slot_index_log = 0; + if ( slot_pointee_log != NULL && !TableNumberingIndex( numbering, (const void *) slot_pointee_log, slot_index_log ) ) { return false; } + w.putleb( slot_index_log ); + } + } + } + } + { + TableListCursor cursor_scores = TableListElements( ctx, value.scores ); // scores + if ( !cursor_scores.ok ) { return false; } + if ( cursor_scores.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_scores = ids.ref( 0x01986b0b27400fb2ull, 26 ); + int64_t body_scores = 0; + body_scores += 1 + TableLebBytes( (uint64_t) ( cursor_scores.count ) ); // the element kind byte and the count + body_scores += (int64_t) ( cursor_scores.count ) * 4; + w.putleb( ref_scores ); w.put8( 14 ); w.putleb( (uint64_t) body_scores ); // scores + w.put8( 4 ); w.putleb( (uint64_t) ( cursor_scores.count ) ); + for ( int32_t elem_i_scores = 0; elem_i_scores < cursor_scores.count; elem_i_scores++ ) + { + w.put32( uint32_t( cursor_scores[elem_i_scores] ) ); + } + } + } + return !w.overflow; +} + +template +inline bool SaveSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Save & value ) +{ + if ( !SaveSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool SaveLoadBody( TableReader & r, const TableNodeMap & nodes, Save & value ) +{ + SaveReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xd24733aa574d4b09ull: // placements + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 13 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.placements, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Placement * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + uint64_t elem_len_placements = 0; + if ( !sub.getleb( elem_len_placements ) || !sub.room( elem_len_placements ) ) { r.report->malformed = true; break; } + { + TableReader elem_placements( sub.buffer + sub.offset, (int64_t) elem_len_placements, r.report, r.ids ); + PlacementLoadBody( elem_placements, ( *slot ) ); + } + sub.offset += (int64_t) elem_len_placements; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x125073191daf5431ull: // log + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 17 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.log, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + TableRef * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + { + uint64_t node_index_log = 0; + if ( !sub.getleb( node_index_log ) ) { r.report->malformed = true; break; } + TableNodeResolve( nodes, ( *slot ), node_index_log, 0x5e781536ac58825full, r.report ); // *LogEntry + } + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x01986b0b27400fb2ull: // scores + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 4 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.scores, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + int32_t * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + if ( !sub.has( 4 ) ) { r.report->malformed = true; break; } + int32_t decoded_v = int32_t( sub.get32( ) ); + ( *slot ) = decoded_v; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +inline int64_t PointMeasureBody( TableIds & ids, const Point & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.x != 0 ) { bytes += TableLebBytes( ids.ref( 0xaf63f54c86021707ull, 19 ) ) + 1 + 4; } // x + if ( value.y != 0 ) { bytes += TableLebBytes( ids.ref( 0xaf63f44c86021554ull, 20 ) ) + 1 + 4; } // y + return bytes; +} + +inline int64_t PointMeasure( const Point & value ) +{ + TableIds ids; + const int64_t body = PointMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool PointSaveBody( TableWriter & w, TableIds & ids, const Point & value ) +{ + if ( value.x != 0 ) + { + w.putleb( ids.ref( 0xaf63f54c86021707ull, 19 ) ); w.put8( 4 ); // x + w.put32( uint32_t( value.x ) ); + } + if ( value.y != 0 ) + { + w.putleb( ids.ref( 0xaf63f44c86021554ull, 20 ) ); w.put8( 4 ); // y + w.put32( uint32_t( value.y ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t PointSave( const Point & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !PointSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == PointMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool PointLoadBody( TableReader & r, Point & value ) +{ + PointReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xaf63f54c86021707ull: // x + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.x = decoded_v; + break; + } + case 0xaf63f44c86021554ull: // y + { + if ( kind != 4 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + int32_t decoded_v = int32_t( r.get32( ) ); + value.y = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict PointLoadVerdict( Point & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + PointReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + PointReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !PointLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool PointLoad( Point & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return PointLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t PointMeasureMessage( const Point & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = PointMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t PointSaveMessage( const Point & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !PointSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == PointMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool PointLoadMessage( Point & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + PointReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return PointLoadBody( r, value ); +} + +template +inline int64_t MixedMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Mixed & value ) +{ + (void) ctx; (void) numbering; + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // grades: a kind 14 array of kind 30 elements, INDEX order (§2.9) + TableListCursor cursor_grades = TableListElements( ctx, value.grades ); + if ( !cursor_grades.ok ) { return -1; } // the slot and the head disagree + if ( cursor_grades.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_grades = ids.ref( 0xd90a4e7682f799c5ull, 13 ); + int64_t body_grades = 0; + body_grades += 1 + TableLebBytes( (uint64_t) ( cursor_grades.count ) ); // the element kind byte and the count + for ( int32_t elem_i_grades = 0; elem_i_grades < cursor_grades.count; elem_i_grades++ ) + { + uint64_t elem_ref_grades = 0; + if ( !TableEnumRef( ids, cursor_grades[elem_i_grades], elem_ref_grades ) ) { return -1; } // no variant names this value + body_grades += TableLebBytes( elem_ref_grades ); + } + bytes += TableLebBytes( ref_grades ) + 1 + TableLebBytes( (uint64_t) ( body_grades ) ) + ( body_grades ); + } + } + { + // perms: a kind 14 array of kind 9 elements, INDEX order (§2.9) + TableListCursor cursor_perms = TableListElements( ctx, value.perms ); + if ( !cursor_perms.ok ) { return -1; } // the slot and the head disagree + if ( cursor_perms.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_perms = ids.ref( 0x4af2ed8470862ea8ull, 14 ); + int64_t body_perms = 0; + body_perms += 1 + TableLebBytes( (uint64_t) ( cursor_perms.count ) ); // the element kind byte and the count + body_perms += (int64_t) ( cursor_perms.count ) * 8; + bytes += TableLebBytes( ref_perms ) + 1 + TableLebBytes( (uint64_t) ( body_perms ) ) + ( body_perms ); + } + } + { + // hits: a kind 14 array of kind 15 elements, INDEX order (§2.9) + TableListCursor cursor_hits = TableListElements( ctx, value.hits ); + if ( !cursor_hits.ok ) { return -1; } // the slot and the head disagree + if ( cursor_hits.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_hits = ids.ref( 0x732dfbcc9b0cf0bbull, 15 ); + int64_t body_hits = 0; + body_hits += 1 + TableLebBytes( (uint64_t) ( cursor_hits.count ) ); // the element kind byte and the count + for ( int32_t elem_i_hits = 0; elem_i_hits < cursor_hits.count; elem_i_hits++ ) + { + if ( cursor_hits[elem_i_hits].type == HitType::None ) { body_hits += 1; } // a None element is the zero reference in its place + else + { + switch ( cursor_hits[elem_i_hits].type ) + { + case HitType::None: break; + case HitType::Point: + { + int64_t arm_payload_hitsu = 0; + const uint64_t arm_ref_hitsu = ids.ref( 0x73feab3544c345b1ull, 36 ); + { + const int64_t arm_body_hitsu = PointMeasureBody( ids, cursor_hits[elem_i_hits].point ); + if ( arm_body_hitsu < 0 ) { return -1; } + arm_payload_hitsu += arm_body_hitsu; // the arm's own table body (§3) + } + body_hits += TableLebBytes( arm_ref_hitsu ) + 1 + TableLebBytes( (uint64_t) ( arm_payload_hitsu ) ) + ( arm_payload_hitsu ); + break; + } + case HitType::Damage: + { + int64_t arm_payload_hitsu = 0; + const uint64_t arm_ref_hitsu = ids.ref( 0x7f6308be8ab37fc0ull, 37 ); + arm_payload_hitsu += 4; // int32 + body_hits += TableLebBytes( arm_ref_hitsu ) + 1 + TableLebBytes( (uint64_t) ( arm_payload_hitsu ) ) + ( arm_payload_hitsu ); + break; + } + default: return -1; // invalid tag — the write side refuses it too + } + } + } + bytes += TableLebBytes( ref_hits ) + 1 + TableLebBytes( (uint64_t) ( body_hits ) ) + ( body_hits ); + } + } + { + // bounds: a kind 14 array of kind 4 elements, INDEX order (§2.9) + TableListCursor cursor_bounds = TableListElements( ctx, value.bounds ); + if ( !cursor_bounds.ok ) { return -1; } // the slot and the head disagree + if ( cursor_bounds.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_bounds = ids.ref( 0x52f60c4caef0b768ull, 16 ); + int64_t body_bounds = 0; + body_bounds += 1 + TableLebBytes( (uint64_t) ( cursor_bounds.count ) ); // the element kind byte and the count + body_bounds += (int64_t) ( cursor_bounds.count ) * 4; + bytes += TableLebBytes( ref_bounds ) + 1 + TableLebBytes( (uint64_t) ( body_bounds ) ) + ( body_bounds ); + } + } + return bytes; +} + +template +inline bool MixedSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Mixed & value ) +{ + (void) ctx; (void) numbering; + { + TableListCursor cursor_grades = TableListElements( ctx, value.grades ); // grades + if ( !cursor_grades.ok ) { return false; } + if ( cursor_grades.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_grades = ids.ref( 0xd90a4e7682f799c5ull, 13 ); + int64_t body_grades = 0; + body_grades += 1 + TableLebBytes( (uint64_t) ( cursor_grades.count ) ); // the element kind byte and the count + for ( int32_t elem_i_grades = 0; elem_i_grades < cursor_grades.count; elem_i_grades++ ) + { + uint64_t elem_ref_grades = 0; + if ( !TableEnumRef( ids, cursor_grades[elem_i_grades], elem_ref_grades ) ) { return false; } // no variant names this value + body_grades += TableLebBytes( elem_ref_grades ); + } + w.putleb( ref_grades ); w.put8( 14 ); w.putleb( (uint64_t) body_grades ); // grades + w.put8( 30 ); w.putleb( (uint64_t) ( cursor_grades.count ) ); + for ( int32_t elem_i_grades = 0; elem_i_grades < cursor_grades.count; elem_i_grades++ ) + { + { + uint64_t element_ref_grades = 0; + if ( !TableEnumRef( ids, cursor_grades[elem_i_grades], element_ref_grades ) ) { return false; } + w.putleb( element_ref_grades ); + } + } + } + } + { + TableListCursor cursor_perms = TableListElements( ctx, value.perms ); // perms + if ( !cursor_perms.ok ) { return false; } + if ( cursor_perms.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_perms = ids.ref( 0x4af2ed8470862ea8ull, 14 ); + int64_t body_perms = 0; + body_perms += 1 + TableLebBytes( (uint64_t) ( cursor_perms.count ) ); // the element kind byte and the count + body_perms += (int64_t) ( cursor_perms.count ) * 8; + w.putleb( ref_perms ); w.put8( 14 ); w.putleb( (uint64_t) body_perms ); // perms + w.put8( 9 ); w.putleb( (uint64_t) ( cursor_perms.count ) ); + for ( int32_t elem_i_perms = 0; elem_i_perms < cursor_perms.count; elem_i_perms++ ) + { + w.put64( uint64_t( cursor_perms[elem_i_perms] ) ); + } + } + } + { + TableListCursor cursor_hits = TableListElements( ctx, value.hits ); // hits + if ( !cursor_hits.ok ) { return false; } + if ( cursor_hits.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_hits = ids.ref( 0x732dfbcc9b0cf0bbull, 15 ); + int64_t body_hits = 0; + body_hits += 1 + TableLebBytes( (uint64_t) ( cursor_hits.count ) ); // the element kind byte and the count + for ( int32_t elem_i_hits = 0; elem_i_hits < cursor_hits.count; elem_i_hits++ ) + { + if ( cursor_hits[elem_i_hits].type == HitType::None ) { body_hits += 1; } // a None element is the zero reference in its place + else + { + switch ( cursor_hits[elem_i_hits].type ) + { + case HitType::None: break; + case HitType::Point: + { + int64_t arm_payload_hitsu = 0; + const uint64_t arm_ref_hitsu = ids.ref( 0x73feab3544c345b1ull, 36 ); + { + const int64_t arm_body_hitsu = PointMeasureBody( ids, cursor_hits[elem_i_hits].point ); + if ( arm_body_hitsu < 0 ) { return -1; } + arm_payload_hitsu += arm_body_hitsu; // the arm's own table body (§3) + } + body_hits += TableLebBytes( arm_ref_hitsu ) + 1 + TableLebBytes( (uint64_t) ( arm_payload_hitsu ) ) + ( arm_payload_hitsu ); + break; + } + case HitType::Damage: + { + int64_t arm_payload_hitsu = 0; + const uint64_t arm_ref_hitsu = ids.ref( 0x7f6308be8ab37fc0ull, 37 ); + arm_payload_hitsu += 4; // int32 + body_hits += TableLebBytes( arm_ref_hitsu ) + 1 + TableLebBytes( (uint64_t) ( arm_payload_hitsu ) ) + ( arm_payload_hitsu ); + break; + } + default: return -1; // invalid tag — the write side refuses it too + } + } + } + w.putleb( ref_hits ); w.put8( 14 ); w.putleb( (uint64_t) body_hits ); // hits + w.put8( 15 ); w.putleb( (uint64_t) ( cursor_hits.count ) ); + for ( int32_t elem_i_hits = 0; elem_i_hits < cursor_hits.count; elem_i_hits++ ) + { + if ( cursor_hits[elem_i_hits].type == HitType::None ) { w.putleb( 0 ); } // a None element rides in its place + else + { + switch ( cursor_hits[elem_i_hits].type ) + { + case HitType::Point: + { + const uint64_t arm_ref_hitsu = ids.ref( 0x73feab3544c345b1ull, 36 ); + int64_t arm_payload_hitsu = 0; + { + const int64_t arm_body_hitsu = PointMeasureBody( ids, cursor_hits[elem_i_hits].point ); + if ( arm_body_hitsu < 0 ) { return false; } + arm_payload_hitsu += arm_body_hitsu; // the arm's own table body (§3) + } + w.putleb( arm_ref_hitsu ); w.put8( 13 ); w.putleb( (uint64_t) arm_payload_hitsu ); // point + if ( !PointSaveBody( w, ids, cursor_hits[elem_i_hits].point ) ) { return false; } + break; + } + case HitType::Damage: + { + const uint64_t arm_ref_hitsu = ids.ref( 0x7f6308be8ab37fc0ull, 37 ); + int64_t arm_payload_hitsu = 0; + arm_payload_hitsu += 4; // int32 + w.putleb( arm_ref_hitsu ); w.put8( 4 ); w.putleb( (uint64_t) arm_payload_hitsu ); // damage + w.put32( uint32_t( cursor_hits[elem_i_hits].damage ) ); + break; + } + default: return false; // write validates the tag before it rides + } + } + } + } + } + { + TableListCursor cursor_bounds = TableListElements( ctx, value.bounds ); // bounds + if ( !cursor_bounds.ok ) { return false; } + if ( cursor_bounds.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_bounds = ids.ref( 0x52f60c4caef0b768ull, 16 ); + int64_t body_bounds = 0; + body_bounds += 1 + TableLebBytes( (uint64_t) ( cursor_bounds.count ) ); // the element kind byte and the count + body_bounds += (int64_t) ( cursor_bounds.count ) * 4; + w.putleb( ref_bounds ); w.put8( 14 ); w.putleb( (uint64_t) body_bounds ); // bounds + w.put8( 4 ); w.putleb( (uint64_t) ( cursor_bounds.count ) ); + for ( int32_t elem_i_bounds = 0; elem_i_bounds < cursor_bounds.count; elem_i_bounds++ ) + { + w.put32( uint32_t( cursor_bounds[elem_i_bounds] ) ); + } + } + } + return !w.overflow; +} + +template +inline bool MixedSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Mixed & value ) +{ + if ( !MixedSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool MixedLoadBody( TableReader & r, const TableNodeMap & nodes, Mixed & value ) +{ + (void) nodes; + MixedReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xd90a4e7682f799c5ull: // grades + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 30 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.grades, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Grade * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + { + uint64_t variant_ref = 0; + if ( !sub.getleb( variant_ref ) ) { r.report->malformed = true; break; } + if ( variant_ref == 0 ) { ( *slot ) = Grade::None; } // the zero reference is the enum's None + else if ( variant_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; break; } + else if ( !TableEnumValue( r.ids->at( variant_ref ), ( *slot ) ) ) + { + ( *slot ) = Grade::None; + r.report->unknown++; + } + } + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x4af2ed8470862ea8ull: // perms + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 9 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.perms, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Perm * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + if ( !sub.has( 8 ) ) { r.report->malformed = true; break; } + uint64_t decoded_v = uint64_t( sub.get64( ) ); + ( *slot ) = decoded_v; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x732dfbcc9b0cf0bbull: // hits + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 15 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.hits, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + Hit * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + { + uint64_t elem_arm_ref_hits = 0; + if ( !sub.getleb( elem_arm_ref_hits ) ) { r.report->malformed = true; break; } + ( *slot ).type = HitType::None; + if ( elem_arm_ref_hits != 0 ) // the zero reference is a None element in its place + { + if ( elem_arm_ref_hits > (uint64_t) r.ids->count ) { r.report->malformed = true; break; } + const uint64_t elem_arm_id_hits = r.ids->at( elem_arm_ref_hits ); + if ( !sub.has( 1 ) ) { r.report->malformed = true; break; } + const uint8_t elem_arm_kind_hits = sub.get8(); + uint64_t elem_arm_len_hits = 0; + if ( !sub.getleb( elem_arm_len_hits ) || !sub.room( elem_arm_len_hits ) ) { r.report->malformed = true; break; } + TableReader elem_arm_hits( sub.buffer + sub.offset, (int64_t) elem_arm_len_hits, r.report, r.ids ); + switch ( elem_arm_id_hits ) // the arm's NAME hash (docs/SPEC-TABLES.md §5) + { + case 0x73feab3544c345b1ull: // point + { + if ( elem_arm_kind_hits != 13 ) { ( *slot ).type = HitType::None; r.report->kind_mismatch++; break; } + ( *slot ).type = HitType::Point; + PointLoadBody( elem_arm_hits, ( *slot ).point ); + if ( elem_arm_hits.offset != elem_arm_hits.size ) { ( *slot ).type = HitType::None; r.report->malformed = true; break; } + break; + } + case 0x7f6308be8ab37fc0ull: // damage + { + if ( elem_arm_kind_hits != 4 ) { ( *slot ).type = HitType::None; r.report->kind_mismatch++; break; } + ( *slot ).type = HitType::Damage; + if ( elem_arm_hits.size != 4 ) { ( *slot ).type = HitType::None; r.report->malformed = true; break; } // an L that is not the kind's width is that arm's own framing damage (§3) + if ( !elem_arm_hits.has( 4 ) ) { ( *slot ).type = HitType::None; r.report->malformed = true; break; } + int32_t decoded_v = int32_t( elem_arm_hits.get32( ) ); + ( *slot ).damage = decoded_v; + break; + } + default: r.report->unknown++; break; // an arm this reader cannot name: the element reads None, the body skips by its length + } + sub.offset += (int64_t) elem_arm_len_hits; + } + } + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0x52f60c4caef0b768ull: // bounds + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 4 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.bounds, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + int32_t * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + if ( !sub.has( 4 ) ) { r.report->malformed = true; break; } + int32_t decoded_v = int32_t( sub.get32( ) ); + if ( decoded_v < 0 ) { decoded_v = 0; r.report->clamped++; } + else if ( decoded_v > 100 ) { decoded_v = 100; r.report->clamped++; } + ( *slot ) = decoded_v; + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// LogEntryWireExtent: the extent LogEntry's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool LogEntryWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + (void) body; (void) length; (void) at; (void) ids; (void) reason; // no list or map below this record + return true; +} + +// LogEntryExtentAt: the node extent LogEntry's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as LogEntryExtentPack advances it (§2.8, §2.9). +template +inline bool LogEntryExtentAt( const Ctx & ctx, const LogEntry & value, int64_t & at ) +{ + (void) ctx; (void) value; (void) at; // no list or map below this record + return true; +} + +// LogEntryExtentPack: carve LogEntry's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset LogEntryExtentAt advances (§2.8, §2.9). +template +inline bool LogEntryExtentPack( const Ctx & ctx, const LogEntry & src, LogEntry & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no list or map below this record + return true; +} + +// SaveWireExtent: the extent Save's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool SaveWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0xd24733aa574d4b09ull && field_kind == 14 ) // placements: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Placement ), (int64_t) alignof( Placement ), 13, 2, NULL, ids, reason ) ) { return false; } + continue; + } + if ( field_id == 0x125073191daf5431ull && field_kind == 14 ) // log: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( TableRef ), (int64_t) alignof( TableRef ), 17, 1, NULL, ids, reason ) ) { return false; } + continue; + } + if ( field_id == 0x01986b0b27400fb2ull && field_kind == 14 ) // scores: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( int32_t ), (int64_t) alignof( int32_t ), 4, 4, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// SaveExtentAt: the node extent Save's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as SaveExtentPack advances it (§2.8, §2.9). +template +inline bool SaveExtentAt( const Ctx & ctx, const Save & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.placements ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Placement ) - 1 ) & ~( (int64_t) alignof( Placement ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Placement ); // the whole array FIRST + } + { + TableListCursor cursor = TableListElements( ctx, value.log ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( TableRef ) - 1 ) & ~( (int64_t) alignof( TableRef ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( TableRef ); // the whole array FIRST + } + { + TableListCursor cursor = TableListElements( ctx, value.scores ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( int32_t ) - 1 ) & ~( (int64_t) alignof( int32_t ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( int32_t ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t SaveExtent( const Ctx & ctx, const Save & value ) +{ + int64_t at = 0; + if ( !SaveExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// SaveExtentPack: carve Save's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset SaveExtentAt advances (§2.8, §2.9). +template +inline bool SaveExtentPack( const Ctx & ctx, const Save & src, Save & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.placements ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Placement ) - 1 ) & ~( (int64_t) alignof( Placement ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Placement ); + if ( at + bytes > capacity ) { return false; } + Placement * placed = (Placement *) ( extent + at ); + at += bytes; + dst.placements.count = cursor.count; + dst.placements.padding = 0; + dst.placements.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.placements.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Placement ) ); // trivially copyable, by construction + } + } + { + TableListCursor cursor = TableListElements( ctx, src.log ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( TableRef ) - 1 ) & ~( (int64_t) alignof( TableRef ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( TableRef ); + if ( at + bytes > capacity ) { return false; } + TableRef * placed = (TableRef *) ( extent + at ); + at += bytes; + dst.log.count = cursor.count; + dst.log.padding = 0; + dst.log.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.log.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( TableRef ) ); // trivially copyable, by construction + } + } + { + TableListCursor cursor = TableListElements( ctx, src.scores ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( int32_t ) - 1 ) & ~( (int64_t) alignof( int32_t ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( int32_t ); + if ( at + bytes > capacity ) { return false; } + int32_t * placed = (int32_t *) ( extent + at ); + at += bytes; + dst.scores.count = cursor.count; + dst.scores.padding = 0; + dst.scores.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.scores.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( int32_t ) ); // trivially copyable, by construction + } + } + return true; +} + +// MixedWireExtent: the extent Mixed's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool MixedWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0xd90a4e7682f799c5ull && field_kind == 14 ) // grades: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Grade ), (int64_t) alignof( Grade ), 30, 1, NULL, ids, reason ) ) { return false; } + continue; + } + if ( field_id == 0x4af2ed8470862ea8ull && field_kind == 14 ) // perms: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Perm ), (int64_t) alignof( Perm ), 9, 8, NULL, ids, reason ) ) { return false; } + continue; + } + if ( field_id == 0x732dfbcc9b0cf0bbull && field_kind == 14 ) // hits: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( Hit ), (int64_t) alignof( Hit ), 15, 1, NULL, ids, reason ) ) { return false; } + continue; + } + if ( field_id == 0x52f60c4caef0b768ull && field_kind == 14 ) // bounds: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( int32_t ), (int64_t) alignof( int32_t ), 4, 4, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// MixedExtentAt: the node extent Mixed's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as MixedExtentPack advances it (§2.8, §2.9). +template +inline bool MixedExtentAt( const Ctx & ctx, const Mixed & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.grades ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Grade ) - 1 ) & ~( (int64_t) alignof( Grade ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Grade ); // the whole array FIRST + } + { + TableListCursor cursor = TableListElements( ctx, value.perms ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Perm ) - 1 ) & ~( (int64_t) alignof( Perm ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Perm ); // the whole array FIRST + } + { + TableListCursor cursor = TableListElements( ctx, value.hits ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Hit ) - 1 ) & ~( (int64_t) alignof( Hit ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( Hit ); // the whole array FIRST + } + { + TableListCursor cursor = TableListElements( ctx, value.bounds ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( int32_t ) - 1 ) & ~( (int64_t) alignof( int32_t ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( int32_t ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t MixedExtent( const Ctx & ctx, const Mixed & value ) +{ + int64_t at = 0; + if ( !MixedExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// MixedExtentPack: carve Mixed's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset MixedExtentAt advances (§2.8, §2.9). +template +inline bool MixedExtentPack( const Ctx & ctx, const Mixed & src, Mixed & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.grades ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Grade ) - 1 ) & ~( (int64_t) alignof( Grade ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Grade ); + if ( at + bytes > capacity ) { return false; } + Grade * placed = (Grade *) ( extent + at ); + at += bytes; + dst.grades.count = cursor.count; + dst.grades.padding = 0; + dst.grades.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.grades.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Grade ) ); // trivially copyable, by construction + } + } + { + TableListCursor cursor = TableListElements( ctx, src.perms ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Perm ) - 1 ) & ~( (int64_t) alignof( Perm ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Perm ); + if ( at + bytes > capacity ) { return false; } + Perm * placed = (Perm *) ( extent + at ); + at += bytes; + dst.perms.count = cursor.count; + dst.perms.padding = 0; + dst.perms.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.perms.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Perm ) ); // trivially copyable, by construction + } + } + { + TableListCursor cursor = TableListElements( ctx, src.hits ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( Hit ) - 1 ) & ~( (int64_t) alignof( Hit ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( Hit ); + if ( at + bytes > capacity ) { return false; } + Hit * placed = (Hit *) ( extent + at ); + at += bytes; + dst.hits.count = cursor.count; + dst.hits.padding = 0; + dst.hits.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.hits.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( Hit ) ); // trivially copyable, by construction + } + } + { + TableListCursor cursor = TableListElements( ctx, src.bounds ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( int32_t ) - 1 ) & ~( (int64_t) alignof( int32_t ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( int32_t ); + if ( at + bytes > capacity ) { return false; } + int32_t * placed = (int32_t *) ( extent + at ); + at += bytes; + dst.bounds.count = cursor.count; + dst.bounds.padding = 0; + dst.bounds.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.bounds.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( int32_t ) ); // trivially copyable, by construction + } + } + return true; +} + +// ---- Save.placements: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Placement * SavePlacementsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool SavePlacementsErase( TableArena & arena, TableList & list, const Placement * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach SavePlacementsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Save.log: the builder's three (§2.9) ---- + +// ADD: the element is appended and handed back to fill. On a []*T that is +// the SLOT at null, which LogEntryEmplace fills as it fills any pointer slot, +// and a second slot may hold the same reference: two slots, one node. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline TableRef * SaveLogAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool SaveLogErase( TableArena & arena, TableList & list, const TableRef * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach SaveLogEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Save.scores: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline int32_t * SaveScoresAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool SaveScoresErase( TableArena & arena, TableList & list, const int32_t * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach SaveScoresEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Mixed.grades: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Grade * MixedGradesAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool MixedGradesErase( TableArena & arena, TableList & list, const Grade * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach MixedGradesEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Mixed.perms: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Perm * MixedPermsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool MixedPermsErase( TableArena & arena, TableList & list, const Perm * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach MixedPermsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Mixed.hits: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline Hit * MixedHitsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool MixedHitsErase( TableArena & arena, TableList & list, const Hit * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach MixedHitsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// ---- Mixed.bounds: the builder's three (§2.9) ---- + +// ADD: the element is appended at its declared defaults and handed back to +// fill. Nothing ever moves (§6.4), so the pointer stays valid while other +// elements arrive. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline int32_t * MixedBoundsAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool MixedBoundsErase( TableArena & arena, TableList & list, const int32_t * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach MixedBoundsEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// LogEntryNumber: number everything LogEntry POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool LogEntryNumber( const Ctx & ctx, TableNumbering & numbering, const LogEntry & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// LogEntryPackMeasure: the packed region bytes of everything LogEntry POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t LogEntryPackMeasure( const Ctx & ctx, TablePackMap & seen, const LogEntry & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// LogEntryPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool LogEntryPackEdges( const Ctx & ctx, TablePackMap & seen, const LogEntry & src, LogEntry & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool LogEntryPack( const Ctx & ctx, TablePackMap & seen, const LogEntry & src, LogEntry & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( LogEntry ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( LogEntry ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !LogEntryExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return LogEntryPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool LogEntryPackEdges( const Ctx & ctx, TablePackMap & seen, const LogEntry & src, LogEntry & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// SaveNumber: number everything Save POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool SaveNumber( const Ctx & ctx, TableNumbering & numbering, const Save & value ) +{ + { // log: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_log = TableListElements( ctx, value.log ); + if ( !cursor_log.ok ) { return false; } + for ( int32_t i = 0; i < cursor_log.count; i++ ) + { + { + const LogEntry * pointee = LogEntryAt( ctx, cursor_log[i] ); // log + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( numbering.seen, (const void *) pointee, + (int64_t) ( numbering.count + 2 ), taken, slot ); // its index, if this is its first visit + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + } + else + { + TableNodeEntry node; + node.node = (const void *) pointee; + node.type_id = 0x5e781536ac58825full; // fnv1a64( "LogEntry" ) + node.type_slot = 50; // its slot in the unit's vocabulary (§3.3) + node.measure = &TableNodeMeasureThunk; + node.save = &TableNodeSaveThunk; + if ( !TableNumberingAppend( numbering, node ) ) { return false; } + if ( !LogEntryNumber( ctx, numbering, *pointee ) ) { return false; } + TablePackMapClose( numbering.seen, (const void *) pointee, slot ); + } + } + } + } + } + return true; +} + +// SavePackMeasure: the packed region bytes of everything Save POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t SavePackMeasure( const Ctx & ctx, TablePackMap & seen, const Save & value ) +{ + int64_t bytes = 0; + { // log: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_log = TableListElements( ctx, value.log ); + if ( !cursor_log.ok ) { return -1; } + for ( int32_t i = 0; i < cursor_log.count; i++ ) + { + { + const LogEntry * pointee = LogEntryAt( ctx, cursor_log[i] ); // log + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, 0, taken, slot ); + if ( entry == NULL ) { return -1; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return -1; } // a data cycle + } + else + { + int64_t inner = LogEntryPackMeasure( ctx, seen, *pointee ); + if ( inner < 0 ) { return -1; } + TablePackMapClose( seen, (const void *) pointee, slot ); + bytes += TableAlignUp64( (int64_t) sizeof( LogEntry ) ) + inner; + } + } + } + } + } + return bytes; +} + +// SavePack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool SavePackEdges( const Ctx & ctx, TablePackMap & seen, const Save & src, Save & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool SavePack( const Ctx & ctx, TablePackMap & seen, const Save & src, Save & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Save ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Save ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !SaveExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return SavePackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool SavePackEdges( const Ctx & ctx, TablePackMap & seen, const Save & src, Save & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + { // log: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_log = TableListElements( ctx, src.log ); + if ( !cursor_log.ok ) { return false; } + TableRef * placed_log = (TableRef *) ( dst.log.elements.value != 0 ? ( (uint8_t *) &dst.log.elements + dst.log.elements.value ) : NULL ); + for ( int32_t i = 0; i < cursor_log.count; i++ ) + { + { + placed_log[i].value = 0; // log + const LogEntry * pointee = LogEntryAt( ctx, cursor_log[i] ); + if ( pointee != NULL ) + { + int64_t at = TableAlignUp64( used ); // where it WOULD land, if this is its first visit + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, at, taken, slot ); + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + placed_log[i].value = (int64_t) ( ( base + entry->offset ) - (const uint8_t *) &placed_log[i] ); // the one body it already has + } + else + { + if ( at + (int64_t) sizeof( LogEntry ) > capacity ) { return false; } + used = at + TableAlignUp64( (int64_t) sizeof( LogEntry ) ); + LogEntry * child = new ( base + at ) LogEntry; // lifetime only: the Pack below memcpy's the whole node over it + placed_log[i].value = (int64_t) ( ( base + at ) - (const uint8_t *) &placed_log[i] ); + if ( !LogEntryPack( ctx, seen, *pointee, *child, base, capacity, used ) ) { return false; } + TablePackMapClose( seen, (const void *) pointee, slot ); + } + } + } + } + } + return true; +} + +// MixedNumber: number everything Mixed POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool MixedNumber( const Ctx & ctx, TableNumbering & numbering, const Mixed & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// MixedPackMeasure: the packed region bytes of everything Mixed POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t MixedPackMeasure( const Ctx & ctx, TablePackMap & seen, const Mixed & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// MixedPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool MixedPackEdges( const Ctx & ctx, TablePackMap & seen, const Mixed & src, Mixed & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool MixedPack( const Ctx & ctx, TablePackMap & seen, const Mixed & src, Mixed & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Mixed ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Mixed ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !MixedExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return MixedPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool MixedPackEdges( const Ctx & ctx, TablePackMap & seen, const Mixed & src, Mixed & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// ---- Save: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: SaveBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Save is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct SaveBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + SaveBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~SaveBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + SaveBuilder( const SaveBuilder & ) = delete; + SaveBuilder & operator=( const SaveBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Save * GetRoot() { return arena.locked ? NULL : (Save *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Save * AsConst() const { return (const Save *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool SaveBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Save & root = *(const Save *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = SavePackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = SaveExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + Save * destination = new ( packed ) Save; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !SavePack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Save on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// SaveNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t SaveNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + case 0x5e781536ac58825full: return TableAlignUp64( (int64_t) sizeof( LogEntry ) ); // LogEntry + default: break; + } + return -1; +} + +// SaveNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void SaveNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + switch ( type_id ) + { + case 0x5e781536ac58825full: { LogEntry * node = new ( at ) LogEntry; LogEntryReset( *node ); break; } // LogEntry + default: break; + } +} + +// SaveNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t SaveNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + case 0x5e781536ac58825full: return TableAlignUp64( (int64_t) sizeof( LogEntry ) ); // LogEntry + default: break; + } + return 0; +} + +// SaveNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t SaveNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + switch ( type_id ) + { + case 0x5e781536ac58825full: return (uint32_t) worker.Alloc().ref.value; // LogEntry + default: break; + } + return 0; +} + +// SaveNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void SaveNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = SaveNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? SaveNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + case 0x5e781536ac58825full: LogEntryLoadBody( r, *(LogEntry *) at ); break; // LogEntry + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool SaveNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Save & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return SaveNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t SaveMeasureWire( const Ctx & ctx, const Save & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( SaveNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = SaveMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t SaveSaveWire( const Ctx & ctx, const Save & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !SaveNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = SaveSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == SaveMeasure( root ) +} + +inline int64_t SaveMeasure( const Save * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SaveMeasureWire( ctx, *root, allocator ); +} + +inline int64_t SaveSave( const Save * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SaveSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t SaveMeasure( const SaveBuilder & builder ) +{ + if ( builder.region != NULL ) { return SaveMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SaveMeasureWire( ctx, *(const Save *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t SaveSave( const SaveBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return SaveSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SaveSaveWire( ctx, *(const Save *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t SaveMeasureMessage( const Save * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SaveMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t SaveSaveMessage( const Save * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SaveSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t SaveMeasureMessage( const SaveBuilder & builder ) +{ + if ( builder.region != NULL ) { return SaveMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return SaveMeasureWire( ctx, *(const Save *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t SaveSaveMessage( const SaveBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return SaveSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return SaveSaveWire( ctx, *(const Save *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// SaveLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t SaveLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !SaveWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SaveNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// SaveLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Save * SaveLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Save ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !SaveWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SaveNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x33f85f24c0f5f008ull; + Save * root = new ( region ) Save; // lifetime only: LoadBody's first act is SaveReset + SaveReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = SaveNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + SaveNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SaveNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Save ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + SaveLoadBody( r, nodes, *root ); + return root; +} + +// SaveLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t SaveLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !SaveWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SaveNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// SaveLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Save * SaveLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Save ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !SaveWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = SaveNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0x33f85f24c0f5f008ull; + Save * root = new ( region ) Save; // lifetime only: LoadBody's first act is SaveReset + SaveReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Save ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = SaveNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + SaveNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SaveNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Save ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + SaveLoadBody( r, nodes, *root ); + return root; +} + +// SaveLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool SaveLoadBuilder( SaveBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Save * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0x33f85f24c0f5f008ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = SaveNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + SaveNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = SaveLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- Mixed: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: MixedBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Mixed is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct MixedBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + MixedBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~MixedBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + MixedBuilder( const MixedBuilder & ) = delete; + MixedBuilder & operator=( const MixedBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Mixed * GetRoot() { return arena.locked ? NULL : (Mixed *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Mixed * AsConst() const { return (const Mixed *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool MixedBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Mixed & root = *(const Mixed *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = MixedPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = MixedExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + Mixed * destination = new ( packed ) Mixed; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !MixedPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Mixed on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// MixedNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t MixedNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + default: break; + } + return -1; +} + +// MixedNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void MixedNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + (void) at; + switch ( type_id ) + { + default: break; + } +} + +// MixedNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t MixedNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + default: break; + } + return 0; +} + +// MixedNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t MixedNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + (void) worker; + switch ( type_id ) + { + default: break; + } + return 0; +} + +// MixedNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void MixedNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = MixedNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? MixedNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool MixedNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Mixed & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return MixedNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t MixedMeasureWire( const Ctx & ctx, const Mixed & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( MixedNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = MixedMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t MixedSaveWire( const Ctx & ctx, const Mixed & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !MixedNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = MixedSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == MixedMeasure( root ) +} + +inline int64_t MixedMeasure( const Mixed * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return MixedMeasureWire( ctx, *root, allocator ); +} + +inline int64_t MixedSave( const Mixed * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return MixedSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t MixedMeasure( const MixedBuilder & builder ) +{ + if ( builder.region != NULL ) { return MixedMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return MixedMeasureWire( ctx, *(const Mixed *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t MixedSave( const MixedBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return MixedSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return MixedSaveWire( ctx, *(const Mixed *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t MixedMeasureMessage( const Mixed * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return MixedMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t MixedSaveMessage( const Mixed * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return MixedSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t MixedMeasureMessage( const MixedBuilder & builder ) +{ + if ( builder.region != NULL ) { return MixedMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return MixedMeasureWire( ctx, *(const Mixed *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t MixedSaveMessage( const MixedBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return MixedSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return MixedSaveWire( ctx, *(const Mixed *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// MixedLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t MixedLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !MixedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = MixedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// MixedLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Mixed * MixedLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Mixed ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !MixedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = MixedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xbb86c6c96f598bb8ull; + Mixed * root = new ( region ) Mixed; // lifetime only: LoadBody's first act is MixedReset + MixedReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = MixedNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + MixedNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + MixedNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Mixed ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + MixedLoadBody( r, nodes, *root ); + return root; +} + +// MixedLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t MixedLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !MixedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = MixedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// MixedLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Mixed * MixedLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Mixed ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !MixedWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = MixedNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xbb86c6c96f598bb8ull; + Mixed * root = new ( region ) Mixed; // lifetime only: LoadBody's first act is MixedReset + MixedReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Mixed ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = MixedNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + MixedNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + MixedNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Mixed ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + MixedLoadBody( r, nodes, *root ); + return root; +} + +// MixedLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool MixedLoadBuilder( MixedBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Mixed * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0xbb86c6c96f598bb8ull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = MixedNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + MixedNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = MixedLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- the cooked form: point at a cook (docs/SPEC-TABLES.md §7) ---- + +// PlacementOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Placement IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Placement * PlacementOpen( const void * bytes, uint64_t length ) +{ + return (const Placement *) TableCookOpen( bytes, length, (uint64_t) sizeof( Placement ), (uint64_t) alignof( Placement ) ); +} + +// LogEntryOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// LogEntry IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const LogEntry * LogEntryOpen( const void * bytes, uint64_t length ) +{ + return (const LogEntry *) TableCookOpen( bytes, length, (uint64_t) sizeof( LogEntry ), (uint64_t) alignof( LogEntry ) ); +} + +// SaveOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH SaveAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Save * SaveOpen( const void * bytes, uint64_t length ) +{ + return (const Save *) TableCookOpen( bytes, length, (uint64_t) sizeof( Save ), (uint64_t) alignof( Save ) ); +} + +// PointOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Point IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Point * PointOpen( const void * bytes, uint64_t length ) +{ + return (const Point *) TableCookOpen( bytes, length, (uint64_t) sizeof( Point ), (uint64_t) alignof( Point ) ); +} + +// MixedOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH MixedAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Mixed * MixedOpen( const void * bytes, uint64_t length ) +{ + return (const Mixed *) TableCookOpen( bytes, length, (uint64_t) sizeof( Mixed ), (uint64_t) alignof( Mixed ) ); +} + +// ---- the cooked form: WRITE a cook (docs/SPEC-TABLES.md §7.6) ---- +// +// The bytes are `schema cook`'s, and the tool stays the reference: the two +// writers are held to one file, byte for byte, in both byte orders. A cook is +// content-addressed by (asset hash, build version), so two writers of one +// instance produce ONE artifact or the pair means nothing. + +inline void PlacementCookBody( uint8_t * at, const Placement & value, TableByteOrder order ); +inline void LogEntryCookBody( uint8_t * at, const LogEntry & value, TableByteOrder order ); +template inline bool SaveCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Save & value, TableByteOrder order ); +inline void PointCookBody( uint8_t * at, const Point & value, TableByteOrder order ); +template inline bool MixedCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Mixed & value, TableByteOrder order ); + +inline void PlacementCookBody( uint8_t * at, const Placement & value, TableByteOrder order ) +{ + { uint32_t bits = 0; memcpy( &bits, &value.x, 4 ); table_cook_put( at + 0, (uint64_t) bits, 4, order ); } + { uint32_t bits = 0; memcpy( &bits, &value.y, 4 ); table_cook_put( at + 4, (uint64_t) bits, 4, order ); } + table_cook_put( at + 8, (uint64_t) value.model, 4, order ); +} + +inline void LogEntryCookBody( uint8_t * at, const LogEntry & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.tick, 4, order ); +} + +template inline bool SaveCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Save & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + (void) value; + table_cook_put( at + 0, 0, 8, order ); // placements: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, 0, 8, order ); // log: the array's delta, filled by the extent writer + table_cook_put( at + 24, 0, 4, order ); // and its count + table_cook_put( at + 32, 0, 8, order ); // scores: the array's delta, filled by the extent writer + table_cook_put( at + 40, 0, 4, order ); // and its count + return true; +} + +inline void PointCookBody( uint8_t * at, const Point & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.x, 4, order ); + table_cook_put( at + 4, (uint64_t) value.y, 4, order ); +} + +template inline bool MixedCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Mixed & value, TableByteOrder order ) +{ + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + (void) value; + table_cook_put( at + 0, 0, 8, order ); // grades: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + table_cook_put( at + 16, 0, 8, order ); // perms: the array's delta, filled by the extent writer + table_cook_put( at + 24, 0, 4, order ); // and its count + table_cook_put( at + 32, 0, 8, order ); // hits: the array's delta, filled by the extent writer + table_cook_put( at + 40, 0, 4, order ); // and its count + table_cook_put( at + 48, 0, 8, order ); // bounds: the array's delta, filled by the extent writer + table_cook_put( at + 56, 0, 4, order ); // and its count + return true; +} + +template inline bool PlacementCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Placement & value, TableByteOrder order ); +template inline bool LogEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const LogEntry & value, TableByteOrder order ); +template inline bool SaveCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Save & value, TableByteOrder order ); +template inline bool PointCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Point & value, TableByteOrder order ); +template inline bool MixedCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Mixed & value, TableByteOrder order ); + +// PlacementCookExtent: Placement's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool PlacementCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Placement & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// LogEntryCookExtent: LogEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool LogEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const LogEntry & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// SaveCookExtent: Save's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SaveCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Save & value, TableByteOrder order ) +{ + { // placements: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.placements ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( Placement ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 12; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + PlacementCookBody( array + i * 12, cursor[i], order ); + } + } + { // log: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.log ); + if ( !cursor.ok ) { return false; } + at = ( at + 7 ) & ~(int64_t) 7; // at alignof( TableRef ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 8; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 16, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 16 ) ) : 0, 8, order ); + table_cook_put( record + 24, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + if ( !table_cook_ref( region, array + i * 8, (const void *) LogEntryAt( ctx, cursor[i] ), order ) ) { return false; } + } + } + { // scores: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.scores ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( int32_t ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 4; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 32, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 32 ) ) : 0, 8, order ); + table_cook_put( record + 40, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + table_cook_put( array + i * 4, (uint64_t) cursor[i], 4, order ); + } + } + return true; +} + +// PointCookExtent: Point's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool PointCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Point & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// MixedCookExtent: Mixed's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool MixedCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Mixed & value, TableByteOrder order ) +{ + (void) region; // a table element's and an entry's references resolve through their own bodies + { // grades: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.grades ); + if ( !cursor.ok ) { return false; } + at = ( at + 0 ) & ~(int64_t) 0; // at alignof( Grade ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 1; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + table_cook_put( array + i * 1, (uint64_t) cursor[i], 1, order ); + } + } + { // perms: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.perms ); + if ( !cursor.ok ) { return false; } + at = ( at + 7 ) & ~(int64_t) 7; // at alignof( Perm ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 8; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 16, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 16 ) ) : 0, 8, order ); + table_cook_put( record + 24, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + table_cook_put( array + i * 8, (uint64_t) cursor[i], 8, order ); // a mask rides raw, in every target + } + } + { // hits: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.hits ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( Hit ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 12; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 32, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 32 ) ) : 0, 8, order ); + table_cook_put( record + 40, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + { + table_cook_put( array + i * 12, (uint64_t) cursor[i].type, 1, order ); // the tag; None is the tag alone + switch ( cursor[i].type ) + { + case HitType::Point: PointCookBody( array + i * 12 + 4, cursor[i].point, order ); break; + case HitType::Damage: + { + table_cook_put( array + i * 12 + 4, (uint64_t) cursor[i].damage, 4, order ); + break; + } + default: break; // every byte outside the set arm stays zero + } + } + } + } + { // bounds: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.bounds ); + if ( !cursor.ok ) { return false; } + at = ( at + 3 ) & ~(int64_t) 3; // at alignof( int32_t ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 4; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 48, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 48 ) ) : 0, 8, order ); + table_cook_put( record + 56, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + table_cook_put( array + i * 4, (uint64_t) cursor[i], 4, order ); + } + } + return true; +} + +// PlacementCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool PlacementCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Placement & value, TableByteOrder order ) +{ + PlacementCookBody( at, value, order ); + int64_t extent_at = 0; + return PlacementCookExtent( ctx, region, at + 16, extent_at, at, value, order ); +} + +// LogEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool LogEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const LogEntry & value, TableByteOrder order ) +{ + LogEntryCookBody( at, value, order ); + int64_t extent_at = 0; + return LogEntryCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// SaveCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool SaveCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Save & value, TableByteOrder order ) +{ + if ( !SaveCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return SaveCookExtent( ctx, region, at + 48, extent_at, at, value, order ); +} + +// PointCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool PointCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Point & value, TableByteOrder order ) +{ + PointCookBody( at, value, order ); + int64_t extent_at = 0; + return PointCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// MixedCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool MixedCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Mixed & value, TableByteOrder order ) +{ + if ( !MixedCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return MixedCookExtent( ctx, region, at + 64, extent_at, at, value, order ); +} + +// PlacementCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Placement IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t PlacementCookMeasure( const Placement & value ) +{ + (void) value; + return 96; // 64 header + 16 data + 16 attribution +} + +// PlacementCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract PlacementMeasure/PlacementSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool PlacementCook( const Placement & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) PlacementCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 16, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + PlacementCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 80, 0, 8, order ); + table_cook_put( raw + 88, 0x41f721f1b93aea80ull, 8, order ); + return true; +} + +// LogEntryCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// LogEntry IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t LogEntryCookMeasure( const LogEntry & value ) +{ + (void) value; + return 88; // 64 header + 8 data + 16 attribution +} + +// LogEntryCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract LogEntryMeasure/LogEntrySave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool LogEntryCook( const LogEntry & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) LogEntryCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 8, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + LogEntryCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 72, 0, 8, order ); + table_cook_put( raw + 80, 0x5e781536ac58825full, 8, order ); + return true; +} + +// SaveCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool SaveCookLayout( const Ctx & ctx, const Save & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = SaveExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 48 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + case 0x5e781536ac58825full: size = 4; node_align = 4; break; // LogEntry + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// SaveCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t SaveCookMeasureFrom( const Ctx & ctx, const Save & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( SaveNumberFrom( ctx, numbering, root ) && SaveCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// SaveCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool SaveCookFrom( const Ctx & ctx, const Save & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = SaveNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && SaveCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = SaveCookNode( ctx, region, region.base, root, order ); + for ( int64_t k = 0; ok && k < numbering.count; k++ ) + { + uint8_t * at = region.base + region.offsets[k + 1]; + const void * node = numbering.entries[k].node; + switch ( numbering.entries[k].type_id ) + { + case 0x5e781536ac58825full: ok = LogEntryCookNode( ctx, region, at, *(const LogEntry *) node, order ); break; // LogEntry + default: ok = false; break; + } + } + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0x33f85f24c0f5f008ull, 8, order ); // the root: fnv1a64( "Save" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// SaveCookMeasure / SaveCook over a REGION root — a locked builder's AsConst, a +// region SaveLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t SaveCookMeasure( const Save * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return SaveCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool SaveCook( const Save * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return SaveCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t SaveCookMeasure( const SaveBuilder & builder ) +{ + if ( builder.region != NULL ) { return SaveCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SaveCookMeasureFrom( ctx, *(const Save *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool SaveCook( const SaveBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return SaveCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return SaveCookFrom( ctx, *(const Save *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// PointCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Point IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t PointCookMeasure( const Point & value ) +{ + (void) value; + return 88; // 64 header + 8 data + 16 attribution +} + +// PointCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract PointMeasure/PointSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool PointCook( const Point & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) PointCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 8, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + PointCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 72, 0, 8, order ); + table_cook_put( raw + 80, 0x8a439296ced9ed11ull, 8, order ); + return true; +} + +// MixedCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool MixedCookLayout( const Ctx & ctx, const Mixed & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = MixedExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 64 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// MixedCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t MixedCookMeasureFrom( const Ctx & ctx, const Mixed & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( MixedNumberFrom( ctx, numbering, root ) && MixedCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// MixedCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool MixedCookFrom( const Ctx & ctx, const Mixed & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = MixedNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && MixedCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = MixedCookNode( ctx, region, region.base, root, order ); + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0xbb86c6c96f598bb8ull, 8, order ); // the root: fnv1a64( "Mixed" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// MixedCookMeasure / MixedCook over a REGION root — a locked builder's AsConst, a +// region MixedLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t MixedCookMeasure( const Mixed * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return MixedCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool MixedCook( const Mixed * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return MixedCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t MixedCookMeasure( const MixedBuilder & builder ) +{ + if ( builder.region != NULL ) { return MixedCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return MixedCookMeasureFrom( ctx, *(const Mixed *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool MixedCook( const MixedBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return MixedCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return MixedCookFrom( ctx, *(const Mixed *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ---- relocatability, enforced: the wire is a pure length-prefixed +// stream AND the decoded storage is pointer-free — every closure type +// must stay trivially copyable and standard-layout, so instances can be +// memcpy'd, mmap'd, shared across processes, and walked through +// descriptor offsets. A failure here means a pointer, virtual or +// non-trivial member crept into generated storage. +// +// They ask the COMPILER ITSELF, which is what every C++ standard library +// answers the same two questions with — and it costs this header no +// include at all. +// A pointer FIELD is a TableRef — eight bytes and no address — so the +// property holds in BOTH forms: a fixed-size table is one relocatable +// struct, and a packed region is one relocatable block whose references +// are self-relative and therefore survive a plain memcpy. +static_assert( __is_trivially_copyable( Placement ), "Placement must stay relocatable" ); +static_assert( __is_standard_layout( Placement ), "Placement must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( LogEntry ), "LogEntry must stay relocatable" ); +static_assert( __is_standard_layout( LogEntry ), "LogEntry must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Save ), "Save must stay relocatable" ); +static_assert( __is_standard_layout( Save ), "Save must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Point ), "Point must stay relocatable" ); +static_assert( __is_standard_layout( Point ), "Point must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Mixed ), "Mixed must stay relocatable" ); +static_assert( __is_standard_layout( Mixed ), "Mixed must stay standard-layout for offsetof" ); + +// ---- the cook's layout contract (docs/SPEC-TABLES.md §20.3) ---- +// +// The compiler derived every number below from the declaration and folded it +// into the BUILD VERSION; these asserts are this compiler saying whether it +// agrees. The model is not self-evidently right — on 32-bit System V +// alignof(uint64_t) is 4, not 8 — which is precisely why it is asserted +// rather than assumed. +static_assert( sizeof( Placement ) == 12, "Placement's sizeof moved: the build version was taken over 12, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Placement ) == 4, "Placement's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Placement, x ) == 0, "Placement's field x moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Placement, y ) == 4, "Placement's field y moved: the build version was taken over offset 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Placement, model ) == 8, "Placement's field model moved: the build version was taken over offset 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( LogEntry ) == 4, "LogEntry's sizeof moved: the build version was taken over 4, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( LogEntry ) == 4, "LogEntry's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( LogEntry, tick ) == 0, "LogEntry's field tick moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Save ) == 48, "Save's sizeof moved: the build version was taken over 48, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Save ) == 8, "Save's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Save, placements ) == 0, "Save's field placements moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Save, log ) == 16, "Save's field log moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Save, scores ) == 32, "Save's field scores moved: the build version was taken over offset 32 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Point ) == 8, "Point's sizeof moved: the build version was taken over 8, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Point ) == 4, "Point's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Point, x ) == 0, "Point's field x moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Point, y ) == 4, "Point's field y moved: the build version was taken over offset 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Mixed ) == 64, "Mixed's sizeof moved: the build version was taken over 64, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Mixed ) == 8, "Mixed's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Mixed, grades ) == 0, "Mixed's field grades moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Mixed, perms ) == 16, "Mixed's field perms moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Mixed, hits ) == 32, "Mixed's field hits moved: the build version was taken over offset 32 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Mixed, bounds ) == 48, "Mixed's field bounds moved: the build version was taken over offset 48 (docs/SPEC-TABLES.md §20.3)" ); + +static_assert( alignof( Placement ) <= kTableAlign, "Save.placements: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( TableRef ) <= kTableAlign, "Save.log: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( int32_t ) <= kTableAlign, "Save.scores: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( Grade ) <= kTableAlign, "Mixed.grades: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( Perm ) <= kTableAlign, "Mixed.perms: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( Hit ) <= kTableAlign, "Mixed.hits: an unbounded array's element alignment must fit the arena's" ); +static_assert( alignof( int32_t ) <= kTableAlign, "Mixed.bounds: an unbounded array's element alignment must fit the arena's" ); + +// ---- reflection descriptors (tables only, docs/SPEC-TABLES.md) ---- + +inline const TableTypeInfo * PlacementTableType(); +inline const TableTypeInfo * LogEntryTableType(); +inline const TableTypeInfo * SaveTableType(); +inline const TableTypeInfo * PointTableType(); +inline const TableTypeInfo * MixedTableType(); +// The descriptors are CONSTANT-INITIALISED data, and a field's target is +// the ADDRESS of another descriptor. These declarations are what let a +// self- or mutually-referential graph — Node naming itself through *Node — +// be expressed as constant data instead of a lazy link, which could not +// have been written race-free OR recursion-safe. The whole reflection +// surface is therefore immutable: read it from any thread, any time. +extern const TableTypeInfo PlacementTableInfo; +extern const TableTypeInfo LogEntryTableInfo; +extern const TableTypeInfo SaveTableInfo; +extern const TableTypeInfo PointTableInfo; +extern const TableTypeInfo MixedTableInfo; + +inline const TableFieldInfo PlacementTableFields[] = { + { "x", "x", "float32", 0xaf63f54c86021707ull, 10, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Placement, x ), (uint32_t) sizeof( Placement::x ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "y", "y", "float32", 0xaf63f44c86021554ull, 10, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Placement, y ), (uint32_t) sizeof( Placement::y ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "model", "model", "uint32", 0x9de543933e6e703aull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Placement, model ), (uint32_t) sizeof( Placement::model ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo PlacementTableInfo = { "Placement", (uint32_t) sizeof( Placement ), 3, PlacementTableFields, +[]( void * p ) { PlacementReset( *(Placement *) p ); }, false }; +inline const TableTypeInfo * PlacementTableType() { return &PlacementTableInfo; } + +inline const TableFieldInfo LogEntryTableFields[] = { + { "tick", "tick", "uint32", 0x1e7683ef2ebc7684ull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( LogEntry, tick ), (uint32_t) sizeof( LogEntry::tick ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo LogEntryTableInfo = { "LogEntry", (uint32_t) sizeof( LogEntry ), 1, LogEntryTableFields, +[]( void * p ) { LogEntryReset( *(LogEntry *) p ); }, false }; +inline const TableTypeInfo * LogEntryTableType() { return &LogEntryTableInfo; } + +inline const TableFieldInfo SaveTableFields[] = { + { "placements", "placements", "Placement", 0xd24733aa574d4b09ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Save, placements ), (uint32_t) sizeof( Placement ), (uint32_t) offsetof( Save, placements.count ), 0xffffffffu, &PlacementTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "log", "log", "LogEntry", 0x125073191daf5431ull, 17, true, true, []( const void * slot ) -> const void * { return (const void *) LogEntryAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) LogEntryEmplace( worker, *(TableRef *) slot ); }, true, false, 0, (uint32_t) offsetof( Save, log ), (uint32_t) sizeof( TableRef ), (uint32_t) offsetof( Save, log.count ), 0xffffffffu, &LogEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "scores", "scores", "int32", 0x01986b0b27400fb2ull, 4, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Save, scores ), (uint32_t) sizeof( int32_t ), (uint32_t) offsetof( Save, scores.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, +}; +inline const TableTypeInfo SaveTableInfo = { "Save", (uint32_t) sizeof( Save ), 3, SaveTableFields, +[]( void * p ) { SaveReset( *(Save *) p ); }, true }; +inline const TableTypeInfo * SaveTableType() { return &SaveTableInfo; } + +inline const TableFieldInfo PointTableFields[] = { + { "x", "x", "int32", 0xaf63f54c86021707ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Point, x ), (uint32_t) sizeof( Point::x ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "y", "y", "int32", 0xaf63f44c86021554ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Point, y ), (uint32_t) sizeof( Point::y ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo PointTableInfo = { "Point", (uint32_t) sizeof( Point ), 2, PointTableFields, +[]( void * p ) { PointReset( *(Point *) p ); }, false }; +inline const TableTypeInfo * PointTableType() { return &PointTableInfo; } + +inline const TableFieldInfo MixedTableFields[] = { + { "grades", "grades", "Grade", 0xd90a4e7682f799c5ull, 30, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Mixed, grades ), (uint32_t) sizeof( Grade ), (uint32_t) offsetof( Mixed, grades.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, 3, +[]( uint64_t v ) { return EnumName( Grade( v ) ); }, +[]( uint64_t v ) -> uint64_t { uint64_t id = 0; TableEnumId( Grade( v ), id ); return id; }, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "perms", "perms", "Perm", 0x4af2ed8470862ea8ull, 9, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Mixed, perms ), (uint32_t) sizeof( Perm ), (uint32_t) offsetof( Mixed, perms.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, 2, +[]( uint64_t v ) { return FlagNamePerm( (int) v ); }, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "hits", "hits", "Hit", 0x732dfbcc9b0cf0bbull, 15, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Mixed, hits ), (uint32_t) sizeof( Hit ), (uint32_t) offsetof( Mixed, hits.count ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, 2, +[]( uint64_t v ) -> const char * { switch ( v ) { case 0: return "None"; case 1: return "point"; case 2: return "damage"; default: return "???"; } }, +[]( uint64_t v ) -> uint64_t { switch ( v ) { case 0: return 0; case 1: return 0x73feab3544c345b1ull; case 2: return 0x7f6308be8ab37fc0ull; default: return 0; } }, NULL, NULL, NULL, +[]() -> const TableUnionInfo * { static const TableFieldInfo arm_fields_Hit[] = { { "damage", "damage", "int32", 0x7f6308be8ab37fc0ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Hit, damage ), (uint32_t) sizeof( Hit::damage ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; static const TableUnionArmInfo arms[] = { { 0, NULL, NULL, 0 }, { (uint32_t) offsetof( Hit, point ), &PointTableInfo, NULL, 8 }, { (uint32_t) offsetof( Hit, damage ), NULL, &arm_fields_Hit[0], 4 }, }; static const TableUnionInfo info = { (uint32_t) offsetof( Hit, type ), (uint32_t) sizeof( Hit::type ), arms }; return &info; }, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "bounds", "bounds", "int32", 0x52f60c4caef0b768ull, 4, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Mixed, bounds ), (uint32_t) sizeof( int32_t ), (uint32_t) offsetof( Mixed, bounds.count ), 0xffffffffu, NULL, true, 0.0, 100.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, +}; +inline const TableTypeInfo MixedTableInfo = { "Mixed", (uint32_t) sizeof( Mixed ), 4, MixedTableFields, +[]( void * p ) { MixedReset( *(Mixed *) p ); }, true }; +inline const TableTypeInfo * MixedTableType() { return &MixedTableInfo; } + +// ---- the text form (docs/SPEC-TABLES.md §16) ---- + +// Placement in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// SaveTable.cpp; link it to use them. +bool PlacementFromJson( Placement & value, const char * text, int64_t bytes, TableReport * report ); +int64_t PlacementToJsonMeasure( const Placement & value ); +int64_t PlacementToJson( const Placement & value, char * buffer, int64_t capacity ); + +// LogEntry in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// SaveTable.cpp; link it to use them. +bool LogEntryFromJson( LogEntry & value, const char * text, int64_t bytes, TableReport * report ); +int64_t LogEntryToJsonMeasure( const LogEntry & value ); +int64_t LogEntryToJson( const LogEntry & value, char * buffer, int64_t capacity ); + +// Save in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in SaveTable.cpp; link it to use them. +bool SaveFromJson( SaveBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t SaveToJsonMeasure( const Save * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t SaveToJson( const Save * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +// Point in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// SaveTable.cpp; link it to use them. +bool PointFromJson( Point & value, const char * text, int64_t bytes, TableReport * report ); +int64_t PointToJsonMeasure( const Point & value ); +int64_t PointToJson( const Point & value, char * buffer, int64_t capacity ); + +// Mixed in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in SaveTable.cpp; link it to use them. +bool MixedFromJson( MixedBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t MixedToJsonMeasure( const Mixed * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t MixedToJson( const Mixed * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/SharedTable.cpp b/testdata/golden/tables/lists/SharedTable.cpp new file mode 100644 index 000000000..5cba2ef74 --- /dev/null +++ b/testdata/golden/tables/lists/SharedTable.cpp @@ -0,0 +1,3134 @@ +// Code generated by the schema compiler from Shared.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — the TABLE wire's text form (docs/SPEC-TABLES.md §16). +// Compile this file to use FromJson / ToJson; a project that +// never reads or writes a text does not compile it and pays nothing. + +#include "SharedTable.h" + +#include // the text form: number formatting +#include // the text form: exact number conversion +#include // the text form: the runtime's decimal point + +// The guard is not vestigial. Several listdemo Table.cpp files may be +// concatenated into ONE translation unit — a unity build — and without it +// each would redefine the walk. It is also why the walk's functions may be +// weak (vague linkage) across separate objects: ODR requires their +// definitions to be token-identical, and the generic-walk gate is what +// proves that, byte for byte, across every generated .cpp. +#ifndef LISTDEMO_SCHEMA_TABLE_JSON +#define LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +// ---- the pointer adapters (docs/SPEC-TABLES.md §16.7) ---- +// +// The walk below is ONE walk, byte-identical in every generated .cpp, and a +// pointer is the one kind it cannot walk alone: reading one needs the +// builder's arena and writing one needs a region's deref, and neither exists +// in a unit that declares no pointer. So the walk calls these three and does +// not define them. A unit with no pointer defines them as stubs no field ever +// reaches; a pointered unit defines them in the graph half that follows the +// walk. + +struct TableJsonIn; +struct TableJsonOut; + +// a pointer field's object, or the `&node` reference standing in for it, into +// the slot; the cursor is on the opening brace +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// the node a pointer slot names, in place — or as `&node` when it is shared +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// the FIRST key of an object the walk is skipping begins with `&`: the cursor is +// on its value. A dropped definition still takes its label (§16.7); a fixed reader +// skips the value whole, as it skips everything else it does not place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); + +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- +// +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map +inline bool TableJsonIsMap( const TableFieldInfo * f ); +// the map as a plain JSON object keyed by the KEY, in ASCENDING key order +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that object back into the slot, in whatever order the text gives it +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); + +// ---- json walk: begin ---- +// +// The TEXT form (docs/SPEC-TABLES.md §16): one table, one text, one walk over the +// reflection descriptors (§8). Reading fills ONE caller-owned instance and +// allocates nothing beyond it; writing targets a caller buffer with the +// wire's measure/write symmetry. Everything AROUND this — which file goes +// with which instance, what key an instance is filed under, how instances +// link into a root table's collections — is a packer's opinion and stays +// with the tool that holds it. +// +// The dialect: trailing commas are accepted on read (the authoring files +// this exists for carry them) and never written; comments are not JSON and +// are refused; unknown keys are skipped and counted; a duplicate key is +// last-wins and counted; a key present with the wrong JSON type is skipped +// and counted, never coerced. + +static const int32_t kTableJsonMaxDepth = 128; + +// A key longer than this cannot name a field, so it is skipped as unknown. +static const int32_t kTableJsonMaxKey = 256; + +// The longest numeric token the walk will convert. Anything longer is a +// value no field can hold and counts as a kind mismatch. +static const int32_t kTableJsonMaxNumber = 512; + +// The decimal point the C runtime is CURRENTLY using. Number conversion is +// the one locale-sensitive corner of the grammar — JSON's point is always +// '.', the runtime's is whatever the program set — so every number crosses +// this one character on the way out and on the way back in. Nothing else in +// the walk consults the locale. +inline char TableJsonDecimalPoint() +{ + const struct lconv * conv = localeconv(); + if ( conv != NULL && conv->decimal_point != NULL && conv->decimal_point[0] != 0 ) + { + return conv->decimal_point[0]; + } + return '.'; +} + +// ---- storage access: the descriptors give an offset and a width, and the +// ---- storage is the HOST's, so every load and store goes through a width +// ---- switch rather than a memcpy into the low bytes of a wider word + +// finite: not a NaN, not an infinity. Written without — the walk's +// runtime surface stays the handful of functions it already names. +// A vocabulary entry the descriptor could not spell. The generated name +// functions answer "???" for a value outside the declared set, and that is +// not a name — writing it would put a spelling in the text that the reader +// then counts as unknown, turning a refusal into a silent loss. +inline bool TableJsonNamed( const char * name ) +{ + return name != NULL && strcmp( name, "???" ) != 0; +} + +inline bool TableJsonFinite( double v ) +{ + return v == v && v <= 1.7976931348623157e308 && v >= -1.7976931348623157e308; +} + +inline uint64_t TableJsonGetRaw( const void * storage, uint32_t width ) +{ + switch ( width ) + { + case 1: { uint8_t v = 0; memcpy( &v, storage, 1 ); return v; } + case 2: { uint16_t v = 0; memcpy( &v, storage, 2 ); return v; } + case 4: { uint32_t v = 0; memcpy( &v, storage, 4 ); return v; } + case 8: { uint64_t v = 0; memcpy( &v, storage, 8 ); return v; } + } + return 0; +} + +inline void TableJsonSetRaw( void * storage, uint32_t width, uint64_t value ) +{ + switch ( width ) + { + case 1: { uint8_t v = (uint8_t) value; memcpy( storage, &v, 1 ); break; } + case 2: { uint16_t v = (uint16_t) value; memcpy( storage, &v, 2 ); break; } + case 4: { uint32_t v = (uint32_t) value; memcpy( storage, &v, 4 ); break; } + case 8: { uint64_t v = value; memcpy( storage, &v, 8 ); break; } + } +} + +inline int64_t TableJsonGetSigned( const void * storage, uint32_t width ) +{ + uint64_t raw = TableJsonGetRaw( storage, width ); + if ( width < 8 ) + { + uint64_t sign = uint64_t( 1 ) << ( width * 8 - 1 ); + if ( ( raw & sign ) != 0 ) + { + raw |= ~( ( sign << 1 ) - 1 ); + } + } + return (int64_t) raw; +} + +// ---- the WIDE kinds (docs/SPEC-TABLES.md §3, §16.2) ---- +// +// The 128-bit integers and the fixed-point family convert EXACTLY, over two +// 64-bit lanes: a 128-bit integer is a decimal integer, a fixed value a +// decimal in WHOLE UNITS (1.0, -0.25, 3.0000152587890625) and nothing +// on either path passes through a double. Nothing here needs a 128-bit type +// either, which is what keeps this walk one text for every unit. +struct TableJsonWide +{ + uint64_t lo; + uint64_t hi; +}; + +inline bool TableJsonKindWide( uint8_t kind ) { return kind >= 18 && kind <= 29; } +inline bool TableJsonKindWideSigned( uint8_t kind ) { return kind == 18 || ( kind >= 20 && kind <= 24 ); } +inline bool TableJsonKindFixed( uint8_t kind ) { return kind >= 20 && kind <= 29; } + +inline bool TableJsonWideZero( TableJsonWide v ) { return v.lo == 0 && v.hi == 0; } +inline bool TableJsonWideNegative( TableJsonWide v ) { return ( v.hi >> 63 ) != 0; } + +inline int TableJsonWideCompare( TableJsonWide a, TableJsonWide b, bool is_signed ) +{ + if ( is_signed && TableJsonWideNegative( a ) != TableJsonWideNegative( b ) ) { return TableJsonWideNegative( a ) ? -1 : 1; } + if ( a.hi != b.hi ) { return a.hi < b.hi ? -1 : 1; } + if ( a.lo != b.lo ) { return a.lo < b.lo ? -1 : 1; } + return 0; +} + +inline TableJsonWide TableJsonWideShl( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.hi = v.lo << ( n - 64 ); return r; } + r.hi = ( v.hi << n ) | ( v.lo >> ( 64 - n ) ); + r.lo = v.lo << n; + return r; +} + +inline TableJsonWide TableJsonWideShr( TableJsonWide v, int n ) +{ + TableJsonWide r = { 0, 0 }; + if ( n <= 0 ) { return v; } + if ( n >= 128 ) { return r; } + if ( n >= 64 ) { r.lo = v.hi >> ( n - 64 ); return r; } + r.lo = ( v.lo >> n ) | ( v.hi << ( 64 - n ) ); + r.hi = v.hi >> n; + return r; +} + +inline TableJsonWide TableJsonWideNeg( TableJsonWide v ) +{ + TableJsonWide r; + r.lo = ~v.lo + 1; + r.hi = ~v.hi + ( r.lo == 0 ? 1 : 0 ); + return r; +} + +// v = v * m + a; the return is the carry out of 128 bits +inline uint32_t TableJsonWideMulAdd( TableJsonWide * v, uint32_t m, uint32_t a ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t carry = a; + for ( int i = 0; i < 4; i++ ) + { + uint64_t p = limb[i] * m + carry; + limb[i] = p & 0xffffffffull; + carry = p >> 32; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) carry; +} + +// v = v / d; the return is the remainder +inline uint32_t TableJsonWideDiv( TableJsonWide * v, uint32_t d ) +{ + uint64_t limb[4] = { v->lo & 0xffffffffull, v->lo >> 32, v->hi & 0xffffffffull, v->hi >> 32 }; + uint64_t rem = 0; + for ( int i = 3; i >= 0; i-- ) + { + uint64_t cur = ( rem << 32 ) | limb[i]; + limb[i] = cur / d; + rem = cur % d; + } + v->lo = limb[0] | ( limb[1] << 32 ); + v->hi = limb[2] | ( limb[3] << 32 ); + return (uint32_t) rem; +} + +// The storage of a wide kind, as lanes. A sixteen-byte storage is serialize's +// pair — native __int128 in the host's byte order, or the emulated struct with +// its low lane first — so the lanes are read in the host's order; a narrower +// storage is one lane, sign-extended for a signed kind. +inline TableJsonWide TableJsonWideLoad( const void * storage, uint32_t width, bool is_signed ) +{ + TableJsonWide v = { 0, 0 }; + if ( width == 16 ) + { + uint64_t half[2]; + memcpy( half, storage, 16 ); + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + v.lo = little ? half[0] : half[1]; + v.hi = little ? half[1] : half[0]; + return v; + } + v.lo = is_signed ? (uint64_t) TableJsonGetSigned( storage, width ) : TableJsonGetRaw( storage, width ); + v.hi = ( is_signed && ( v.lo >> 63 ) != 0 ) ? ~uint64_t( 0 ) : 0; + return v; +} + +inline void TableJsonWideStore( void * storage, uint32_t width, TableJsonWide v ) +{ + if ( width == 16 ) + { + uint16_t probe = 1; + bool little = *(const uint8_t *) &probe == 1; + uint64_t half[2]; + half[0] = little ? v.lo : v.hi; + half[1] = little ? v.hi : v.lo; + memcpy( storage, half, 16 ); + return; + } + TableJsonSetRaw( storage, width, v.lo ); +} + +// a counted field's companion: a string's length, a bytes' length, a counted +// array's count. Bounded by the declared extent on the way out, so a storage +// invariant a caller broke cannot walk off the end of the array. +inline int32_t TableJsonCount( const void * base, const TableFieldInfo * f ) +{ + if ( !f->counted ) + { + return f->array_bound; + } + int32_t count = 0; + memcpy( &count, (const uint8_t *) base + f->count_offset, sizeof( count ) ); + if ( count < 0 ) { count = 0; } + if ( count > f->array_bound ) { count = f->array_bound; } + return count; +} + +inline void TableJsonSetCount( void * base, const TableFieldInfo * f, int32_t count ) +{ + if ( f->counted ) + { + memcpy( (uint8_t *) base + f->count_offset, &count, sizeof( count ) ); + } +} + +// ---- what a field's kind expects to see in the text ---- +// +// One classifier, consulted by both directions, so a reader and a writer can +// never disagree about a kind's JSON form. 'o' object, 'a' array, 's' +// string, 'n' number, 'b' boolean. +// +// A vocabulary field is spelled by NAME: an enum is one name, a flags mask +// is the array of the names of its set bits. The two are told apart by the +// id column — an enum variant rides under a wire id, a flags BIT never does +// (docs/SPEC-TABLES.md §4), so a name function with no id function is flags. +// +// bytes(N) is the one kind whose element kind does not decide its form: it +// shares u8 with a plain array of u8, and rides as base64. The schema type +// name settles it, and "bytes" is a keyword no declaration can claim. +inline bool TableJsonIsBytes( const TableFieldInfo * f ) +{ + return f->is_array && f->kind == 6 && strcmp( f->type_name, "bytes" ) == 0; +} + +// An ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): its JSON form is an OBJECT +// keyed by variant name, not a positional array, because that is what the +// storage is — one slot per variant, addressed by the variant. +inline bool TableJsonIsKeyed( const TableFieldInfo * f ) +{ + return f->key_name != NULL; +} + +// THE KEY A STORAGE SLOT HOLDS (§2.4, §8): the storage shifts left, so slot i +// holds the key i + 1 and nothing is stored for None. This is the ONE place +// the walker spells the shift. +inline uint64_t TableJsonKeyedSlotKey( int64_t slot ) +{ + return (uint64_t) ( slot + 1 ); +} + +// A slot whose key names a variant of the keying enum. Every slot in +// [0, array_bound) does, unless the enum carries max-headroom variants outside +// a table closure, where a reserved value names nothing and its key id is 0 — +// the reserved id no declared name can fold to (§5). +inline bool TableJsonKeyedSlotValid( const TableFieldInfo * f, int64_t slot ) +{ + return f->key_id( TableJsonKeyedSlotKey( slot ) ) != 0; +} + +inline bool TableJsonIsFlags( const TableFieldInfo * f ) +{ + return f->enum_name != NULL && f->variant_id == NULL; +} + +inline bool TableJsonIsEnum( const TableFieldInfo * f ) +{ + return f->variant_id != NULL && f->arms == NULL; +} + +inline char TableJsonShape( const TableFieldInfo * f ) +{ + if ( TableJsonIsMap( f ) ) return 'o'; // a MAP: an object keyed by the KEY (§2.8) + if ( f->kind == 12 ) return 's'; // string + if ( TableJsonIsBytes( f ) ) return 's'; // bytes: base64 + if ( TableJsonIsKeyed( f ) ) return 'o'; // an object keyed by variant NAME + if ( f->is_array ) return 'a'; + if ( f->arms != NULL ) return 'o'; // union: an object with ONE key + if ( f->kind == 13 ) return 'o'; // nested table or type + if ( f->kind == 17 ) return f->table != NULL ? 'o' : 's'; // a pointer: the pointee's object in place, or null (§16.7); a byte buffer's string (§2.5) + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// the ELEMENT shape of an array field — the same classifier one level down +inline char TableJsonElementShape( const TableFieldInfo * f ) +{ + if ( f->arms != NULL ) return 'o'; // an element of an array of unions: one key, the arm (§2.6) + if ( f->kind == 13 ) return 'o'; + if ( TableJsonIsEnum( f ) ) return 's'; + if ( TableJsonIsFlags( f ) ) return 'a'; + if ( f->kind == 1 ) return 'b'; + return 'n'; +} + +// A guarded group rides only when its guard reads true — the wire's own +// elision (§4), carried into the text so a text and a wire written from one +// instance say the same thing. The guard is spelled as its branch condition +// over bool fields of the SAME type ("at_rest", "!at_rest", +// "active && has_target"), so evaluating it is a walk of the same +// descriptor. Nothing is inferred in the other direction: reading places +// every key it can name, and the guard is a plain bool key (§16.2). +inline bool TableJsonGuardHolds( const void * base, const TableTypeInfo * info, const char * guard ) +{ + const char * p = guard; + for ( ;; ) + { + while ( *p == ' ' || *p == '&' ) { p++; } + if ( *p == 0 ) { return true; } + bool want = true; + if ( *p == '!' ) { want = false; p++; } + const char * start = p; + while ( *p != 0 && *p != ' ' && *p != '&' ) { p++; } + size_t length = (size_t) ( p - start ); + bool value = false; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( strlen( f->name ) == length && strncmp( f->name, start, length ) == 0 ) + { + value = TableJsonGetRaw( (const uint8_t *) base + f->offset, f->elem_size ) != 0; + break; + } + } + if ( value != want ) { return false; } + } +} + +// ---- writing ---- + +// The writer sink MEASURES when the buffer is NULL and WRITES when it is +// not, over one code path — so measure and write agree byte for byte, the +// wire's invariant (§9) carried across. +struct TableJsonOut +{ + char * buffer; + int64_t capacity; + int64_t offset; + bool overflow; + void * graph; // the pointered write's identity map (§16.7); NULL for a fixed table + + void raw( const char * data, int64_t count ) + { + if ( buffer != NULL ) + { + if ( offset + count > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) count ); + } + offset += count; + } + void put( char c ) { raw( &c, 1 ); } + void text( const char * s ) { raw( s, (int64_t) strlen( s ) ); } + void line( int32_t depth ) + { + put( '\n' ); + for ( int32_t i = 0; i < depth; i++ ) { raw( " ", 2 ); } + } +}; + +inline const char * TableJsonBase64Alphabet() +{ + return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; +} + +inline void TableJsonWriteBase64( TableJsonOut & out, const uint8_t * data, int32_t length ) +{ + const char * alphabet = TableJsonBase64Alphabet(); + out.put( '"' ); + int32_t i = 0; + for ( ; i + 3 <= length; i += 3 ) + { + uint32_t triple = ( uint32_t( data[i] ) << 16 ) | ( uint32_t( data[i+1] ) << 8 ) | uint32_t( data[i+2] ); + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], + alphabet[ ( triple >> 6 ) & 0x3f ], alphabet[ triple & 0x3f ] }; + out.raw( quad, 4 ); + } + if ( i < length ) + { + int32_t left = length - i; + uint32_t triple = uint32_t( data[i] ) << 16; + if ( left == 2 ) { triple |= uint32_t( data[i+1] ) << 8; } + char quad[4] = { alphabet[ ( triple >> 18 ) & 0x3f ], alphabet[ ( triple >> 12 ) & 0x3f ], '=', '=' }; + if ( left == 2 ) { quad[2] = alphabet[ ( triple >> 6 ) & 0x3f ]; } + out.raw( quad, 4 ); + } + out.put( '"' ); +} + +// One UTF-8 sequence at s, or -1 when the bytes there are not one. Rejects +// the lot: a stray continuation, an overlong form, a surrogate half, and +// anything past U+10FFFF. +inline int32_t TableJsonUtf8( const char * s, int32_t remaining, int32_t * width ) +{ + unsigned char lead = (unsigned char) s[0]; + int32_t want = 0; + int32_t code = 0; + if ( lead < 0x80 ) { *width = 1; return lead; } + else if ( lead >= 0xc2 && lead <= 0xdf ) { want = 2; code = lead & 0x1f; } + else if ( lead >= 0xe0 && lead <= 0xef ) { want = 3; code = lead & 0x0f; } + else if ( lead >= 0xf0 && lead <= 0xf4 ) { want = 4; code = lead & 0x07; } + else { return -1; } + if ( remaining < want ) { return -1; } + for ( int32_t i = 1; i < want; i++ ) + { + unsigned char next = (unsigned char) s[i]; + if ( ( next & 0xc0 ) != 0x80 ) { return -1; } + code = ( code << 6 ) | ( next & 0x3f ); + } + if ( want == 3 && code < 0x800 ) { return -1; } // overlong + if ( want == 4 && code < 0x10000 ) { return -1; } // overlong + if ( code >= 0xd800 && code <= 0xdfff ) { return -1; } // a surrogate half + if ( code > 0x10ffff ) { return -1; } + *width = want; + return code; +} + +// A JSON text MUST be valid UTF-8 (RFC 8259 §8.1). The read path is +// byte-transparent — the wire imposes no encoding (§3) and a string may hold +// anything — so the WRITER is where that obligation is met: a byte that is +// not part of a well-formed sequence is written as U+FFFD, one per bad byte, +// and never raw. A text this walk writes is therefore readable by any +// conforming parser, which a raw byte would not be. The cost is stated +// plainly: for a string holding invalid UTF-8, the round trip is NOT +// byte-identical, because the alternative is emitting a text that is not +// JSON. +inline void TableJsonWriteString( TableJsonOut & out, const char * s, int32_t length ) +{ + static const char hex[] = "0123456789abcdef"; + out.put( '"' ); + for ( int32_t i = 0; i < length; i++ ) + { + unsigned char c = (unsigned char) s[i]; + switch ( c ) + { + case '"': out.raw( "\\\"", 2 ); break; + case '\\': out.raw( "\\\\", 2 ); break; + case '\b': out.raw( "\\b", 2 ); break; + case '\f': out.raw( "\\f", 2 ); break; + case '\n': out.raw( "\\n", 2 ); break; + case '\r': out.raw( "\\r", 2 ); break; + case '\t': out.raw( "\\t", 2 ); break; + default: + if ( c < 0x20 ) + { + char escape[6] = { '\\', 'u', '0', '0', hex[ c >> 4 ], hex[ c & 0xf ] }; + out.raw( escape, 6 ); + } + else if ( c < 0x80 ) + { + out.put( (char) c ); + } + else + { + int32_t width = 0; + if ( TableJsonUtf8( s + i, length - i, &width ) < 0 ) + { + out.raw( "\xef\xbf\xbd", 3 ); // U+FFFD, one per bad byte + } + else + { + out.raw( s + i, width ); + i += width - 1; + } + } + break; + } + } + out.put( '"' ); +} + +inline void TableJsonWriteUnsigned( TableJsonOut & out, uint64_t value ) +{ + char digits[24]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) ( value % 10 ) ); + value /= 10; + } while ( value != 0 ); + char text[24]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); +} + +inline void TableJsonWriteSigned( TableJsonOut & out, int64_t value ) +{ + if ( value < 0 ) + { + out.put( '-' ); + TableJsonWriteUnsigned( out, uint64_t( 0 ) - (uint64_t) value ); + return; + } + TableJsonWriteUnsigned( out, (uint64_t) value ); +} + +// A wide kind writes its raw storage as §16.2's text: a 128-bit integer as a +// decimal integer; a fixed value in WHOLE UNITS as the shortest exact decimal +// with at least one fractional digit (1.0, -0.25), the spelling the schema text +// gives a fixed default. The fraction terminates because a dyadic fraction has +// a finite decimal expansion — at most F digits. +inline void TableJsonWriteWide( TableJsonOut & out, const void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + TableJsonWide v = TableJsonWideLoad( storage, f->elem_size, is_signed ); + if ( is_signed && TableJsonWideNegative( v ) ) + { + out.put( '-' ); + v = TableJsonWideNeg( v ); + } + int frac = f->frac_bits; + TableJsonWide whole = TableJsonWideShr( v, frac ); + char digits[40]; + int32_t n = 0; + do + { + digits[n++] = (char) ( '0' + (int) TableJsonWideDiv( &whole, 10 ) ); + } while ( !TableJsonWideZero( whole ) ); + char text[40]; + for ( int32_t i = 0; i < n; i++ ) { text[i] = digits[n - 1 - i]; } + out.raw( text, n ); + if ( !TableJsonKindFixed( f->kind ) ) { return; } + out.put( '.' ); + // the fraction bits alone: v with everything at and above bit F cleared + TableJsonWide fraction = v; + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + if ( frac == 0 ) { fraction.lo = 0; } + if ( TableJsonWideZero( fraction ) ) + { + out.put( '0' ); + return; + } + while ( !TableJsonWideZero( fraction ) ) + { + // ×10: the digit is what lands at and above bit F, including the + // carry out of 128 bits when F leaves no room for it below + uint32_t carry = TableJsonWideMulAdd( &fraction, 10, 0 ); + uint64_t digit = TableJsonWideShr( fraction, frac ).lo; + if ( frac > 64 ) { digit |= uint64_t( carry ) << ( 128 - frac ); } + out.put( (char) ( '0' + (int) digit ) ); + if ( frac < 64 ) { fraction.hi = 0; fraction.lo &= ( uint64_t( 1 ) << frac ) - 1; } + else { fraction.hi &= ( uint64_t( 1 ) << ( frac - 64 ) ) - 1; } + } +} + +// A float writes at the SHORTEST precision that reads back as the same value +// at the field's own width, so a round trip is exact and a text stays +// readable. Non-finite values have no JSON spelling at all, and the writer +// REFUSES rather than losing one silently — the same rule measure and save +// already apply to an enum value no variant names (§5). +inline bool TableJsonWriteFloat( TableJsonOut & out, double value, bool single ) +{ + if ( !TableJsonFinite( value ) ) { return false; } + char text[64]; + int low = single ? 6 : 15; + int high = single ? 9 : 17; + int length = 0; + for ( int digits = low; ; digits++ ) + { + length = snprintf( text, sizeof( text ), "%.*g", digits, value ); + if ( length <= 0 || length >= (int) sizeof( text ) ) { return false; } + if ( digits >= high ) { break; } + // the round-trip check runs BEFORE the decimal point is normalised: + // the token still carries whatever point snprintf just produced + if ( single ) + { + if ( (double) strtof( text, NULL ) == value ) { break; } + } + else + { + if ( strtod( text, NULL ) == value ) { break; } + } + } + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int i = 0; i < length; i++ ) + { + if ( text[i] == point ) { text[i] = '.'; } + } + } + out.raw( text, length ); + return true; +} + +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration writes through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ); + +// one scalar, at one storage address: a nested object, a union, a +// vocabulary, or a number +inline bool TableJsonWriteScalar( TableJsonOut & out, const void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; None is {} + const TableUnionInfo * arms = f->arms(); + uint64_t tag = TableJsonGetRaw( (const uint8_t *) storage + arms->tag_offset, arms->tag_size ); + if ( tag == 0 ) + { + out.raw( "{}", 2 ); + return true; + } + if ( (int64_t) tag > f->enum_max ) + { + return false; // a tag no arm names, exactly as measure refuses it + } + const char * arm = f->enum_name( tag ); + // and refuse on the NAME, not merely on the bound: §16.2 says a value + // no variant NAMES is refused, so the check is the name. Writing + // whatever came back would emit "???", a spelling the reader counts + // as unknown — a silent round-trip loss in place of a refusal. + if ( !TableJsonNamed( arm ) ) { return false; } + out.put( '{' ); + out.line( depth + 1 ); + TableJsonWriteString( out, arm, (int32_t) strlen( arm ) ); + out.raw( ": ", 2 ); + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2): an arm that names + // no declaration carries the FIELD descriptor a field of its type + // would carry, offsets taken inside the union storage (§2.6), so the + // value walks through the field writer one key down. + if ( arms->arms[tag].field != NULL ) + { + if ( !TableJsonWriteField( out, storage, arms->arms[tag].field, depth + 1 ) ) + { + return false; + } + } + else if ( arms->arms[tag].table == NULL ) + { + out.raw( "null", 4 ); // a payload-free arm: the name selects it (§2.6) + } + else if ( !TableJsonWriteValue( out, (const uint8_t *) storage + arms->arms[tag].offset, arms->arms[tag].table, depth + 1 ) ) + { + return false; + } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->kind == 13 ) + { + return TableJsonWriteValue( out, storage, f->table, depth ); + } + if ( TableJsonIsEnum( f ) ) + { + uint64_t value = TableJsonGetRaw( storage, f->elem_size ); + // a value no variant names has no text spelling, exactly as it has no + // wire identity: the writer REFUSES rather than writing None over it, + // the rule measure and save already apply (docs/SPEC-TABLES.md §5) + if ( (int64_t) value > f->enum_max ) { return false; } + if ( value != 0 && f->variant_id( value ) == 0 ) { return false; } + const char * name = f->enum_name( value ); + if ( !TableJsonNamed( name ) ) { return false; } + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + return true; + } + if ( TableJsonIsFlags( f ) ) + { + uint64_t bits = TableJsonGetRaw( storage, f->elem_size ); + if ( bits == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + bool first = true; + for ( int64_t bit = 0; bit < 64; bit++ ) + { + if ( ( bits & ( uint64_t( 1 ) << bit ) ) == 0 ) { continue; } + if ( bit > f->enum_max ) + { + return false; // a bit no variant names has no text spelling + } + const char * name = f->enum_name( (uint64_t) bit ); + if ( !TableJsonNamed( name ) ) { return false; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + TableJsonWriteString( out, name, (int32_t) strlen( name ) ); + } + out.line( depth ); + out.put( ']' ); + return true; + } + switch ( f->kind ) + { + case 1: + out.text( TableJsonGetRaw( storage, f->elem_size ) != 0 ? "true" : "false" ); + return true; + case 10: + { + float v = 0.0f; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, (double) v, true ); + } + case 11: + { + double v = 0.0; + memcpy( &v, storage, sizeof( v ) ); + return TableJsonWriteFloat( out, v, false ); + } + case 2: case 3: case 4: case 5: + TableJsonWriteSigned( out, TableJsonGetSigned( storage, f->elem_size ) ); + return true; + default: + if ( TableJsonKindWide( f->kind ) ) + { + TableJsonWriteWide( out, storage, f ); + return true; + } + TableJsonWriteUnsigned( out, TableJsonGetRaw( storage, f->elem_size ) ); + return true; + } +} + +inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const TableFieldInfo * f, int32_t depth ) +{ + const uint8_t * storage = (const uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonWriteMap( out, (const void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } + if ( f->kind == 17 && !f->is_array ) + { + return TableJsonWritePointer( out, storage, f, depth ); + } + if ( f->kind == 17 ) + { + // an ARRAY OF POINTERS (§2.1): the pointer row per element — the + // pointee's object in place, null, or `&node` for a shared one (§16.7) + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWritePointer( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; + } + if ( f->kind == 12 ) + { + TableJsonWriteString( out, (const char *) storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + TableJsonWriteBase64( out, storage, TableJsonCount( base, f ) ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + // one entry per SLOT, keyed by the variant that owns it, so inserting + // a variant next season moves nothing in the text either. Slot i holds + // the key i + 1: nothing is stored for None, so nothing is written for it. + out.put( '{' ); + bool first = true; + for ( int64_t slot = 0; slot < f->array_bound; slot++ ) + { + if ( !TableJsonKeyedSlotValid( f, slot ) ) { continue; } + if ( !first ) { out.put( ',' ); } + first = false; + out.line( depth + 1 ); + const char * key = f->key_name( TableJsonKeyedSlotKey( slot ) ); + TableJsonWriteString( out, key, (int32_t) strlen( key ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteScalar( out, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + if ( first ) { out.raw( "}", 1 ); return true; } + out.line( depth ); + out.put( '}' ); + return true; + } + if ( f->is_array ) + { + int32_t count = TableJsonCount( base, f ); + if ( count == 0 ) + { + out.raw( "[]", 2 ); + return true; + } + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + if ( !TableJsonWriteScalar( out, storage + (int64_t) i * f->elem_size, f, depth + 1 ) ) + { + return false; + } + } + out.line( depth ); + out.put( ']' ); + return true; + } + return TableJsonWriteScalar( out, storage, f, depth ); +} + +// One instance's fields, in DECLARATION ORDER, defaults included — a text is +// for people and tools, and a text that elides is a text a reader has to know +// the schema to complete. `any` says whether the object is already open on +// entry — a shared node's `&node` opens it before the fields (§16.7) — and +// whether it is open on return. +inline bool TableJsonWriteFields( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth, bool & any ) +{ + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + const TableFieldInfo * f = &info->fields[i]; + if ( f->guard[0] != 0 && !TableJsonGuardHolds( base, info, f->guard ) ) { continue; } + // an ABSENT optional writes no key: presence of the key IS the + // presence (§16.2), so an absent field is an absent key and nothing + // else would read back as absent + if ( f->optional && + TableJsonGetRaw( (const uint8_t *) base + f->present_offset, 1 ) == 0 ) + { + continue; + } + if ( !any ) { out.put( '{' ); } + else { out.put( ',' ); } + any = true; + out.line( depth + 1 ); + TableJsonWriteString( out, f->json, (int32_t) strlen( f->json ) ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, base, f, depth + 1 ) ) { return false; } + } + return true; +} + +// One instance as one object. The writer carries the reader's depth cap +// (§16.2): a pointer chain nests as deep as it is long (§16.7), and a text the +// writer produced past the cap would be a text the reader refuses. +inline bool TableJsonWriteValue( TableJsonOut & out, const void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { return false; } + bool any = false; + if ( !TableJsonWriteFields( out, base, info, depth, any ) ) { return false; } + if ( !any ) + { + out.raw( "{}", 2 ); + return true; + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- reading ---- + +struct TableJsonIn +{ + const char * text; + int64_t size; + int64_t pos; + TableReport * report; + bool bad; // the text is not JSON: the walk stops and keeps what it placed + void * graph; // the pointered read's builder and label map (§16.7); NULL for a fixed table +}; + +inline void TableJsonSpace( TableJsonIn & in ) +{ + while ( in.pos < in.size ) + { + char c = in.text[in.pos]; + if ( c == ' ' || c == '\t' || c == '\n' || c == '\r' ) { in.pos++; continue; } + // comments are not JSON, and a walk that guessed at one would be + // reading a dialect nobody wrote down + if ( c == '/' ) { in.bad = true; } + return; + } +} + +inline char TableJsonPeek( TableJsonIn & in ) +{ + TableJsonSpace( in ); + return in.pos < in.size ? in.text[in.pos] : 0; +} + +// the shape of the value sitting at the cursor, without consuming it +inline char TableJsonValueShape( TableJsonIn & in ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return 'o'; + case '[': return 'a'; + case '"': return 's'; + case 't': case 'f': return 'b'; + case 'n': return 'z'; + case 0: return 0; + default: return 'n'; + } +} + +inline bool TableJsonLiteral( TableJsonIn & in, const char * word ) +{ + int64_t length = (int64_t) strlen( word ); + if ( in.pos + length > in.size || memcmp( in.text + in.pos, word, (size_t) length ) != 0 ) + { + in.bad = true; + return false; + } + in.pos += length; + return true; +} + +// one \uXXXX escape body; -1 when the four hex digits are not there +inline int TableJsonHex4( TableJsonIn & in ) +{ + if ( in.pos + 4 > in.size ) { return -1; } + int value = 0; + for ( int i = 0; i < 4; i++ ) + { + char c = in.text[in.pos + i]; + int digit; + if ( c >= '0' && c <= '9' ) { digit = c - '0'; } + else if ( c >= 'a' && c <= 'f' ) { digit = c - 'a' + 10; } + else if ( c >= 'A' && c <= 'F' ) { digit = c - 'A' + 10; } + else { return -1; } + value = ( value << 4 ) | digit; + } + in.pos += 4; + return value; +} + +inline int32_t TableJsonEncodeUtf8( uint32_t code, char * unit ) +{ + if ( code < 0x80 ) { unit[0] = (char) code; return 1; } + if ( code < 0x800 ) + { + unit[0] = (char) ( 0xc0 | ( code >> 6 ) ); + unit[1] = (char) ( 0x80 | ( code & 0x3f ) ); + return 2; + } + if ( code < 0x10000 ) + { + unit[0] = (char) ( 0xe0 | ( code >> 12 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( code & 0x3f ) ); + return 3; + } + unit[0] = (char) ( 0xf0 | ( code >> 18 ) ); + unit[1] = (char) ( 0x80 | ( ( code >> 12 ) & 0x3f ) ); + unit[2] = (char) ( 0x80 | ( ( code >> 6 ) & 0x3f ) ); + unit[3] = (char) ( 0x80 | ( code & 0x3f ) ); + return 4; +} + +// Scan one JSON string into a caller buffer. Bytes are appended ONE CODE +// POINT AT A TIME — an escape's encoding, or a UTF-8 sequence read whole — +// so a string longer than the field is clamped AT A CODE POINT BOUNDARY and +// never cut through a multi-byte character. Clamping is counted, never +// fatal, exactly as it is on the wire (§4). A NULL destination scans past a +// string without keeping it. +inline bool TableJsonScanString( TableJsonIn & in, char * out, int32_t capacity, int32_t * length ) +{ + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + int32_t placed = 0; + bool clamped = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos]; + if ( c == '"' ) { in.pos++; break; } + char unit[4]; + int32_t unit_length = 0; + if ( c == '\\' ) + { + in.pos++; + if ( in.pos >= in.size ) { in.bad = true; return false; } + char escape = in.text[in.pos++]; + switch ( escape ) + { + case '"': unit[0] = '"'; unit_length = 1; break; + case '\\': unit[0] = '\\'; unit_length = 1; break; + case '/': unit[0] = '/'; unit_length = 1; break; + case 'b': unit[0] = '\b'; unit_length = 1; break; + case 'f': unit[0] = '\f'; unit_length = 1; break; + case 'n': unit[0] = '\n'; unit_length = 1; break; + case 'r': unit[0] = '\r'; unit_length = 1; break; + case 't': unit[0] = '\t'; unit_length = 1; break; + case 'u': + { + int high = TableJsonHex4( in ); + if ( high < 0 ) { in.bad = true; return false; } + uint32_t code = (uint32_t) high; + if ( high >= 0xd800 && high <= 0xdbff && in.pos + 2 <= in.size && + in.text[in.pos] == '\\' && in.text[in.pos + 1] == 'u' ) + { + int64_t mark = in.pos; + in.pos += 2; + int low = TableJsonHex4( in ); + if ( low >= 0xdc00 && low <= 0xdfff ) + { + code = 0x10000 + ( ( (uint32_t) high - 0xd800 ) << 10 ) + ( (uint32_t) low - 0xdc00 ); + } + else + { + in.pos = mark; // a lone lead surrogate rides as itself + } + } + // a surrogate half that never found its partner has no + // UTF-8 encoding: encoding it anyway would manufacture + // CESU-8 — invalid UTF-8 — out of input that was valid + // JSON, so it reads as the replacement character + if ( code >= 0xd800 && code <= 0xdfff ) { code = 0xfffd; } + unit_length = TableJsonEncodeUtf8( code, unit ); + break; + } + default: in.bad = true; return false; + } + } + else if ( (unsigned char) c < 0x20 ) + { + in.bad = true; // a raw control character is not a JSON string body + return false; + } + else + { + // a UTF-8 sequence read WHOLE, so the clamp below can only land + // between code points. Only bytes that ACTUALLY look like + // continuations are taken: the wire imposes no encoding (§3), so + // a string may legitimately hold a stray lead byte, and one at + // the end of a text must not swallow the closing quote. + unsigned char lead = (unsigned char) c; + int32_t want = 1; + if ( ( lead & 0xe0 ) == 0xc0 ) { want = 2; } + else if ( ( lead & 0xf0 ) == 0xe0 ) { want = 3; } + else if ( ( lead & 0xf8 ) == 0xf0 ) { want = 4; } + unit[0] = c; + in.pos++; + unit_length = 1; + while ( unit_length < want && in.pos < in.size && + ( (unsigned char) in.text[in.pos] & 0xc0 ) == 0x80 ) + { + unit[unit_length++] = in.text[in.pos++]; + } + } + if ( out == NULL ) + { + placed += unit_length; // measured and not kept: a byte buffer's read sizes its node this way (§2.5) + } + else if ( placed + unit_length <= capacity ) + { + memcpy( out + placed, unit, (size_t) unit_length ); + placed += unit_length; + } + else + { + clamped = true; + } + } + if ( clamped ) { in.report->clamped++; } + if ( length != NULL ) { *length = placed; } + return true; +} + +// the numeric token at the cursor, copied out whole; false = not a number +// Scan one number, to JSON's OWN grammar (RFC 8259 §6) and not to a run of +// number-ish characters: +// +// number = [ "-" ] int [ frac ] [ exp ] +// int = "0" / ( digit1-9 *digit ) +// frac = "." 1*digit +// exp = ( "e" / "E" ) [ "-" / "+" ] 1*digit +// +// Scanning the production is what makes a typo in an authoring file a +// DIAGNOSTIC rather than a value: "1-2" scans as 1 and leaves "-2" where the +// object expects a comma, so the text is malformed — which is what §16.2 +// already promises. A permissive scan would hand "1-2" to a digit loop and +// report a clamp, and a config pipeline would never hear about it. Leading +// "+", leading zeros, ".5" and "3." are not JSON either. +inline bool TableJsonWalkNumber( TableJsonIn & in, bool * integral ) +{ + TableJsonSpace( in ); + bool whole = true; + if ( in.pos < in.size && in.text[in.pos] == '-' ) { in.pos++; } + // int: a lone zero, or a non-zero digit and any digits after it + if ( in.pos >= in.size ) { return false; } + if ( in.text[in.pos] == '0' ) + { + in.pos++; + } + else if ( in.text[in.pos] >= '1' && in.text[in.pos] <= '9' ) + { + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + } + else + { + return false; + } + // frac + if ( in.pos < in.size && in.text[in.pos] == '.' ) + { + in.pos++; + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + // exp + if ( in.pos < in.size && ( in.text[in.pos] == 'e' || in.text[in.pos] == 'E' ) ) + { + in.pos++; + if ( in.pos < in.size && ( in.text[in.pos] == '-' || in.text[in.pos] == '+' ) ) { in.pos++; } + if ( in.pos >= in.size || in.text[in.pos] < '0' || in.text[in.pos] > '9' ) { return false; } + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) { in.pos++; } + whole = false; + } + *integral = whole; + return true; +} + +// the same production, with the token kept for conversion +inline bool TableJsonScanNumber( TableJsonIn & in, char * token, int32_t capacity, int32_t * length, bool * integral ) +{ + TableJsonSpace( in ); + int64_t start = in.pos; + if ( !TableJsonWalkNumber( in, integral ) ) { return false; } + int64_t count = in.pos - start; + if ( count <= 0 || count >= capacity ) { return false; } + memcpy( token, in.text + start, (size_t) count ); + token[count] = 0; + *length = (int32_t) count; + return true; +} + +// the token's exact double, through the runtime's own converter — which +// speaks the LOCALE's decimal point, so the token crosses back over that +// character on its way in +inline double TableJsonTokenDouble( const char * token, int32_t length, bool single ) +{ + char work[kTableJsonMaxNumber]; + memcpy( work, token, (size_t) length ); + work[length] = 0; + char point = TableJsonDecimalPoint(); + if ( point != '.' ) + { + for ( int32_t i = 0; i < length; i++ ) + { + if ( work[i] == '.' ) { work[i] = point; } + } + } + if ( single ) { return (double) strtof( work, NULL ); } + return strtod( work, NULL ); +} + +// the token's exact integer, parsed digit by digit so no width and no +// locale can move it. Saturation is reported as a clamp, the wire's rule for +// a value outside what the reader can hold (§4). +inline int64_t TableJsonTokenInteger( const char * token, int32_t length, bool is_signed, bool * saturated ) +{ + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) + { + negative = token[i] == '-'; + i++; + } + uint64_t magnitude = 0; + bool over = false; + for ( ; i < length; i++ ) + { + uint64_t digit = (uint64_t) ( token[i] - '0' ); + if ( magnitude > ( UINT64_MAX - digit ) / 10 ) { over = true; break; } + magnitude = magnitude * 10 + digit; + } + if ( !is_signed ) + { + // -0 IS zero, and clamping it would report an event that did not + // happen; only a real negative magnitude is out of range here + if ( negative ) { *saturated = magnitude != 0; return 0; } + if ( over ) { *saturated = true; return (int64_t) UINT64_MAX; } + *saturated = false; + return (int64_t) magnitude; + } + if ( negative ) + { + if ( over || magnitude > ( uint64_t( 1 ) << 63 ) ) { *saturated = true; return INT64_MIN; } + *saturated = false; + if ( magnitude == ( uint64_t( 1 ) << 63 ) ) { return INT64_MIN; } + return -(int64_t) magnitude; + } + if ( over || magnitude > (uint64_t) INT64_MAX ) { *saturated = true; return INT64_MAX; } + *saturated = false; + return (int64_t) magnitude; +} + +// A number token into a wide kind's raw storage (docs/SPEC-TABLES.md §16.2). A +// 128-bit integer takes any token whose VALUE is integral; a fixed field any +// token whose value is EXACTLY representable in its Q I.F — a finer fraction +// is the wrong shape for the field, counted as a kind mismatch and never +// rounded, the rule SPEC.md §4.6 gives a fixed default. A magnitude past 128 +// bits saturates and counts as a clamp, as an int64 field saturates at +// INT64_MAX; the declared range clamps after it, on the RAW scale, as it does +// for every bounded scalar. +// +// The token is normalized to its digits with the decimal point after "point" +// of them. An integer part past 40 digits is above 2^128 whatever the digits +// are, and a value below 10^-40 is finer than 2^-127, the finest fraction any +// F can spell — so outside that band the answer is known without the +// arithmetic, and a token spelling 1e999999999 costs nothing to refuse. +inline bool TableJsonReadWide( TableJsonIn & in, const char * token, int32_t length, void * storage, const TableFieldInfo * f ) +{ + bool is_signed = TableJsonKindWideSigned( f->kind ); + int frac = f->frac_bits; + int32_t i = 0; + bool negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { negative = token[i] == '-'; i++; } + const char * int_digits = token + i; + int32_t int_len = 0; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { int_len++; i++; } + const char * frac_digits = token + i; + int32_t frac_len = 0; + if ( i < length && token[i] == '.' ) + { + i++; + frac_digits = token + i; + while ( i < length && token[i] >= '0' && token[i] <= '9' ) { frac_len++; i++; } + } + int64_t exp = 0; + if ( i < length && ( token[i] == 'e' || token[i] == 'E' ) ) + { + i++; + bool exp_negative = false; + if ( i < length && ( token[i] == '-' || token[i] == '+' ) ) { exp_negative = token[i] == '-'; i++; } + while ( i < length && token[i] >= '0' && token[i] <= '9' ) + { + if ( exp < 100000 ) { exp = exp * 10 + ( token[i] - '0' ); } + i++; + } + if ( exp_negative ) { exp = -exp; } + } + // the digits, with the point after "point" of them; leading and trailing + // zeros stripped. digit( k ) reads the k-th of the int and frac runs. + int32_t start = 0, end = int_len + frac_len; + int64_t point = int_len + exp; + while ( start < end && ( start < int_len ? int_digits[start] : frac_digits[start - int_len] ) == '0' ) { start++; point--; } + while ( end > start && ( end - 1 < int_len ? int_digits[end - 1] : frac_digits[end - 1 - int_len] ) == '0' ) { end--; } + + TableJsonWide raw = { 0, 0 }; + bool saturated = false; + TableJsonWide signed_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) >> 1 }; + TableJsonWide signed_min = { 0, uint64_t( 1 ) << 63 }; + TableJsonWide unsigned_max = { ~uint64_t( 0 ), ~uint64_t( 0 ) }; + if ( start == end ) + { + // zero, and -0 IS zero + } + else if ( point > 40 ) + { + saturated = true; + if ( !negative ) { raw = is_signed ? signed_max : unsigned_max; } + else if ( is_signed ) { raw = signed_min; } + } + else if ( point < -40 ) + { + in.report->kind_mismatch++; // finer than any F can spell + return true; + } + else + { + // the fraction FIRST, so an inexact value is the wrong shape whatever + // its magnitude: its digits, with the zeros a negative point puts in + // front, doubled F times; each doubling's carry is the next bit, and + // the value is exact iff nothing is left after the last one + char fd[kTableJsonMaxNumber + 48]; + int32_t fn = 0; + for ( int64_t z = point; z < 0; z++ ) { fd[fn++] = 0; } + for ( int32_t k = (int32_t) ( point > 0 ? point : 0 ) + start; k < end; k++ ) + { + fd[fn++] = (char) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ); + } + TableJsonWide fraction = { 0, 0 }; + for ( int b = 0; b < frac; b++ ) + { + int carry = 0; + for ( int32_t k = fn - 1; k >= 0; k-- ) + { + int d = fd[k] * 2 + carry; + fd[k] = (char) ( d % 10 ); + carry = d / 10; + } + fraction = TableJsonWideShl( fraction, 1 ); + fraction.lo |= (uint64_t) carry; + } + for ( int32_t k = 0; k < fn; k++ ) + { + if ( fd[k] != 0 ) + { + in.report->kind_mismatch++; + return true; + } + } + // then the whole part, saturating past 128 bits + TableJsonWide whole = { 0, 0 }; + for ( int64_t k = start; k < start + point && !saturated; k++ ) + { + uint32_t digit = k < end ? (uint32_t) ( ( k < int_len ? int_digits[k] : frac_digits[k - int_len] ) - '0' ) : 0; + if ( TableJsonWideMulAdd( &whole, 10, digit ) != 0 ) { saturated = true; } + } + if ( !saturated && frac > 0 && !TableJsonWideZero( TableJsonWideShr( whole, 128 - frac ) ) ) { saturated = true; } + if ( !saturated ) + { + raw = TableJsonWideShl( whole, frac ); + raw.lo |= fraction.lo; + raw.hi |= fraction.hi; + } + if ( is_signed ) + { + if ( !saturated && !negative && TableJsonWideNegative( raw ) ) { saturated = true; } + if ( !saturated && negative && TableJsonWideCompare( raw, signed_min, false ) > 0 ) { saturated = true; } + if ( saturated ) { raw = negative ? signed_min : signed_max; } + else if ( negative ) { raw = TableJsonWideNeg( raw ); } + } + else + { + if ( saturated ) { raw = unsigned_max; } + if ( negative && !TableJsonWideZero( raw ) ) { raw.lo = 0; raw.hi = 0; saturated = true; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->wide != NULL ) + { + TableJsonWide lo = { f->wide->lo[0], f->wide->lo[1] }; + TableJsonWide hi = { f->wide->hi[0], f->wide->hi[1] }; + if ( TableJsonWideCompare( raw, lo, is_signed ) < 0 ) { raw = lo; in.report->clamped++; } + else if ( TableJsonWideCompare( raw, hi, is_signed ) > 0 ) { raw = hi; in.report->clamped++; } + } + TableJsonWideStore( storage, f->elem_size, raw ); + return true; +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ); + +inline bool TableJsonSkipContainer( TableJsonIn & in, char close, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; // the opening bracket + bool first = true; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == close ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + if ( close == '}' ) + { + // the key is kept, because a skipped OBJECT may still be a + // pointer's: an `&node` opening it names a node the storage could + // not hold, and the numbering has to survive the drop (§16.7). + // Anywhere but first, the prefix is the reserved key out of place + // — in a pointered unit; a fixed unit skips the value whole. + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( key[0] == '&' && in.graph != NULL ) + { + if ( !first ) { in.report->malformed = true; in.bad = true; return false; } + if ( !TableJsonSkippedAmpersand( in, key, depth ) ) { return false; } + first = false; + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } + } + first = false; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == close ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +inline bool TableJsonSkipValue( TableJsonIn & in, int32_t depth ) +{ + char c = TableJsonPeek( in ); + switch ( c ) + { + case '{': return TableJsonSkipContainer( in, '}', depth ); + case '[': return TableJsonSkipContainer( in, ']', depth ); + case '"': return TableJsonScanString( in, NULL, 0, NULL ); + case 't': return TableJsonLiteral( in, "true" ); + case 'f': return TableJsonLiteral( in, "false" ); + case 'n': return TableJsonLiteral( in, "null" ); + case 0: in.bad = true; return false; + default: + { + // consumed, never converted: skipping needs no buffer, and this + // is the one walk a hostile text drives to the depth cap. It is + // the SAME production the value path scans, so an unknown key + // cannot smuggle past a number a named key would refuse. + bool integral = false; + if ( !TableJsonWalkNumber( in, &integral ) ) { in.bad = true; return false; } + return true; + } + } +} + +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ); +// a UNION ARM that names no declaration reads through the field walk one key +// down (docs/SPEC-TABLES.md §2.6, §16.2), which is defined below +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ); + +// place one scalar at one storage address +inline bool TableJsonReadScalar( TableJsonIn & in, void * storage, const TableFieldInfo * f, int32_t depth ) +{ + if ( f->arms != NULL ) + { + // a union is an object with ONE key, the arm's name; {} is None, and + // two keys is a text this walk will not guess at + const TableUnionInfo * arms = f->arms(); + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, 0 ); + if ( TableJsonPeek( in ) == '}' ) { in.pos++; return true; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t tag = 0; + for ( int64_t t = 1; t <= f->enum_max; t++ ) + { + if ( strcmp( f->enum_name( (uint64_t) t ), key ) == 0 ) { tag = t; break; } + } + if ( tag == 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + void * payload = (uint8_t *) storage + arms->arms[tag].offset; + const TableFieldInfo * arm = arms->arms[tag].field; + bool placed = true; + if ( arm != NULL ) + { + // THE ARM'S VALUE TAKES THE ARM'S OWN ROW (§16.2). A value of + // the wrong shape for that row is a KIND MISMATCH: the union + // reads None, the event is counted, and the enclosing object + // continues — the rule a FIELD's value lives under, one key + // down. A pointer arm's null is a null pointer, not a shape + // error, exactly as a pointer field's is (§16.7). + char got = TableJsonValueShape( in ); + if ( arm->kind == 17 && !arm->is_array && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + memset( payload, 0, (size_t) arms->arms[tag].size ); + } + else if ( got != TableJsonShape( arm ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( arm->kind == 17 && !arm->is_array ) + { + // A POINTER ARM'S VALUE IS THE POINTEE IN PLACE, or a + // node reference to one (§16.7) — the read a pointer + // FIELD takes, which is not the scalar walk + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadPointer( in, payload, arm, depth + 1 ) ) { return false; } + } + else + { + // SELECTION ZERO-ESTABLISHES THE ARM (SPEC §5): an arm + // takes no specified default, so zero is the establish + memset( payload, 0, (size_t) arms->arms[tag].size ); + if ( !TableJsonReadField( in, storage, arm, depth + 1 ) ) { return false; } + } + } + else if ( arms->arms[tag].table != NULL ) + { + if ( TableJsonValueShape( in ) != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else + { + arms->arms[tag].table->reset( payload ); + if ( !TableJsonReadTable( in, payload, arms->arms[tag].table, depth + 1 ) ) { return false; } + } + } + else + { + // A PAYLOAD-FREE ARM'S VALUE IS null (§2.6): the arm name + // selects it and there is nothing to place + if ( TableJsonValueShape( in ) != 'z' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed = false; + } + else if ( !TableJsonLiteral( in, "null" ) ) + { + return false; + } + } + if ( placed ) + { + TableJsonSetRaw( (uint8_t *) storage + arms->tag_offset, arms->tag_size, (uint64_t) tag ); + } + } + char c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; // a second key: a one-of with two arms is not a value + return false; + } + if ( f->kind == 13 ) + { + f->table->reset( storage ); + return TableJsonReadTable( in, storage, f->table, depth + 1 ); + } + if ( TableJsonIsEnum( f ) ) + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + for ( int64_t v = 0; v <= f->enum_max; v++ ) + { + if ( strcmp( f->enum_name( (uint64_t) v ), name ) == 0 ) + { + TableJsonSetRaw( storage, f->elem_size, (uint64_t) v ); + return true; + } + } + // a name this build cannot name reads as None and counts as unknown, + // exactly as an unknown variant id does on the wire (§4) + TableJsonSetRaw( storage, f->elem_size, 0 ); + in.report->unknown++; + return true; + } + if ( TableJsonIsFlags( f ) ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + uint64_t bits = 0; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( c != '"' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + char name[kTableJsonMaxKey]; + int32_t name_length = 0; + if ( !TableJsonScanString( in, name, kTableJsonMaxKey - 1, &name_length ) ) { return false; } + name[name_length] = 0; + bool found = false; + for ( int64_t bit = 0; bit <= f->enum_max; bit++ ) + { + if ( strcmp( f->enum_name( (uint64_t) bit ), name ) == 0 ) + { + bits |= uint64_t( 1 ) << bit; + found = true; + break; + } + } + if ( !found ) { in.report->unknown++; } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + TableJsonSetRaw( storage, f->elem_size, bits ); + return true; + } + if ( f->kind == 1 ) + { + char c = TableJsonPeek( in ); + if ( c == 't' ) { if ( !TableJsonLiteral( in, "true" ) ) { return false; } TableJsonSetRaw( storage, f->elem_size, 1 ); return true; } + if ( !TableJsonLiteral( in, "false" ) ) { return false; } + TableJsonSetRaw( storage, f->elem_size, 0 ); + return true; + } + char token[kTableJsonMaxNumber]; + int32_t length = 0; + bool integral = false; + if ( !TableJsonScanNumber( in, token, kTableJsonMaxNumber, &length, &integral ) ) + { + in.bad = true; + return false; + } + if ( TableJsonKindWide( f->kind ) ) + { + return TableJsonReadWide( in, token, length, storage, f ); + } + if ( f->kind == 10 || f->kind == 11 ) + { + bool single = f->kind == 10; + double value = TableJsonTokenDouble( token, length, single ); + // A magnitude the field's format cannot hold is the WRONG SHAPE for + // the kind, and it never reaches storage: 1e400 is not a float64 and + // 1e300 is not a float32. Storing the infinity the conversion + // produced would leave an instance this walk called CLEAN that + // ToJsonMeasure then refuses forever (a non-finite float has no JSON + // spelling), and §16.1's one invariant is that a text which reads + // clean writes back. + if ( !TableJsonFinite( value ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( f->has_range ) + { + if ( value < f->range_min ) { value = f->range_min; in.report->clamped++; } + else if ( value > f->range_max ) { value = f->range_max; in.report->clamped++; } + } + if ( single ) + { + float narrow = (float) value; + if ( !TableJsonFinite( (double) narrow ) ) + { + in.report->kind_mismatch++; + return true; + } + memcpy( storage, &narrow, sizeof( narrow ) ); + } + else + { + memcpy( storage, &value, sizeof( value ) ); + } + return true; + } + // JSON HAS ONE NUMBER TYPE. 2.0 IS the integer 2 and 1e3 IS 1000, and a + // library that round-trips numbers through a double emits them that way — + // this walker's own float writer emits 1e+21. So an integer field takes + // any number whose VALUE is integral, however it was spelled; only a + // genuinely fractional value is the wrong shape for it. + bool is_signed = f->kind >= 2 && f->kind <= 5; + bool saturated = false; + int64_t value = 0; + if ( integral ) + { + value = TableJsonTokenInteger( token, length, is_signed, &saturated ); + } + else + { + double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) + { + in.report->kind_mismatch++; + return true; + } + if ( is_signed ) + { + if ( d >= 9223372036854775808.0 ) { value = INT64_MAX; saturated = true; } + else if ( d < -9223372036854775808.0 ) { value = INT64_MIN; saturated = true; } + else if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) d; } + } + else + { + if ( d < 0.0 ) + { + // a negative for an unsigned field clamps to zero, as the + // exact digit path already does + if ( d != (double) (int64_t) d ) { in.report->kind_mismatch++; return true; } + value = 0; + saturated = true; + } + else if ( d >= 18446744073709551616.0 ) { value = (int64_t) UINT64_MAX; saturated = true; } + else if ( d != (double) (uint64_t) d ) { in.report->kind_mismatch++; return true; } + else { value = (int64_t) (uint64_t) d; } + } + } + if ( saturated ) { in.report->clamped++; } + if ( f->has_range ) + { + if ( (double) value < f->range_min ) { value = (int64_t) f->range_min; in.report->clamped++; } + else if ( (double) value > f->range_max ) { value = (int64_t) f->range_max; in.report->clamped++; } + } + // the field's own storage width is the last bound: a value past it + // clamps rather than wrapping, which is what the wire does too + if ( f->elem_size < 8 ) + { + if ( is_signed ) + { + int64_t high = ( int64_t( 1 ) << ( f->elem_size * 8 - 1 ) ) - 1; + int64_t low = -high - 1; + if ( value > high ) { value = high; in.report->clamped++; } + else if ( value < low ) { value = low; in.report->clamped++; } + } + else + { + uint64_t high = ( uint64_t( 1 ) << ( f->elem_size * 8 ) ) - 1; + if ( value < 0 ) { value = 0; in.report->clamped++; } + else if ( (uint64_t) value > high ) { value = (int64_t) high; in.report->clamped++; } + } + } + // at eight bytes the storage IS the parser's width, and an unsigned value + // past INT64_MAX rides here as a negative int64 by design — the token + // parser already turned a NEGATIVE token for an unsigned field into a + // clamped zero, so there is nothing left to bound. + TableJsonSetRaw( storage, f->elem_size, (uint64_t) value ); + return true; +} + +inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldInfo * f, int32_t depth ) +{ + uint8_t * storage = (uint8_t *) base + f->offset; + if ( TableJsonIsMap( f ) ) + { + return TableJsonReadMap( in, (void *) storage, f, depth ); + } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + + if ( f->kind == 12 ) + { + int32_t length = 0; + if ( !TableJsonScanString( in, (char *) storage, f->array_bound, &length ) ) { return false; } + storage[length] = 0; + TableJsonSetCount( base, f, length ); + return true; + } + if ( TableJsonIsBytes( f ) ) + { + // base64 decodes STRAIGHT INTO the field's storage, six bits at a + // time — no window, no temporary, so a bytes(N) of any declared + // extent reads the same way. A base64 body carries no escapes, so a + // backslash in one is simply not an alphabet character. + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + in.pos++; + memset( storage, 0, (size_t) f->array_bound ); + TableJsonSetCount( base, f, 0 ); + const char * alphabet = TableJsonBase64Alphabet(); + int32_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + bool clamped = false; + bool malformed = false; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + const char * at = c != 0 ? strchr( alphabet, c ) : NULL; + if ( at == NULL ) { malformed = true; continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( at - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < f->array_bound ) + { + storage[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); + } + else + { + clamped = true; + } + } + } + if ( malformed ) + { + // a body that is not base64 is the wrong shape for the kind: the + // field keeps its default and the event is counted + in.report->kind_mismatch++; + return true; + } + if ( clamped ) { in.report->clamped++; } + TableJsonSetCount( base, f, placed ); + return true; + } + if ( TableJsonIsKeyed( f ) ) + { + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + // every slot back to its declared defaults first, so a key the text + // omits keeps them and a repeated field key cannot leave an earlier + // occurrence's slots standing + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + void * slot = storage + (int64_t) i * f->elem_size; + if ( f->kind == 13 ) { f->table->reset( slot ); } + else { memset( slot, 0, (size_t) f->elem_size ); } + } + char shape = TableJsonElementShape( f ); + // A KEYED OBJECT'S KEYS ARE KEYS: a variant named twice is a duplicate + // key like any other, last-wins and counted (§16.2). Tracked the way + // a table's own field keys are — a bounded, allocation-free bitmask; + // a vocabulary wider than this still reads, its repeats simply stop + // being counted. + uint64_t seen[8] = {}; + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t slot = -1; + for ( int64_t v = 0; v < f->array_bound; v++ ) + { + // nothing is stored for None, so "None" finds no slot and is + // an unknown key like any other name this reader cannot place + if ( !TableJsonKeyedSlotValid( f, v ) ) { continue; } + if ( strcmp( f->key_name( TableJsonKeyedSlotKey( v ) ), key ) == 0 ) { slot = v; break; } + } + if ( slot >= 0 && slot < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( slot & 63 ); + if ( ( seen[slot >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[slot >> 6] |= bit; + } + if ( slot < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, storage + slot * f->elem_size, f, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; + } + if ( f->is_array ) + { + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + in.pos++; + // LAST WINS has to be true of a repeated ARRAY key too, and it is + // wire-visible: a fixed array writes every slot, so a second, shorter + // occurrence overlaying a prefix would leave the first occurrence's + // tail standing. The field goes back to its declared defaults before + // this occurrence's elements are placed — the re-establishment a nested + // table and a union arm already get. A table element's defaults are + // its own (the reset hook); every other element kind's storage + // default is zero, which is what the generated array declares. + if ( f->kind == 13 ) + { + for ( int32_t i = 0; i < f->array_bound; i++ ) + { + f->table->reset( storage + (int64_t) i * f->elem_size ); + } + } + else + { + memset( storage, 0, (size_t) f->array_bound * (size_t) f->elem_size ); + } + TableJsonSetCount( base, f, 0 ); + int32_t placed = 0; + char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + if ( placed >= f->array_bound ) + { + // more elements than the reader's bound: the bounded prefix + // is kept and the excess counts, the wire's rule (§4) + in.report->clamped++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( f->kind == 17 ) + { + // an element of an ARRAY OF POINTERS (§2.1): null is a null slot, an + // object is the pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( storage + (int64_t) placed * f->elem_size, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + else if ( TableJsonValueShape( in ) != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + placed++; + } + else + { + if ( !TableJsonReadScalar( in, storage + (int64_t) placed * f->elem_size, f, depth + 1 ) ) { return false; } + placed++; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + // a fixed array's tail keeps the defaults the prefill left there, + // exactly as a short wire count does + TableJsonSetCount( base, f, placed ); + return true; + } + return TableJsonReadScalar( in, storage, f, depth ); +} + +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ); + +// ONE table object: keys are field keys, unknown ones are skipped and +// counted, a repeated key is last-wins and counted. The instance is already +// at its declared defaults when this is entered, so a key the text never +// mentions keeps the default an absent field takes on the wire (§4). +inline bool TableJsonReadTable( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth ) +{ + if ( depth > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + return TableJsonReadTableKeys( in, base, info, depth, NULL ); +} + +// The keys of an object whose brace is already consumed. A pointer's object +// opens the same way a table's does, but its FIRST key may be `&node` (§16.7) +// and the adapter that reads it has to scan the key to know — so it hands the +// key it scanned in as `first_key`, with the colon consumed, and this places +// it before scanning the rest. +inline bool TableJsonReadTableKeys( TableJsonIn & in, void * base, const TableTypeInfo * info, int32_t depth, const char * first_key ) +{ + // duplicate tracking, bounded and allocation-free: a table with more + // fields than this still reads, its repeats simply stop being counted + uint64_t seen[8] = {}; + for ( ;; ) + { + char key[kTableJsonMaxKey]; + char c = 0; + if ( first_key != NULL ) + { + memcpy( key, first_key, strlen( first_key ) + 1 ); // scanned into a buffer this size by the caller + first_key = NULL; + } + else + { + c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; return true; } + if ( c == 0 ) { in.bad = true; return false; } + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + } + int32_t index = -1; + for ( int32_t i = 0; i < info->num_fields; i++ ) + { + if ( strcmp( info->fields[i].json, key ) == 0 ) { index = i; break; } + } + if ( key[0] == '&' ) + { + // THE AMPERSAND PREFIX IS RESERVED TO THE FORM (docs/SPEC-TABLES.md + // §16.7). No declaration may take a key beginning with it, so this + // is never a field this build lacks — it is the sharing construct + // somewhere it cannot stand: `&node` is the FIRST key of a pointer's + // object and nothing else, and the adapter that reads a pointer + // has consumed it before these keys are read. MALFORMED, refused + // and counted; never counted as unknown, never skipped. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( index < 0 ) + { + in.report->unknown++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else + { + const TableFieldInfo * f = &info->fields[index]; + if ( index < 512 ) + { + uint64_t bit = uint64_t( 1 ) << ( index & 63 ); + if ( ( seen[index >> 6] & bit ) != 0 ) { in.report->duplicate++; } + seen[index >> 6] |= bit; + } + // PRESENCE OF THE KEY IS THE PRESENCE (§16.2): reaching this line + // is the key being present, so an optional is set present + // whatever its value — with one exception the page names: a JSON + // null, which reads as ABSENT rather than as a value. + char got = TableJsonValueShape( in ); + if ( f->kind == 17 && !f->is_array ) + { + // a pointer: null is a null pointer, an object is the pointee + // in place or an `&node` reference to one (§16.7), a string is + // a BYTE BUFFER's bytes (§2.5), and anything else is the wrong + // shape for the kind + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) base + f->offset, f->elem_size, 0 ); + } + else if ( got != TableJsonShape( f ) ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) base + f->offset, f, depth ) ) + { + return false; + } + } + else if ( f->optional && got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + // absent, and back at its defaults: a repeated key whose last + // occurrence is null must not leave an earlier value standing + if ( f->table != NULL ) { f->table->reset( (uint8_t *) base + f->offset ); } + else { memset( (uint8_t *) base + f->offset, 0, (size_t) f->elem_size ); } + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 0 ); + } + else + { + if ( got != TableJsonShape( f ) ) + { + // the wrong JSON type for the kind: skipped, never coerced + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, base, f, depth ) ) + { + return false; + } + if ( f->optional ) + { + TableJsonSetRaw( (uint8_t *) base + f->present_offset, 1, 1 ); + } + } + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; return true; } + in.bad = true; + return false; + } +} + +// ---- the two entry points the per-table wrappers name ---- + +inline bool TableJsonRead( void * value, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = NULL; + info->reset( value ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, value, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +inline int64_t TableJsonWrite( const void * value, const TableTypeInfo * info, char * buffer, int64_t capacity ) +{ + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = NULL; + if ( !TableJsonWriteValue( out, value, info, 0 ) ) { return -1; } + // THE CANONICAL TEXT ENDS WITH EXACTLY ONE NEWLINE (docs/SPEC-TABLES.md + // §16.1). Every writer emits it — this walk, the C# walk and + // "schema unpack" — and every reader accepts a text with or without one, + // because the trailing whitespace a read already skips is what makes the + // two the same text. It is a byte of the FORM rather than a file + // convention: a text that is written to a file, pasted into a diff and + // handed back through a pipe has to be one text in all three places, and a + // buffer whose last byte is a closing brace is the one shape that is not. + out.put( '\n' ); + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json walk: end ---- + +// ---- json graph walk: begin ---- +// +// THE VARIABLE CLASS's half of the text form (docs/SPEC-TABLES.md §16.7). The +// walk above places every kind but one; this defines the three adapters it +// calls for that one, and the two entry points a pointered table's wrappers +// name. The text is the fixed class's — a pointee is an object in place — and a +// node named more than once carries `&node`: defined once, with its fields, +// and referenced after by `{ "&node": N }` alone. + +// ---- the identity map ---- +// +// ONE map shape serves both directions. Writing keys it by a node's ADDRESS and +// counts the slots that name the node, so the second pass knows at a node's +// first occurrence whether it will be named again; reading keys it by the +// text's own label and answers the node it defined. Open addressing, a +// multiply-shift hash and quadrupling growth — TablePackMap's shape (§6.2), on +// the same terms: proportional to nodes, never to bytes, on the authoring +// side, and released before the call returns. + +struct TableJsonGraphEntry +{ + uint64_t key; // a node's address (write) or a label (read); 0 is an empty slot + int64_t count; // write: how many slots name this node + int64_t label; // write: the `&node` label assigned at its first write, 0 until then + uint8_t open; // the descent is still open: a reference here is a cycle (write), a self-reference (read) + uint32_t node; // read: the node's arena offset; 0 for a definition the reader dropped + const TableTypeInfo * type; // read: the node's table; NULL for a dropped one +}; + +struct TableJsonGraphMap +{ + TableJsonGraphEntry * entries; + int64_t capacity; // a power of two, or zero while empty + int64_t count; + TableAllocator allocator; // the caller's pair (§6.5): the builder's on read, the one handed to ToJson on write +}; + +inline void TableJsonGraphMapInit( TableJsonGraphMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TableJsonGraphMapShutdown( TableJsonGraphMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TableJsonGraphMapInit( map, map.allocator ); +} + +inline int64_t TableJsonGraphMapSlot( const TableJsonGraphMap & map, uint64_t key ) +{ + uint64_t hash = key * 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != 0 && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TableJsonGraphEntry * TableJsonGraphMapFind( TableJsonGraphMap & map, uint64_t key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +inline bool TableJsonGraphMapGrow( TableJsonGraphMap & map ) +{ + TableJsonGraphMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 64; + grown.count = 0; + grown.entries = (TableJsonGraphEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TableJsonGraphEntry ) ); // zeroed, by the pair's contract + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == 0 ) { continue; } + grown.entries[ TableJsonGraphMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// the entry for a key, made if it was not there; `taken` says which. NULL is the +// allocator refusing, and the walk refuses with it. +inline TableJsonGraphEntry * TableJsonGraphMapReach( TableJsonGraphMap & map, uint64_t key, bool & taken ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TableJsonGraphMapGrow( map ) ) { return NULL; } + } + TableJsonGraphEntry * entry = &map.entries[ TableJsonGraphMapSlot( map, key ) ]; + taken = entry->key != key; + if ( taken ) + { + entry->key = key; + map.count++; + } + return entry; +} + +// ---- reading: into a builder ---- + +struct TableJsonGraphIn +{ + TableWorker * worker; // where every node comes from + TableJsonGraphMap labels; // a label -> the node it defined +}; + +// `&node`'s value, the LABEL: a positive integer spelled as one — digits, no sign, no +// fraction, no exponent, no leading zero (§16.7). Anything else is malformed. +inline bool TableJsonScanLabel( TableJsonIn & in, uint64_t & label ) +{ + TableJsonSpace( in ); + if ( in.pos >= in.size || in.text[in.pos] < '1' || in.text[in.pos] > '9' ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + uint64_t value = 0; + while ( in.pos < in.size && in.text[in.pos] >= '0' && in.text[in.pos] <= '9' ) + { + uint64_t digit = (uint64_t) ( in.text[in.pos] - '0' ); + if ( value > ( UINT64_MAX - digit ) / 10 ) + { + in.report->malformed = true; + in.bad = true; + return false; + } + value = value * 10 + digit; + in.pos++; + } + label = value; + return true; +} + +// A BYTE BUFFER's text (docs/SPEC-TABLES.md §2.5, §16.2): a string. For a +// *string the string's bytes become the blob; for a *bytes the string is base64 +// and its decoded bytes do. The blob is allocated at EXACTLY the decoded +// length — the string is scanned once without keeping it to learn the length, +// and once into the node — so a blob of any size reads with no window and no +// bound to clamp against. A *bytes body that is not base64 is the wrong shape +// for the kind: the reference stays null and the event is counted. +inline bool TableJsonReadBlob( TableJsonIn & in, void * slot, const TableFieldInfo * f ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '"' ) { in.bad = true; return false; } + TableRef * ref = (TableRef *) slot; + ref->value = 0; + if ( strcmp( f->type_name, "string" ) == 0 ) + { + const int64_t mark = in.pos; + int32_t length = 0; + if ( !TableJsonScanString( in, NULL, 0, &length ) ) { return false; } + in.pos = mark; + char * data = TableStringEmplace( *graph->worker, *ref, NULL, (int64_t) length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int32_t placed = 0; + return TableJsonScanString( in, data, length, &placed ); + } + // base64: the alphabet characters decide the length, six bits apiece + const char * alphabet = TableJsonBase64Alphabet(); + const int64_t mark = in.pos + 1; + int64_t symbols = 0; + bool malformed = false; + in.pos++; + for ( ;; ) + { + if ( in.pos >= in.size ) { in.bad = true; return false; } + char c = in.text[in.pos++]; + if ( c == '"' ) { break; } + if ( c == '=' || malformed ) { continue; } + if ( c == 0 || strchr( alphabet, c ) == NULL ) { malformed = true; continue; } + symbols++; + } + if ( malformed ) + { + in.report->kind_mismatch++; + return true; + } + const int64_t length = ( symbols * 6 ) / 8; + uint8_t * data = TableBytesEmplace( *graph->worker, *ref, length ); + if ( data == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + int64_t placed = 0; + uint32_t accumulator = 0; + int32_t held = 0; + for ( int64_t at = mark; ; at++ ) + { + char c = in.text[at]; + if ( c == '"' ) { break; } + const char * symbol = c != '=' ? strchr( alphabet, c ) : NULL; + if ( symbol == NULL ) { continue; } + accumulator = ( accumulator << 6 ) | (uint32_t) ( symbol - alphabet ); + held += 6; + if ( held >= 8 ) + { + held -= 8; + if ( placed < length ) { data[placed++] = (uint8_t) ( ( accumulator >> held ) & 0xff ); } + } + } + return true; +} + +// A pointer's object. Its FIRST key decides what it is: `&node` naming a label not +// yet defined, with fields after it, is a DEFINITION; `&node` naming one already +// defined, alone, is a REFERENCE; any other key is a node named once, its +// object in place. The node comes from the +// builder's arena, and the slot holds its arena offset (§6.3). A pointer whose +// target is a BYTE BUFFER — no table — takes a string instead (§2.5). +inline bool TableJsonReadPointer( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( f->table == NULL ) { return TableJsonReadBlob( in, slot, f ); } + // the pointee nests one level down, exactly as a by-value table does, and + // takes the same cap: a chain nests as deep as it is long (§16.7) + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + in.pos++; + char c = TableJsonPeek( in ); + if ( c == '}' ) + { + // an empty object: a node at its defaults, named once + in.pos++; + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } // the arena refused + return true; + } + if ( c == 0 ) { in.bad = true; return false; } + char key[kTableJsonMaxKey]; + int32_t key_length = 0; + if ( !TableJsonScanString( in, key, kTableJsonMaxKey - 1, &key_length ) ) { return false; } + key[key_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + if ( strcmp( key, "&node" ) != 0 ) + { + // a node named once: the pointee's object in place, and this key is + // its first field — unless it is the reserved prefix under a spelling + // this form does not have, which ReadTableKeys refuses + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return TableJsonReadTableKeys( in, node, f->table, depth + 1, key ); + } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->labels, label, taken ); + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + // ONE SPELLING, and what follows the label says which half it is: fields + // after a label the text has not defined DEFINE it, and a label alone that + // the text has defined REFERS to it. The other two are malformed — a label + // alone that the text never defined, which would otherwise read as a default + // node under a silent report, and a field after a label already defined, + // which would be a second definition. That is what keeps a typo loud. + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; c = TableJsonPeek( in ); } + bool bare = c == '}'; + if ( bare == taken ) { in.report->malformed = true; in.bad = true; return false; } + if ( bare ) + { + // A REFERENCE. A label is defined when its object CLOSES, so a + // reference met inside its own definition — at any depth of by-value + // nesting — names a node whose descent is still open: the cycle the + // wire refuses (§3.1), refused here where it is written. A definition + // the reader dropped names no node, so the slot stays null with + // nothing more counted — the drop was counted where it happened. A + // node of another table than the slot declares is a kind mismatch, as + // on the wire. + in.pos++; + if ( entry->open != 0 ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + if ( entry->type == NULL ) + { + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + if ( entry->type != f->table ) + { + memcpy( slot, &ref, sizeof( ref ) ); + in.report->kind_mismatch++; + return true; + } + ref.value = (int64_t) entry->node; + memcpy( slot, &ref, sizeof( ref ) ); + return true; + } + // A DEFINITION: the node is allocated, the label is its, and the keys after + // `&node` are its fields. The entry is OPEN until the object closes, so a + // reference to the label from inside the node's own fields is refused as + // the cycle it is; the node and its table are filled in at the close. + void * node = f->emplace( *graph->worker, slot ); + if ( node == NULL ) { in.report->malformed = true; in.bad = true; return false; } + entry->open = 1; + if ( !TableJsonReadTableKeys( in, node, f->table, depth + 1, NULL ) ) { return false; } + entry = TableJsonGraphMapFind( graph->labels, label ); // the map may have grown under the descent + if ( entry == NULL ) { in.report->malformed = true; in.bad = true; return false; } + TableRef ref; + memcpy( &ref, slot, sizeof( ref ) ); + entry->node = (uint32_t) ref.value; + entry->type = f->table; + entry->open = 0; + return true; +} + +// An `&`-prefixed key opening an object the walk is SKIPPING — a value past an +// array's bound, an unknown key's value, a value of the wrong shape. A +// definition in there still takes its label, so the numbering survives whatever +// the storage could not hold (§16.7): the label is registered with no node, and a +// reference to it reads null. Any other prefixed key is the reserved prefix +// out of place. +inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL || strcmp( key, "&node" ) != 0 ) { in.report->malformed = true; in.bad = true; return false; } + uint64_t label = 0; + if ( !TableJsonScanLabel( in, label ) ) { return false; } + bool taken = false; + if ( TableJsonGraphMapReach( graph->labels, label, taken ) == NULL ) { in.report->malformed = true; in.bad = true; return false; } + return true; // a fresh entry is node 0, type NULL: a definition with no node +} + +// ---- writing: from a region's const root ---- + +struct TableJsonGraphOut +{ + TableJsonGraphMap nodes; // a node's address -> how many slots name it, and its `&node` once assigned + bool counting; // PASS ONE: count the references, refuse a cycle, emit nothing + int64_t next_label; +}; + +// The node a slot names: null as `null`, a node named once as its object in +// place, and a node named more than once under the construct. Which of the +// last two it is was learned in pass one; pass two spells it. +inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphOut * graph = (TableJsonGraphOut *) out.graph; + if ( graph == NULL ) { return false; } + const void * node = f->resolve( slot ); + if ( node == NULL ) + { + out.raw( "null", 4 ); + return true; + } + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph->nodes, (uint64_t) (uintptr_t) node, taken ); + if ( entry == NULL ) { return false; } + if ( f->table == NULL ) + { + // A BYTE BUFFER (§2.5, §16.7): its text is a string, which has no + // first key to carry `&node`, so a blob named from more than one + // slot has no spelling this form can carry and the graph is refused — + // as a shared node with nothing to write is. A blob named once is its + // bytes in place: base64 for a *bytes, the string itself for a *string. + if ( graph->counting ) { entry->count++; return true; } + if ( entry->count > 1 ) { return false; } + const TableBlob * blob = (const TableBlob *) node; + if ( blob->length > (uint32_t) 0x7fffffff ) { return false; } + if ( strcmp( f->type_name, "string" ) == 0 ) { TableJsonWriteString( out, (const char *) ( blob + 1 ), (int32_t) blob->length ); } + else { TableJsonWriteBase64( out, (const uint8_t *) ( blob + 1 ), (int32_t) blob->length ); } + return true; + } + if ( graph->counting ) + { + // PASS ONE: one visit per node, every slot that names it counted, and + // a reference to a node whose descent is still open is a cycle — + // refused here as the wire refuses it (§3.1) + entry->count++; + if ( !taken ) { return entry->open == 0; } + entry->open = 1; + if ( !TableJsonWriteValue( out, node, f->table, depth ) ) { return false; } + entry = TableJsonGraphMapFind( graph->nodes, (uint64_t) (uintptr_t) node ); // the map may have grown under the descent + if ( entry == NULL ) { return false; } + entry->open = 0; + return true; + } + // PASS TWO: a node named once is its object in place; a node named more + // than once is DEFINED at its first occurrence — `&node` first, then its + // fields — and REFERENCED by `&node` alone after that, spelled the same way at + // every site. Labels run from 1 in first-write order and are the text's own, + // so a stray number in a hand-edited text is most often one never defined. + if ( entry->count <= 1 ) + { + return TableJsonWriteValue( out, node, f->table, depth ); + } + if ( depth > kTableJsonMaxDepth ) { return false; } + if ( entry->label != 0 ) + { + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + out.line( depth ); + out.put( '}' ); + return true; + } + entry->label = ++graph->next_label; + out.put( '{' ); + out.line( depth + 1 ); + out.raw( "\"&node\": ", 9 ); + TableJsonWriteUnsigned( out, (uint64_t) entry->label ); + bool any = true; + int64_t before = out.offset; + if ( !TableJsonWriteFields( out, node, f->table, depth, any ) ) { return false; } + // a definition carries at least one field, because a label alone is a + // reference: a shared node with nothing to write has no definition this + // form can spell, and the writer refuses it as it refuses any value it + // cannot spell (§16.3) + if ( out.offset == before ) { return false; } + out.line( depth ); + out.put( '}' ); + return true; +} + +// ---- the two entry points a pointered table's wrappers name ---- + +// The text into the builder's root. Every node the text names is allocated in +// the builder's arena through the field's own Emplace; the label map is the +// walk's, released before this returns. The root itself takes no label — nothing +// may name it (§16.7) — so an `&node` at the root is refused like any other key +// of the prefix. +inline bool TableJsonReadGraph( TableWorker & worker, void * root, const TableTypeInfo * info, const char * text, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + if ( worker.arena == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } + TableJsonGraphIn graph; + graph.worker = &worker; + TableJsonGraphMapInit( graph.labels, worker.arena->allocator ); + TableJsonIn in; + in.text = text; + in.size = bytes; + in.pos = 0; + in.report = report != NULL ? report : &ignored; + in.bad = false; + in.graph = &graph; + info->reset( root ); + if ( text == NULL || bytes < 0 ) + { + in.report->malformed = true; + return false; + } + bool ok = TableJsonReadTable( in, root, info, 0 ); + if ( ok ) + { + TableJsonSpace( in ); + if ( in.pos != in.size ) { in.bad = true; } // trailing rubbish is not one text + } + TableJsonGraphMapShutdown( graph.labels ); + if ( in.bad || !ok ) + { + in.report->malformed = true; + return false; + } + return true; +} + +// The text of a region's const root: measured when the buffer is NULL, written +// when it is not, over one code path. Two passes over one walk — the first +// counts how many slots name each node and refuses a cycle, the second writes +// — so a node's first occurrence knows whether it will be named again. The +// ROOT's entry is open for the whole first pass, so a reference back at it is +// the cycle it is (§3.1), and it takes no label. +inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * info, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + if ( root == NULL ) { return -1; } + TableJsonGraphOut graph; + TableJsonGraphMapInit( graph.nodes, allocator ); + graph.counting = true; + graph.next_label = 0; + bool taken = false; + TableJsonGraphEntry * entry = TableJsonGraphMapReach( graph.nodes, (uint64_t) (uintptr_t) root, taken ); + if ( entry == NULL ) { TableJsonGraphMapShutdown( graph.nodes ); return -1; } + entry->open = 1; + TableJsonOut count; + count.buffer = NULL; + count.capacity = 0; + count.offset = 0; + count.overflow = false; + count.graph = &graph; + bool ok = TableJsonWriteValue( count, root, info, 0 ); + graph.counting = false; + TableJsonOut out; + out.buffer = buffer; + out.capacity = capacity; + out.offset = 0; + out.overflow = false; + out.graph = &graph; + if ( ok ) { ok = TableJsonWriteValue( out, root, info, 0 ); } + TableJsonGraphMapShutdown( graph.nodes ); + if ( !ok ) { return -1; } + out.put( '\n' ); // the canonical text ends with exactly one newline (§16.1) + if ( out.overflow ) { return -1; } + return out.offset; +} + +// ---- json graph walk: end ---- + +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + +// ---- json map walk: begin ---- + +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} + +// the entry's two rows: fields[0] IS the key and fields[1] IS the value, which +// is what makes a user's own table of pairs the same bytes (§2.8) +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } + +inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } +inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } + +// AN INTEGER KEY IS THE INTEGER'S DECIMAL SPELLING, QUOTED, because a JSON +// object's keys are strings. Written digit by digit so no locale can move it. +inline void TableJsonWriteMapIntegerKey( TableJsonOut & out, const void * storage, const TableFieldInfo * key ) +{ + uint64_t magnitude = 0; + bool negative = false; + if ( TableJsonMapKeySigned( key ) ) + { + int64_t value = 0; + switch ( key->kind ) + { + case 2: value = (int64_t) *(const int8_t *) storage; break; + case 3: value = (int64_t) *(const int16_t *) storage; break; + case 4: value = (int64_t) *(const int32_t *) storage; break; + default: value = *(const int64_t *) storage; break; + } + negative = value < 0; + magnitude = negative ? ( ~(uint64_t) value ) + 1 : (uint64_t) value; + } + else + { + switch ( key->kind ) + { + case 6: magnitude = (uint64_t) *(const uint8_t *) storage; break; + case 7: magnitude = (uint64_t) *(const uint16_t *) storage; break; + case 8: magnitude = (uint64_t) *(const uint32_t *) storage; break; + default: magnitude = *(const uint64_t *) storage; break; + } + } + char digits[24]; + int32_t at = (int32_t) sizeof( digits ); + do { digits[--at] = (char) ( '0' + ( magnitude % 10 ) ); magnitude /= 10; } while ( magnitude != 0 ); + if ( negative ) { digits[--at] = '-'; } + TableJsonWriteString( out, digits + at, (int32_t) sizeof( digits ) - at ); +} + +inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const TableFieldInfo * key ) +{ + const uint8_t * storage = (const uint8_t *) entry + key->offset; + if ( TableJsonMapKeyIsString( key ) ) + { + // A STRING KEY IS THE STRING (§2.8): every JSON key of a map object is + // a KEY OF THE MAP and none is a field key, so the `&` prefix §16.7 + // reserves for field keys is ordinary data here. + TableJsonWriteString( out, (const char *) storage, *(const int32_t *) ( (const uint8_t *) entry + key->count_offset ) ); + return; + } + TableJsonWriteMapIntegerKey( out, (const void *) storage, key ); +} + +// ToJson WRITES ENTRIES IN ASCENDING KEY ORDER, so unpack then pack is +// byte-stable and a diff of two texts is a diff of two maps (§2.8, §17.2). +// A region holds them in that order already, so this is the array in place. +inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "{}", 2 ); return true; } + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); + out.put( '{' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); + TableJsonWriteMapKey( out, entry, key ); + out.raw( ": ", 2 ); + if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( '}' ); + return true; +} + +// AN INTEGER KEY IS READ BY §16.2's INTEGER RULE AND BY NOTHING ELSE, so +// "2.0" and "1e3" are the integers 2 and 1000 and "-0" is zero. The token is +// walked as a JSON number over its own bytes; a token that rule calls +// malformed makes the KEY malformed, and a genuinely fractional value, or one +// outside the key kind's range, is kind_mismatch for that entry. +inline bool TableJsonMapKeyValue( const char * token, int32_t length, const TableFieldInfo * key, + int64_t & value, bool & fits ) +{ + fits = false; + TableReport scratch; + TableJsonIn probe = { token, (int64_t) length, 0, &scratch, false, NULL }; + bool integral = false; + if ( !TableJsonWalkNumber( probe, &integral ) ) { return false; } + if ( probe.pos != (int64_t) length ) { return false; } // trailing bytes: not a number + if ( !integral ) + { + const double d = TableJsonTokenDouble( token, length, false ); + if ( !TableJsonFinite( d ) ) { return true; } // a value no key kind holds + const double whole = d < 0 ? -d : d; + if ( whole != (double) (int64_t) whole ) { return true; } // genuinely fractional + } + bool saturated = false; + const bool is_signed = TableJsonMapKeySigned( key ); + value = integral ? TableJsonTokenInteger( token, length, is_signed, &saturated ) + : (int64_t) TableJsonTokenDouble( token, length, false ); + if ( saturated ) { return true; } // outside every width: kind_mismatch, never clamped + switch ( key->kind ) + { + case 2: fits = value >= -128 && value <= 127; break; + case 3: fits = value >= -32768 && value <= 32767; break; + case 4: fits = value >= -2147483647 - 1 && value <= 2147483647; break; + case 5: fits = true; break; + case 6: fits = value >= 0 && value <= 255; break; + case 7: fits = value >= 0 && value <= 65535; break; + case 8: fits = value >= 0 && (uint64_t) value <= 4294967295ull; break; + default: fits = integral; break; // uint64: the token's own magnitude + } + return true; +} + +// FromJson READS KEYS IN WHATEVER ORDER THE TEXT GIVES THEM. A repeated key is +// last-wins and counted duplicate, the object rule (§16.2) applied inside the +// map. An empty object is an empty map, and null is kind_mismatch. +inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '{' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + const TableFieldInfo * key = TableJsonMapKeyField( f ); + const TableFieldInfo * value = TableJsonMapValueField( f ); + const char shape = TableJsonShape( value ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == '}' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + char token[kTableJsonMaxKey]; + int32_t token_length = 0; + if ( !TableJsonScanString( in, token, kTableJsonMaxKey - 1, &token_length ) ) { return false; } + token[token_length] = 0; + if ( TableJsonPeek( in ) != ':' ) { in.bad = true; return false; } + in.pos++; + int64_t key_value = 0; + bool place = true; + if ( !TableJsonMapKeyIsString( key ) ) + { + bool fits = false; + if ( !TableJsonMapKeyValue( token, token_length, key, key_value, fits ) ) + { + // A MALFORMED KEY STOPS THE READ where §16.1's rule stops it, + // with the instance holding what was placed before the stop. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( !fits ) { in.report->kind_mismatch++; place = false; } + } + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; + if ( place && entry == NULL ) + { + // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the + // wire's rule, because a clamped key is a merged entry (§2.8). + in.report->clamped++; + } + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) + { + in.report->duplicate++; // last-wins, the object rule inside the map + } + const char got = TableJsonValueShape( in ); + if ( entry == NULL ) + { + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( value->kind == 17 && !value->is_array ) + { + // A POINTER VALUE IS SHARED EXACTLY AS A POINTER FIELD IS (§2.8): + // null is a null slot, an object is the pointee in place or an + // &node reference to one (§16.7), anything else is the wrong shape — + // the same three the field-key loop gives a pointer field, because + // an entry's value IS a field line. + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) entry + value->offset, value->elem_size, 0 ); + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, (uint8_t *) entry + value->offset, value, depth + 1 ) ) + { + return false; + } + } + else if ( got != shape ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadField( in, entry, value, depth + 1 ) ) + { + return false; + } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == '}' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json map walk: end ---- + +// ---- json list walk: begin ---- + +// an unbounded array is the out-of-line array that is not a map (§8.1) +inline bool TableJsonIsList( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && !TableJsonIsMap( f ); +} + +// ToJson WRITES THE ELEMENTS IN INDEX ORDER, which is the only order there is, +// so unpack then pack is byte-stable without a rule of its own (§2.9, §17.2). +// A region holds the array in place, so this steps it at the descriptor's pitch. +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) +{ + const int32_t count = TableJsonExtentCount( slot ); + if ( count == 0 ) { out.raw( "[]", 2 ); return true; } + const uint8_t * elements = TableJsonExtentElements( slot ); + out.put( '[' ); + for ( int32_t i = 0; i < count; i++ ) + { + if ( i > 0 ) { out.put( ',' ); } + out.line( depth + 1 ); + const uint8_t * element = elements + (int64_t) i * f->elem_size; + if ( f->kind == 17 ) + { + // a []*T's elements take the pointer row (§16.7): the pointee's + // object in place, null, or `&node` for a shared one + if ( !TableJsonWritePointer( out, element, f, depth + 1 ) ) { return false; } + } + else if ( !TableJsonWriteScalar( out, element, f, depth + 1 ) ) { return false; } + } + out.line( depth ); + out.put( ']' ); + return true; +} + +// FromJson READS EVERY ELEMENT THE TEXT CARRIES, appending each through the +// descriptor's place resolver: `[]` is an empty list, and null is +// kind_mismatch, the array row's own rule (§16.2). LAST WINS holds for a +// repeated key: the list goes back to EMPTY before this occurrence's elements +// land, the builder's storage being reclaimed at reset (§2.9). +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ) +{ + TableJsonGraphIn * graph = (TableJsonGraphIn *) in.graph; + if ( graph == NULL ) { in.report->malformed = true; in.bad = true; return false; } + if ( TableJsonPeek( in ) != '[' ) { in.bad = true; return false; } + if ( depth + 1 > kTableJsonMaxDepth ) { in.bad = true; return false; } + in.pos++; + TableJsonSetRaw( (uint8_t *) slot, 8, 0 ); + TableJsonSetRaw( (uint8_t *) slot + 8, 4, 0 ); + const char shape = TableJsonElementShape( f ); + for ( ;; ) + { + char c = TableJsonPeek( in ); + if ( c == ']' ) { in.pos++; break; } + if ( c == 0 ) { in.bad = true; return false; } + void * element = f->place( *graph->worker, slot, NULL, 0, 0 ); + if ( element == NULL ) + { + // NOT ADDED: the arena could not carve another segment, or the + // count met the int32 cap. The text cannot be placed whole, and + // the read stops where §16.1's rule stops it. + in.report->malformed = true; + in.bad = true; + return false; + } + if ( f->kind == 17 ) + { + // an element of a []*T (§2.9): null is a null slot, an object is the + // pointee in place or an `&node` reference (§16.7) + char got = TableJsonValueShape( in ); + if ( got == 'z' ) + { + if ( !TableJsonLiteral( in, "null" ) ) { return false; } + TableJsonSetRaw( (uint8_t *) element, f->elem_size, 0 ); + } + else if ( got != 'o' ) + { + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadPointer( in, element, f, depth + 1 ) ) { return false; } + } + else if ( TableJsonValueShape( in ) != shape ) + { + // the wrong shape for the element kind: the slot keeps its + // defaults and the event counts, the array row's rule (§16.2) + in.report->kind_mismatch++; + if ( !TableJsonSkipValue( in, depth + 1 ) ) { return false; } + } + else if ( !TableJsonReadScalar( in, element, f, depth + 1 ) ) { return false; } + c = TableJsonPeek( in ); + if ( c == ',' ) { in.pos++; continue; } // a trailing comma is accepted + if ( c == ']' ) { in.pos++; break; } + in.bad = true; + return false; + } + return true; +} + +// ---- json list walk: end ---- + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_JSON + +namespace listdemo { + +bool PhotoFromJson( Photo & value, const char * text, int64_t bytes, TableReport * report ) +{ + return TableJsonRead( &value, PhotoTableType(), text, bytes, report ); +} + +int64_t PhotoToJsonMeasure( const Photo & value ) +{ + return TableJsonWrite( &value, PhotoTableType(), NULL, 0 ); +} + +int64_t PhotoToJson( const Photo & value, char * buffer, int64_t capacity ) +{ + return TableJsonWrite( &value, PhotoTableType(), buffer, capacity ); +} + +bool AlbumFromJson( AlbumBuilder & builder, const char * text, int64_t bytes, TableReport * report ) +{ + Album * root = builder.GetRoot(); + if ( root == NULL ) { if ( report != NULL ) { report->malformed = true; } return false; } // locked, or the root allocation failed + return TableJsonReadGraph( builder.main, root, AlbumTableType(), text, bytes, report ); +} + +int64_t AlbumToJsonMeasure( const Album * root, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, AlbumTableType(), NULL, 0, allocator ); +} + +int64_t AlbumToJson( const Album * root, char * buffer, int64_t capacity, TableAllocator allocator ) +{ + return TableJsonWriteGraph( root, AlbumTableType(), buffer, capacity, allocator ); +} + +} // namespace listdemo diff --git a/testdata/golden/tables/lists/SharedTable.h b/testdata/golden/tables/lists/SharedTable.h new file mode 100644 index 000000000..de8d0934d --- /dev/null +++ b/testdata/golden/tables/lists/SharedTable.h @@ -0,0 +1,5533 @@ +// Code generated by the schema compiler from Shared.schema. DO NOT EDIT. +// SPDX-License-Identifier: NONE — this generated output is yours, under terms of +// your choice. See the LICENSE exception in the schema compiler; the compiler is +// AGPL-3.0, its output is not. +// package listdemo — protocol id 0xa5fbe602c119cdd9 (packets only: tables version by field id, not by protocol id) +// The TABLE wire (evolution-tolerant, docs/SPEC-TABLES.md): no serialize +// dependency — includable from any TU. + +#pragma once + +#include +#include // the prefill's scalar-array fills +#include // offsetof, for the reflection descriptors + +// ---- the hooks (docs/USAGE.md, "the C++ table runtime's hooks") ---- +// +// schema_assert — the runtime's own assert, and the refusal a debugger reads. +// NDEBUG removes it, exactly as it removes assert. A caller who already routes +// serialize's asserts writes `#define schema_assert serialize_assert` before +// including this header and both halves land in one handler. +#ifndef schema_assert +#include +#define schema_assert assert +#endif // #ifndef schema_assert + +// schema_fatal — what stands after the assert on a path that cannot continue. +// NDEBUG does not remove it. Supply it and is never included. +#ifndef schema_fatal +#include // abort +#define schema_fatal abort +#endif // #ifndef schema_fatal + +// schema_allocate / schema_release — what "no allocator handed in" means for +// this program. schema_allocate hands back ZEROED bytes and NULL on failure: +// an arena segment is copied whole, padding included, so anything left +// uninitialized here would reach a packed region. Supply both and +// is never included; hand a TableAllocator to a builder to route one +// structure's allocations somewhere else again. +#ifndef schema_allocate +#include // calloc, free +#define schema_allocate( bytes ) calloc( (size_t) 1, (size_t) ( bytes ) ) +#define schema_release( pointer ) free( pointer ) +#endif // #ifndef schema_allocate +#include // a node's lifetime starts in arena storage (placement new) +#include // one atomic per slab: the arena is lock-free by ownership + +#include "Shared.h" + +#ifndef LISTDEMO_SCHEMA_TABLE_PRIMITIVES +#define LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +// THE CODEC DOES NOT DEPEND ON THE COMPILER'S INLINING BUDGET. A table of a +// realistic field count emits one large body per type, and the cursor a body +// writes through lives in the caller's `TableWriter`: across a call boundary +// that cursor round-trips through memory, and a `uint8_t *` store may alias the +// writer itself, so every put reloads it. When a budget runs out mid-body the +// codec silently degrades to that shape. Forcing the primitives and the +// fixed-class bodies inline is what keeps the cursor in registers and lets +// adjacent constant framing bytes merge into one store. +#if defined( _MSC_VER ) +#define LISTDEMO_TABLE_INLINE __forceinline +#elif defined( __GNUC__ ) || defined( __clang__ ) +#define LISTDEMO_TABLE_INLINE inline __attribute__(( always_inline )) +#else +#define LISTDEMO_TABLE_INLINE inline +#endif + +namespace listdemo { + +// WHY A READ WAS REFUSED, by name (docs/SPEC-TABLES.md §3.3, §11). A REFUSAL +// is not one of §4's events: nothing is decoded, no counter moves and no +// damage is reported, so five zero counters and a false flag are what a clean +// read prints too and only the verdict tells them apart. The reason says which +// refusal it was. +// +// This is the MESSAGE PATH's vocabulary and not the cooked form's (§7.4): a +// caller meeting one of these has been refused a MESSAGE on a connection, +// which is a different recovery with a different owner than a file a header +// match turned down. +enum TableMessageReason +{ + newer_form, // a FORM BYTE this reader does not carry (§3) + no_vocabulary, // no table for this connection: the message arrived before the announcement, or after a refused one + second_announcement, // a second announcement on a connection: it sets nothing, amends nothing, and the connection closes + vocabulary_too_large, // an announcement above the receiver's declared bound, refused before an entry is touched + message_form_as_file // a form 2 wire where a FILE was expected: its table is somewhere else +}; + +// The table-wire read report — the permissive contract's ledger. Silence +// (all zero) means the data matched this reader's schema exactly. +struct TableReport +{ + int32_t unknown = 0; // unknown field ids skipped (newer data) + int32_t kind_mismatch = 0; // known id, changed type — skipped, never misdecoded + int32_t clamped = 0; // out-of-range values clamped to declared bounds + // a key the TEXT form saw twice: last wins, and the repeat is counted + // (docs/SPEC-TABLES.md §16.2). The wire never raises it — a body carrying an + // id twice is legal input whose last occurrence wins, silently (§3). + int32_t duplicate = 0; + bool malformed = false; // framing damage; decode stopped, partial result kept + // THE REFUSAL VERDICT, which is not one of §4's events and moves no counter + // (docs/SPEC-TABLES.md §3): a FORM BYTE this reader does not carry. Five + // zero counters and a false flag are what a clean read prints too, so the + // verdict is what tells the two apart. + bool refused = false; + // WHICH refusal, and it is read only when refused is set: a read that + // was not refused has no reason, and this member is the one the caller + // must not look at then (docs/SPEC-TABLES.md §3.3). + TableMessageReason reason = newer_form; +}; + + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; +// ---- reflection (tables only, docs/SPEC-TABLES.md) ---- +// +// Static field descriptors for every type in the table closure: name, wire +// id/kind, storage offset, bounds, ranges, enum names and branch guards — +// enough to walk, print, diff, edit or bind any table value at runtime with +// no RTTI and no schema files. TableType() returns X's descriptor. + +struct TableTypeInfo; + +// One arm of a union field: where its payload sits inside the union's storage +// and what its payload looks like. The arm's NAME and its table-wire id come +// from the field's enum_name/variant_id functions at the same tag, so nothing +// is spelled twice (docs/SPEC-TABLES.md §8). +struct TableFieldInfo; + +struct TableUnionArmInfo +{ + uint32_t offset; // offsetof the arm's payload within the union storage + const TableTypeInfo * table; // the arm payload's descriptor, or NULL + // AN ARM IS A FIELD LINE (docs/SPEC-TABLES.md §2.6): an arm that names no + // declared type or table carries the FIELD descriptor a field of that + // type would carry instead — offsets taken within the union storage — so + // a generic walk meets an arm's kind, width, bounds and companions where + // it meets a field's. Exactly one of the two is non-NULL on a set arm. + const TableFieldInfo * field; + uint32_t size; // the arm's whole storage, which selection zero-establishes +}; + +// A union field's shape: the tag, and the arms indexed by it. Arms run +// [0, enum_max]; index 0 is the EMPTY arm and carries no payload. +struct TableUnionInfo +{ + uint32_t tag_offset; // offsetof the tag within the union storage + uint32_t tag_size; // sizeof the tag + const TableUnionArmInfo * arms; +}; + +// The exact raw range of a wide-kind field (docs/SPEC-TABLES.md §8.2): two 128-bit +// values as 64-bit lanes, low lane first, two's complement for the signed kinds. +struct TableWideRange +{ + uint64_t lo[2]; + uint64_t hi[2]; +}; + +// the arena's allocation front, defined with the variable-length runtime +// below; a descriptor names it only through a pointer parameter. +struct TableWorker; + +struct TableFieldInfo +{ + const char * name; // schema field name, e.g. "health" + const char * json; // the TEXT form's key: the json = "key" attribute, else name (§16.3) + const char * type_name; // schema type name, e.g. "float32", "Grade" + uint64_t id; // table-wire field id: fnv1a64 of the name, of the was alias after a rename (§5) + uint8_t kind; // table-wire kind; for arrays/strings/bytes, the ELEMENT kind + bool is_array; // fixed or counted array (bytes included) + bool is_pointer; // a *T pointer field: storage is an 8-byte TableRef; the target is a table + // THE TWO THE TEXT FORM NEEDS (docs/SPEC-TABLES.md §16.7), and they + // are here for the same reason is_pointer is: the walk is ONE walk + // over descriptors and cannot spell a target's own At or + // Emplace. `resolve` reads a slot in a REGION and answers the + // node it names, or NULL; `emplace` allocates one in a BUILDER's + // arena and points the slot at it. NULL on every field that is not + // a pointer, and emitted only in a unit that declares one. + const void * (*resolve)( const void * slot ); + void * (*emplace)( TableWorker & worker, void * slot ); + bool counted; // a _count/_length int32 companion exists (counted arrays, strings, bytes) + bool optional; // a ?T field: a _present bool companion decides whether it rides + int32_t array_bound; // array capacity / string max length; 0 for plain scalars + uint32_t offset; // offsetof the storage member + uint32_t elem_size; // sizeof the member (element size for arrays) + uint32_t count_offset; // offsetof the _count/_length companion, or 0xffffffff + uint32_t present_offset; // offsetof the _present companion, or 0xffffffff + const TableTypeInfo * table; // nested table's descriptor, or NULL + bool has_range; // a declared [min, max] (int or float) + double range_min; // NOTE: int64 ranges beyond 2^53 lose precision here + double range_max; + // the WIDE kinds (18-29, docs/SPEC-TABLES.md §3, §8.2): frac_bits is a fixed + // field's F — its storage holds units × 2^F — and wide is the declared + // range on that RAW scale, exact, as two 128-bit two's-complement values + // in 64-bit lanes (low lane first). NULL where the declaration bounds + // nothing (a bare uint128) and for every other kind; frac_bits is 0 for + // every kind that is not fixed-point. range_min/range_max still carry + // the declared bounds as doubles — whole units for a fixed field — for + // a walker that only shows them. + uint8_t frac_bits; + const TableWideRange * wide; + int64_t enum_max; // enums: highest valid value (None = 0 always valid); + // unions: the arm count (tag range [0, enum_max]); + // flags: the highest declared BIT INDEX; else -1 + // the vocabulary's names, indexed the same way enum_max bounds: an enum's + // value -> name, a union's tag -> arm name, a FLAGS field's bit index -> + // variant name. NULL for every other kind. + const char * (*enum_name)( uint64_t value ); + // the TABLE-WIRE id of one variant (docs/SPEC-TABLES.md §5): for an enum, the + // hash of the variant's name; for a union, the hash of the arm's name. + // 0 is the reserved id — an enum's None, a union's empty. NULL for every + // other kind — a FLAGS field's variants have no per-variant wire id (§4), + // so a NULL here beside a non-NULL enum_name is what says "flags". + // Walk [0, enum_max] to enumerate a vocabulary and its ids. + uint64_t (*variant_id)( uint64_t value ); + // an ENUM-KEYED array (docs/SPEC-TABLES.md §2.4): the array has one slot per + // variant of key_type_name, indexed by the variant's value, and its slots + // ride under variant ids rather than positions. key_name and key_id are + // the key's vocabulary — walk [0, array_bound) to print slots by name. + // NULL on every other field. + const char * key_type_name; + const char * (*key_name)( uint64_t value ); + uint64_t (*key_id)( uint64_t value ); + // union fields: the tag and its arms, behind a function so the whole + // descriptor stays CONSTANT-INITIALISED (a captureless lambda converts to + // a function pointer at compile time; the arms themselves are a static + // inside it). NULL for every other kind. + const TableUnionInfo * (*arms)(); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded +}; + +struct TableTypeInfo +{ + const char * name; // schema type name + uint32_t size; // sizeof the storage struct + int32_t num_fields; + const TableFieldInfo * fields; + // put one instance back at its declared defaults, in place. A generic + // walker that fills a value has to be able to establish the defaults an + // absent field takes, and it holds no type to spell — this is the one + // thing the descriptors could not express without it. Placement-new + // value-init, exactly what the wire's read path does, and no temporary. + void (*reset)( void * storage ); + // the DERIVED mode (docs/SPEC-TABLES.md): false = fixed-size, a plain + // relocatable struct; true = variable-length, built through a Builder + // and read through a region root. Nobody declares it; the compiler + // works it out. + bool variable; +}; + +struct TableWriter +{ + uint8_t * buffer; + int64_t capacity; + int64_t offset = 0; + bool overflow = false; + + // the parameters do not repeat the member names: a parameter that hides a + // member is a warning the estate's compilers disagree about (gcc's + // -Wshadow and cl's C4458 refuse it, clang's -Wshadow does not), and this + // is a header a consumer compiles under its OWN flags + TableWriter( uint8_t * to_buffer, int64_t to_capacity ) : buffer( to_buffer ), capacity( to_capacity ) {} + + LISTDEMO_TABLE_INLINE void raw( const void * data, int64_t bytes ) + { + if ( offset + bytes > capacity ) { overflow = true; return; } + memcpy( buffer + offset, data, (size_t) bytes ); + offset += bytes; + } + LISTDEMO_TABLE_INLINE void put8( uint8_t v ) { raw( &v, 1 ); } + LISTDEMO_TABLE_INLINE void put16( uint16_t v ) { uint8_t b[2] = { uint8_t( v ), uint8_t( v >> 8 ) }; raw( b, 2 ); } + LISTDEMO_TABLE_INLINE void put32( uint32_t v ) { uint8_t b[4] = { uint8_t( v ), uint8_t( v >> 8 ), uint8_t( v >> 16 ), uint8_t( v >> 24 ) }; raw( b, 4 ); } + LISTDEMO_TABLE_INLINE void put64( uint64_t v ) { put32( uint32_t( v ) ); put32( uint32_t( v >> 32 ) ); } + // a 128-bit value as two lanes, the low half first (docs/SPEC-TABLES.md §3) + LISTDEMO_TABLE_INLINE void put128( uint64_t lo, uint64_t hi ) { put64( lo ); put64( hi ); } + // EVERY LENGTH, COUNT, INDEX AND ID REFERENCE IS ONE CANONICAL UNSIGNED + // LEB128 (docs/SPEC-TABLES.md §3): seven value bits a byte, the lowest + // group first, the high bit set on every byte but the last. One value has + // one spelling, so two conforming writers agree byte for byte. + LISTDEMO_TABLE_INLINE void putleb( uint64_t v ) + { + while ( v >= 0x80 ) { put8( uint8_t( v ) | 0x80 ); v >>= 7; } + put8( uint8_t( v ) ); + } +}; + +// TableLebBytes is one value's spelling length, which a MEASURE needs before +// the bytes exist — the length of a body has to be known before it is written, +// because a length whose own width moves cannot be patched in place. +inline int64_t TableLebBytes( uint64_t v ) +{ + int64_t n = 1; + while ( v >= 0x80 ) { v >>= 7; n++; } + return n; +} + +// THE ID TABLE, WRITER SIDE (docs/SPEC-TABLES.md §3). It holds every id the +// body used, once each, in FIRST-USE order over the whole wire, and the body +// names them by position: reference k is the kth entry, counted from 1, and +// reference 0 names NO ID. +// +// Its capacity is a COMPILE-TIME fact of the unit — the distinct names its +// table closure can spell — so a save allocates nothing: the table is a local +// of Measure and of Save. The bucket chain makes ref constant time and makes +// truncate constant time too, which is what an ELIDED field needs: a field +// that turns out not to ride costs nothing in the id table either, so the walk +// interns its id, builds the payload that decides, and undoes the entry when +// nothing rides. +struct TableIds +{ + static const int32_t kCapacity = 60; + static const int32_t kBuckets = 128; + + uint64_t ids[ kCapacity ]; + int32_t chain[ kCapacity ]; + int32_t head[ kBuckets ]; + int32_t count; + bool overflow; + // THE MESSAGE FORM'S SLOTS (docs/SPEC-TABLES.md §3.3). A form 2 wire + // names ids through the CONNECTION's table, which is the unit's whole + // vocabulary in a compiler-settled order — so every reference is known at + // compile time and rides at the header as a literal beside the id. This + // flag is what selects it: false interns the id in first-use order and + // writes a trailer, true answers the slot and writes none, and the walk + // that decides is one walk. + bool vocabulary; + + TableIds() : count( 0 ), overflow( false ), vocabulary( false ) + { + for ( int32_t i = 0; i < kBuckets; i++ ) { head[i] = -1; } + } + + static LISTDEMO_TABLE_INLINE uint32_t bucket_of( uint64_t id ) + { + return uint32_t( ( id * 0x9E3779B97F4A7C15ull ) >> 57 ) & uint32_t( kBuckets - 1 ); + } + + // the reference an id takes: its message-form SLOT under the connection's + // table, or the file's own first-use entry + LISTDEMO_TABLE_INLINE uint64_t ref( uint64_t id, uint64_t slot ) + { + if ( vocabulary ) { return slot; } + return intern( id ); + } + + // the FILE form's half, appending the id on first use + uint64_t intern( uint64_t id ) + { + const uint32_t b = bucket_of( id ); + for ( int32_t i = head[b]; i >= 0; i = chain[i] ) + { + if ( ids[i] == id ) { return uint64_t( i ) + 1; } + } + if ( count >= kCapacity ) { overflow = true; return 1; } + ids[count] = id; chain[count] = head[b]; head[b] = count; count++; + return uint64_t( count ); + } + + // undo every entry appended since mark. An entry removed is the most + // recent one in its bucket, so it sits at that bucket's head. + void truncate( int32_t mark ) + { + // a SLOT costs no entry, so an elided field has nothing to undo + if ( vocabulary ) { return; } + while ( count > mark ) + { + count--; + head[ bucket_of( ids[count] ) ] = chain[count]; + } + } +}; + +// TableIdsBytes is the trailer's own size: the entries, each a fixed +// little-endian u64, and the ENTRY COUNT, the one fixed-width number on the +// wire (docs/SPEC-TABLES.md §3). +inline int64_t TableIdsBytes( const TableIds & ids ) { return int64_t( ids.count ) * 8 + 8; } + +// TableIdsWrite puts the trailer where the walk ended: a writer never patches, +// because first-use order is known only when the walk ends. +inline void TableIdsWrite( TableWriter & w, const TableIds & ids ) +{ + for ( int32_t i = 0; i < ids.count; i++ ) { w.put64( ids.ids[i] ); } + w.put64( uint64_t( ids.count ) ); +} + +// THE ID TABLE, READER SIDE (docs/SPEC-TABLES.md §3). A reader locates it from +// the END of the wire and resolves it ONCE, at open: the entries are eight +// bytes each and a body names them by position, so every field dispatches +// through an index rather than through a search over hashes. +struct TableIdTable +{ + const uint8_t * entries = NULL; + int64_t count = 0; + + // the id a reference names. ref is 1-based and bounds-checked by the + // caller: a reference ABOVE the entry count is framing damage on the body + // that carries it, and 0 names no id at all. + uint64_t at( uint64_t ref ) const + { + const uint8_t * e = entries + ( ref - 1 ) * 8; + uint64_t lo = uint64_t( e[0] ) | uint64_t( e[1] ) << 8 | uint64_t( e[2] ) << 16 | uint64_t( e[3] ) << 24; + uint64_t hi = uint64_t( e[4] ) | uint64_t( e[5] ) << 8 | uint64_t( e[6] ) << 16 | uint64_t( e[7] ) << 24; + return lo | ( hi << 32 ); + } +}; + +struct TableReader +{ + const uint8_t * buffer; + int64_t size; + int64_t offset = 0; + TableReport * report; + const TableIdTable * ids = NULL; + // ONLY THE ROOT BODY CARRIES THE NODE TABLE (docs/SPEC-TABLES.md §3.1), so + // a body has to know which it is: the reserved id inside a NESTED body is + // malformed, because a second numbering cannot exist. Every reader made + // for a payload is nested; the two the wire surfaces make for a root say so. + bool nested = true; + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report ) + : buffer( from_buffer ), size( from_size ), report( to_report ) {} + + TableReader( const uint8_t * from_buffer, int64_t from_size, TableReport * to_report, const TableIdTable * to_ids ) + : buffer( from_buffer ), size( from_size ), report( to_report ), ids( to_ids ) {} + + LISTDEMO_TABLE_INLINE bool has( int64_t bytes ) const { return offset + bytes <= size; } + // A LENGTH IS A 64-BIT NUMBER AND A BUFFER IS NOT (docs/SPEC-TABLES.md + // §3): every length, count and index on this wire has sixty-four bits of + // capability, so one past what remains must be compared UNSIGNED. Casting + // it to int64 first turns 0xFFFFFFFFFFFFFFFF into -1, and a negative + // length looks like room. + LISTDEMO_TABLE_INLINE bool room( uint64_t bytes ) const { return bytes <= (uint64_t) ( size - offset ); } + LISTDEMO_TABLE_INLINE uint8_t get8() { return buffer[offset++]; } + LISTDEMO_TABLE_INLINE uint16_t get16() { uint16_t v = uint16_t( buffer[offset] ) | uint16_t( buffer[offset+1] ) << 8; offset += 2; return v; } + LISTDEMO_TABLE_INLINE uint32_t get32() { uint32_t v = uint32_t( buffer[offset] ) | uint32_t( buffer[offset+1] ) << 8 | uint32_t( buffer[offset+2] ) << 16 | uint32_t( buffer[offset+3] ) << 24; offset += 4; return v; } + LISTDEMO_TABLE_INLINE uint64_t get64() { uint64_t lo = get32(); uint64_t hi = get32(); return lo | ( hi << 32 ); } + LISTDEMO_TABLE_INLINE void get128( uint64_t & lo, uint64_t & hi ) { lo = get64(); hi = get64(); } + + // ONE CANONICAL UNSIGNED LEB128 (docs/SPEC-TABLES.md §3), and a + // non-minimal spelling is MALFORMED: 0x80 0x00 and 0x00 both spell zero, + // and only the second is legal input. An encoding past ten bytes, or a + // tenth byte with a bit above the 64th value bit, is malformed on the same + // rule. false = framing damage on the body carrying it. + bool getleb( uint64_t & value ) + { + // A NUMBER THIS READER REFUSES LEAVES THE CURSOR WHERE IT WAS. The + // caller's next question is often "did this body end exactly at its + // L", and a rejected number that had moved the cursor would answer + // that question with the damage already stepped over. + const int64_t at = offset; + value = 0; + uint32_t shift = 0; + for ( int32_t i = 0; i < 10; i++ ) + { + if ( !has( 1 ) ) { offset = at; return false; } + const uint8_t b = get8(); + if ( i == 9 && b > 1 ) { offset = at; return false; } + value |= uint64_t( b & 0x7F ) << shift; + if ( ( b & 0x80 ) == 0 ) + { + if ( i > 0 && b == 0 ) { offset = at; return false; } // a redundant continuation + return true; + } + shift += 7; + } + offset = at; + return false; + } + + // resolve one id reference against the file's table. false = a reference + // ABOVE the entry count, or a 0 where an id is required, both of which + // are framing damage on the body that carries it. + bool getid( uint64_t & id ) + { + uint64_t ref = 0; + if ( !getleb( ref ) ) { return false; } + if ( ref == 0 || ids == NULL || ref > (uint64_t) ids->count ) { return false; } + id = ids->at( ref ); + return true; + } + + // skip one payload by kind; false = framing damage. FOUR RULES COVER THE + // SET (docs/SPEC-TABLES.md §3), and a kind outside it is not skippable — + // which is why the set is closed and why kind 31 exists. + bool skip( uint8_t kind ) + { + switch ( kind ) + { + // the fixed-width kinds, each by its width: 18-29 are the 128-bit integers and + // the fixed-point family at every storage width (docs/SPEC-TABLES.md §3) + case 1: case 2: case 6: case 20: case 25: return has( 1 ) ? ( offset += 1, true ) : false; + case 3: case 7: case 21: case 26: return has( 2 ) ? ( offset += 2, true ) : false; + case 4: case 8: case 10: case 22: case 27: return has( 4 ) ? ( offset += 4, true ) : false; + case 5: case 9: case 11: case 23: case 28: return has( 8 ) ? ( offset += 8, true ) : false; + case 18: case 19: case 24: case 29: return has( 16 ) ? ( offset += 16, true ) : false; + case 17: case 30: // a NODE INDEX (§3.1) and an ENUM's variant reference: one LEB128 and stop + { + uint64_t ignored = 0; + return getleb( ignored ); + } + case 12: case 13: case 14: case 16: case 31: case 32: // 31 is the ESCAPE, 32 the payload-free kind + { + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + case 15: // union: the arm id reference, then its kind, its L and its payload (reference 0 = empty) + { + uint64_t arm = 0; + if ( !getleb( arm ) ) return false; + if ( arm == 0 ) return true; + if ( !has( 1 ) ) return false; + offset += 1; // the arm's kind byte + uint64_t n = 0; + if ( !getleb( n ) ) return false; + return room( n ) ? ( offset += (int64_t) n, true ) : false; + } + } + return false; + } +}; + +// The RESERVED node-table id, the one id the language holds back +// (docs/SPEC-TABLES.md §3.1, §5). It rides in every unit, pointered or not, +// because every body has to know that a NESTED body claiming one is damaged. +static const uint64_t kTableNodeTableFieldId = 0xFFFFFFFFFFFFFFFFull; + +// TableWireForm is the FORM BYTE, and it is the whole header +// (docs/SPEC-TABLES.md §3). A reader that meets a byte it does not know +// refuses the wire by name and never reports damage. +const uint8_t kTableWireForm = 1; + +// TableOpen reads the form byte and the trailer, in that order, and hands back +// the ROOT BODY. It answers one of three verdicts, because five zero counters +// and a false flag are what a clean read prints too: +// +// TableOpenOk the form is known and the table read whole +// TableOpenRefused a FORM BYTE this reader does not carry: nothing is +// decoded, nothing is counted, and no damage is reported +// TableOpenDamaged a table that cannot be read whole — fewer than eight +// bytes, a count whose entries run past the front of the +// file, a count that leaves no room for the form byte, or +// ONE ID IN TWO ENTRIES. The whole wire is malformed, +// nothing is decoded, and one event is counted. +// TableOpenBodyStopped the form and the table were good and the ROOT BODY +// could not be walked to its own terminator. What it +// decoded before that is kept, as everywhere on this wire. +enum TableOpenVerdict { TableOpenOk, TableOpenRefused, TableOpenDamaged, TableOpenBodyStopped }; + +inline TableOpenVerdict TableOpen( const uint8_t * buffer, int64_t bytes, TableIdTable & table, int64_t & body_bytes ) +{ + if ( bytes < 1 ) { return TableOpenDamaged; } + if ( buffer[0] != kTableWireForm ) { return TableOpenRefused; } + if ( bytes < 9 ) { return TableOpenDamaged; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + uint64_t count = lo | ( hi << 32 ); + if ( count > (uint64_t) ( bytes / 8 ) ) { return TableOpenDamaged; } + const int64_t span = (int64_t) count * 8 + 8; + if ( span + 1 > bytes ) { return TableOpenDamaged; } + table.entries = buffer + bytes - span; + table.count = (int64_t) count; + // THE ENTRIES ARE DISTINCT: a table that carries one id twice is malformed + // for the whole wire, because no wire this schema writes carries a repeat + // and it would leave one more shape of table for a hostile writer to aim + // at (docs/SPEC-TABLES.md §3). + for ( int64_t i = 1; i < table.count; i++ ) + { + const uint64_t id = table.at( uint64_t( i ) + 1 ); + for ( int64_t j = 0; j < i; j++ ) + { + if ( table.at( uint64_t( j ) + 1 ) == id ) { return TableOpenDamaged; } + } + } + body_bytes = bytes - span - 1; + return TableOpenOk; +} + +// TableBodyExtent walks a body's framing to the zero reference that ends it, +// so a reader can tell a body that ENDED EARLY — leaving bytes no field claims +// — from one that is merely damaged. ANY BYTE BETWEEN THE ROOT'S TERMINATOR +// AND THE TABLE'S FIRST ENTRY IS MALFORMED, because no field claims it and the +// two ends of the file have met (docs/SPEC-TABLES.md §3). +inline bool TableBodyEndsEarly( const uint8_t * body, int64_t bytes, const TableIdTable & table ) +{ + TableReport ignored; + TableReader r( body, bytes, &ignored, &table ); + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { return false; } + if ( ref == 0 ) { return r.offset != bytes; } + if ( ref > (uint64_t) table.count ) { return false; } + if ( !r.has( 1 ) ) { return false; } + if ( !r.skip( r.get8() ) ) { return false; } + } +} + +// THE MESSAGE FORM (docs/SPEC-TABLES.md §3.3): a FILE carries its own id +// table and a MESSAGE STREAM announces one and then carries none. +// +// A form 2 wire is TWO PARTS, the form byte and the root body: the body ends +// at its own zero reference as it does in a file, there is no trailer, and the +// message's last byte is the body's terminator. Its references resolve against +// the CONNECTION's table, which is the unit's whole vocabulary in the order +// the compiler settled. +const uint8_t kTableWireMessageForm = 2; + +// The RESERVED build-version id, the second id the language holds back (§5, +// §11), beside the node table's. It is the announcement's one required field, +// and a reserved id in any body but the one whose transport it is, is +// malformed (§3.1). +static const uint64_t kTableBuildVersionFieldId = 0xFFFFFFFFFFFFFFFEull; + +// The reserved NODE-TABLE id's own slot in this unit's vocabulary (§3.3). A +// pointered message names the node table through it, exactly as every other +// field header names its id through a slot. +static const uint64_t kTableNodeTableFieldSlot = 38; + +// THE UNIT'S ANNOUNCEMENT, byte for byte: 61 entries and 508 bytes. It is an +// ordinary form 1 FILE — the form byte, a body carrying the BUILD VERSION +// under the reserved id at kind 9, and the trailer that IS the connection's +// table, slot 1 the reserved id and slots 2 and up the vocabulary under one +// numbering. +// +// The vocabulary is the unit's whole closure in the COOK PROJECTION's order +// (§20.2) — each record in the order the projection renders it and each +// record's fields in the order the projection renders them, then each enum's +// variants and each union's arms — followed by the tail the projection does +// not name: the reserved node-table id, the three blob type ids as bytes, +// string and wstring, and every table's own name id in the projection's sorted +// record order. The tail is UNCONDITIONAL, so an ordinary edit only ever grows +// it at its end and never moves a slot a generated field header carries as a +// literal. +static const int64_t kTableAnnounceBytes = 508; +static const uint8_t kTableAnnounce[ kTableAnnounceBytes ] = { + 0x01, 0x01, 0x09, 0xc7, 0x71, 0x45, 0xca, 0xda, 0x0e, 0x7c, 0x8d, 0x00, + 0xfe, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x30, 0xb1, 0x3a, 0xff, + 0x4a, 0xd9, 0xb1, 0x40, 0x20, 0xea, 0x4d, 0x40, 0x8e, 0xa7, 0x19, 0xaa, + 0x26, 0xa9, 0x02, 0x0c, 0x9b, 0x01, 0x48, 0x78, 0xe9, 0xea, 0x71, 0x6f, + 0x0f, 0x01, 0x82, 0xbf, 0x6f, 0x2c, 0x41, 0x4f, 0xbf, 0x84, 0x78, 0x3e, + 0xf3, 0xa4, 0x48, 0x44, 0x19, 0xab, 0xd7, 0x56, 0x05, 0x4a, 0xa3, 0x30, + 0x67, 0x55, 0x5b, 0x85, 0xc9, 0xe2, 0x4e, 0x30, 0x69, 0x6a, 0xb4, 0x81, + 0xfb, 0x67, 0x4d, 0x1a, 0xcf, 0x7b, 0x27, 0x21, 0x74, 0xa2, 0x79, 0x44, + 0x8e, 0xe2, 0xe5, 0xb1, 0x84, 0x76, 0xbc, 0x2e, 0xef, 0x83, 0x76, 0x1e, + 0xc5, 0x99, 0xf7, 0x82, 0x76, 0x4e, 0x0a, 0xd9, 0xa8, 0x2e, 0x86, 0x70, + 0x84, 0xed, 0xf2, 0x4a, 0xbb, 0xf0, 0x0c, 0x9b, 0xcc, 0xfb, 0x2d, 0x73, + 0x68, 0xb7, 0xf0, 0xae, 0x4c, 0x0c, 0xf6, 0x52, 0xbf, 0xe9, 0xd1, 0x2f, + 0x93, 0xcd, 0xda, 0xdb, 0x22, 0x72, 0x34, 0x7d, 0xf6, 0x0b, 0x72, 0x17, + 0x07, 0x17, 0x02, 0x86, 0x4c, 0xf5, 0x63, 0xaf, 0x54, 0x15, 0x02, 0x86, + 0x4c, 0xf4, 0x63, 0xaf, 0x3a, 0x70, 0x6e, 0x3e, 0x93, 0x43, 0xe5, 0x9d, + 0x3d, 0x62, 0xcb, 0x8f, 0xec, 0xfc, 0xf7, 0x39, 0x09, 0x06, 0x02, 0x86, + 0x4c, 0xeb, 0x63, 0xaf, 0x09, 0x4b, 0x4d, 0x57, 0xaa, 0x33, 0x47, 0xd2, + 0x31, 0x54, 0xaf, 0x1d, 0x19, 0x73, 0x50, 0x12, 0xb2, 0x0f, 0x40, 0x27, + 0x0b, 0x6b, 0x98, 0x01, 0x38, 0x81, 0x0a, 0xf1, 0x1f, 0x06, 0xa7, 0xa3, + 0x0f, 0x62, 0xad, 0x07, 0x77, 0x47, 0x82, 0x5f, 0x42, 0x4f, 0x4f, 0x30, + 0x0d, 0x39, 0x84, 0x1c, 0x86, 0x1b, 0x63, 0x8e, 0xba, 0xad, 0xbc, 0xc4, + 0xec, 0x10, 0x5b, 0x36, 0x19, 0x4a, 0xc9, 0x3d, 0xea, 0x0c, 0xe8, 0x30, + 0x94, 0xfd, 0xe4, 0x7c, 0xec, 0x22, 0x02, 0x86, 0x4c, 0xfc, 0x63, 0xaf, + 0x05, 0x28, 0x02, 0x86, 0x4c, 0xff, 0x63, 0xaf, 0x52, 0x26, 0x02, 0x86, + 0x4c, 0xfe, 0x63, 0xaf, 0xb1, 0x45, 0xc3, 0x44, 0x35, 0xab, 0xfe, 0x73, + 0xc0, 0x7f, 0xb3, 0x8a, 0xbe, 0x08, 0x63, 0x7f, 0xff, 0xff, 0xff, 0xff, + 0xff, 0xff, 0xff, 0xff, 0xe4, 0x4f, 0x1c, 0x4f, 0x47, 0xc0, 0x2e, 0x2f, + 0x58, 0xfc, 0xaf, 0xfa, 0xd8, 0xe0, 0x4b, 0x70, 0xc7, 0xd4, 0x7b, 0x26, + 0xb0, 0x9d, 0x29, 0x5f, 0xcc, 0x14, 0x15, 0x7f, 0xcb, 0xc2, 0x58, 0xd8, + 0x84, 0x9e, 0x3a, 0x55, 0x8b, 0x37, 0xe2, 0x06, 0x2a, 0x89, 0xf5, 0x12, + 0x09, 0xc1, 0xca, 0x0a, 0x44, 0xa2, 0x31, 0xc1, 0xad, 0xa7, 0xee, 0xee, + 0xe8, 0xcf, 0xbf, 0x43, 0x73, 0x18, 0x43, 0xd0, 0x42, 0xad, 0xf6, 0xf8, + 0x59, 0x86, 0x63, 0x91, 0xb7, 0xce, 0x00, 0x7c, 0xd1, 0xc5, 0x34, 0x20, + 0x06, 0x68, 0x47, 0x98, 0xd1, 0xa1, 0xcf, 0x52, 0x5f, 0x82, 0x58, 0xac, + 0x36, 0x15, 0x78, 0x5e, 0xb8, 0x8b, 0x59, 0x6f, 0xc9, 0xc6, 0x86, 0xbb, + 0xc3, 0x64, 0x89, 0x50, 0xd2, 0x8d, 0xa7, 0xf1, 0x80, 0xea, 0x3a, 0xb9, + 0xf1, 0x21, 0xf7, 0x41, 0x11, 0xed, 0xd9, 0xce, 0x96, 0x92, 0x43, 0x8a, + 0xfb, 0x06, 0xc9, 0xfe, 0x19, 0xe1, 0x13, 0xa0, 0xa7, 0x0a, 0xc7, 0x54, + 0x12, 0xd6, 0x40, 0xdc, 0x08, 0xf0, 0xf5, 0xc0, 0x24, 0x5f, 0xf8, 0x33, + 0xc8, 0xfb, 0x85, 0x9a, 0xaf, 0xe0, 0xc9, 0x0c, 0x91, 0x0a, 0x55, 0x60, + 0xf7, 0xa2, 0x07, 0xec, 0x8b, 0x6d, 0x02, 0x86, 0x43, 0xf3, 0xc2, 0x2e, + 0x87, 0x27, 0xcc, 0x86, 0xf0, 0xe0, 0x26, 0x8f, 0x3d, 0x00, 0x00, 0x00, + 0x00, 0x00, 0x00, 0x00, +}; + +// AnnounceMeasure is the announcement's byte count, which is a constant of the +// unit and not a walk. +inline int64_t AnnounceMeasure() { return kTableAnnounceBytes; } + +// Announce writes the announcement into the caller's buffer and answers the +// bytes written — exactly AnnounceMeasure's answer — or -1 when the buffer is +// too small. It allocates nothing and walks nothing. +inline int64_t Announce( uint8_t * buffer, int64_t capacity ) +{ + if ( buffer == NULL || capacity < kTableAnnounceBytes ) { return -1; } + memcpy( buffer, kTableAnnounce, (size_t) kTableAnnounceBytes ); + return kTableAnnounceBytes; +} + +// TableVocabulary is ONE DIRECTION of ONE CONNECTION's id table (§3.3): the +// entries an announcement carried, whole, under one numbering with slot 1 the +// reserved build-version id. +// +// A peer holds TWO of these for a connection, the one it writes with and the +// one it reads with, and neither is the other's. A restart opens a fresh +// connection with empty tables and nothing is cached across connections, so +// its whole life is one connection's. It BORROWS the announcement's bytes rather than +// copying them, so a receiver holds one table a direction and its memory is +// the bound below and nothing else. +struct TableVocabulary +{ + // THE CONFORMING DEFAULT BOUND (§3.3): 32 KiB a direction, eight times the + // 500-id unit that is already a large one. A connection's table is bounded + // by nothing the wire carries, so the receiver declares the maximum and an + // announcement above it is refused by name before an entry is touched. + static const int64_t kDefaultMaxEntries = 4096; + + TableIdTable table; + uint64_t build_version = 0; + bool announced = false; + int64_t max_entries = kDefaultMaxEntries; +}; + +// AnnounceRead reads an announcement into one direction's table (§3.3). +// +// THE BOUND IS CHECKED BEFORE ANYTHING IS ALLOCATED: the entry count is a +// fixed little-endian u64 at the end, so a receiver reads it, compares it and +// refuses without touching an entry. After that it is §3's ordinary FILE read, +// because the announcement IS a file, with EXACTLY ONE STRICT CHECK over its +// body: the reserved build-version field present, exactly once, under kind 9, +// eight bytes wide. Everything else is an ordinary field under §4's tolerance, +// so an unknown one is skipped and counted and the announcement can GAIN a +// field in a later minor without a lockstep redeploy. +// +// The FIRST announcement sets the table and it is the only one that can. A +// SECOND is refused by name: it does not replace the table, it does not amend +// it and it changes nothing. A refused announcement sets NO TABLE. +inline bool AnnounceRead( TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + if ( vocabulary.announced ) + { + to->refused = true; + to->reason = second_announcement; + return false; + } + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireForm ) + { + to->refused = true; + to->reason = buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + return false; + } + if ( bytes < 9 ) { to->malformed = true; return false; } + const uint8_t * tail = buffer + bytes - 8; + uint64_t lo = uint64_t( tail[0] ) | uint64_t( tail[1] ) << 8 | uint64_t( tail[2] ) << 16 | uint64_t( tail[3] ) << 24; + uint64_t hi = uint64_t( tail[4] ) | uint64_t( tail[5] ) << 8 | uint64_t( tail[6] ) << 16 | uint64_t( tail[7] ) << 24; + if ( ( lo | ( hi << 32 ) ) > (uint64_t) vocabulary.max_entries ) + { + to->refused = true; + to->reason = vocabulary_too_large; + return false; + } + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else { to->refused = true; to->reason = newer_form; } + return false; + } + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) { to->malformed = true; return false; } + // the body, under §4's tolerance and this form's one strict check + TableReader r( buffer + 1, body_bytes, to, &table ); + uint64_t version = 0; + int32_t seen = 0; + for ( ;; ) + { + uint64_t ref = 0; + if ( !r.getleb( ref ) ) { to->malformed = true; return false; } + if ( ref == 0 ) { break; } + if ( ref > (uint64_t) table.count || !r.has( 1 ) ) { to->malformed = true; return false; } + const uint64_t id = table.at( ref ); + const uint8_t kind = r.get8(); + if ( id != kTableBuildVersionFieldId ) + { + to->unknown++; + if ( !r.skip( kind ) ) { to->malformed = true; return false; } + continue; + } + if ( kind != 9 || !r.has( 8 ) ) { to->refused = true; to->reason = no_vocabulary; return false; } + version = r.get64(); + seen++; + } + if ( seen != 1 ) { to->refused = true; to->reason = no_vocabulary; return false; } + vocabulary.table = table; + vocabulary.build_version = version; + vocabulary.announced = true; + return true; +} + +inline float table_bits_to_float( uint32_t bits ) { float f; memcpy( &f, &bits, 4 ); return f; } +inline uint32_t table_float_to_bits( float f ) { uint32_t b; memcpy( &b, &f, 4 ); return b; } +inline double table_bits_to_double( uint64_t bits ) { double d; memcpy( &d, &bits, 8 ); return d; } +inline uint64_t table_double_to_bits( double d ) { uint64_t b; memcpy( &b, &d, 8 ); return b; } + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_PRIMITIVES + +#ifndef LISTDEMO_SCHEMA_TABLE_ARENA +#define LISTDEMO_SCHEMA_TABLE_ARENA + +namespace listdemo { + +// ---- variable-length tables: tuning constants (docs/SPEC-TABLES.md) ---- +// +// The segment size and the count multiply to exactly 2^32: the u32 reference +// is the arena's hard ceiling, and these constants saturate it rather than +// leaving address space unreachable. Slab handout costs one atomic per slab, +// so per-node allocation costs no synchronization at all. + +static const uint32_t kTableSegmentBits = 22; // 4 MiB segments +static const uint32_t kTableSegmentSize = 1u << kTableSegmentBits; +static const uint32_t kTableSegmentMask = kTableSegmentSize - 1u; +static const uint32_t kTableMaxSegments = 1u << ( 32 - kTableSegmentBits ); // 1024 -> 4 GiB +static const uint32_t kTableSlabBytes = 64u * 1024u; // one atomic per slab +static const uint32_t kTableAlign = 8; // every node starts 8-aligned +static const uint32_t kTableAllocFailed = 0xFFFFFFFFu; + +// ---- THE CALLER'S ALLOCATOR (docs/SPEC-TABLES.md §6.5) ---- +// +// Every allocation the variable-length runtime makes goes through one of +// these — the arena's segments, the pack walk's identity map, the numbering's +// entry array, the packed region, and the tool path's node directory. There is +// no other call to the C library on this path, so a counting allocator sees +// every byte and a game's own heap can own all of it. +// +// It is the shape TableBlockAllocator already has (§19.1): two function +// pointers and a context the caller carries. What it adds is a CONTRACT ON +// alloc — the bytes come back ZEROED. Lock copies whole nodes, PADDING +// INCLUDED, so anything left uninitialized reaches a packed region; the default +// pair reaches that through calloc, which costs nothing measurable because a +// fresh segment is untouched pages either way. +struct TableAllocator +{ + void * ( *alloc )( void * context, int64_t bytes ); // ZEROED bytes, NULL on failure + void ( *free )( void * context, void * pointer ); + void * context; +}; + +// The default pair, and it is the one every entry point takes when the caller +// names none. It calls schema_allocate / schema_release, so a program with its +// own C-library replacement can move the floor without writing a struct at all. +inline void * table_default_alloc( void * context, int64_t bytes ) { (void) context; return schema_allocate( bytes ); } +inline void table_default_free( void * context, void * pointer ) { (void) context; schema_release( pointer ); } + +inline TableAllocator TableDefaultAllocator() +{ + TableAllocator allocator; + allocator.alloc = table_default_alloc; + allocator.free = table_default_free; + allocator.context = NULL; + return allocator; +} + +// ---- TableRef: a relocatable reference (never a machine pointer) ---- +// +// Two encodings, one slot, and the FORM says which is in force: +// +// in the arena — the node's arena offset (segment index in the high bits) +// in a region — the SELF-RELATIVE byte delta from this slot's own address, +// so a deref is one add, needs no base pointer, and a whole +// region relocates by memcpy with zero fix-up +// +// 0 is null in both, and a slot can never name the node that contains it, so +// zero names nothing real in either form. +// +// A REGION DELTA HAS NO REQUIRED SIGN (§6.3). A region is packed depth-first, +// so a node's FIRST reference points forward; every LATER reference to that +// same node points BACK at the one body it already has, which is exactly what +// makes one node one node in a region. Sharing and a back-reference are the +// same fact, and nothing validates a reference by its sign. +// +// IT IS EIGHT BYTES, SIGNED, so ONE REGION REACHES EVERYTHING (§6.3, §7): a +// four-byte slot bounded a region at 2 GiB, and the scale a cook exists for is +// *"100mbs or many gigabytes of data in Assets.bin"*. +struct TableRef +{ + int64_t value = 0; + bool null() const { return value == 0; } +}; + +// TableSlot is what Alloc hands back: usable as the node pointer (write +// fields through it) AND as the reference to store in a pointer field. +template struct TableSlot +{ + T * ptr = NULL; + TableRef ref; + T * operator->() const { return ptr; } + T & operator*() const { return *ptr; } + operator T *() const { return ptr; } + operator TableRef() const { return ref; } + bool null() const { return ptr == NULL; } +}; + +inline uint32_t TableAlignUp( uint32_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( kTableAlign - 1 ); } +inline int64_t TableAlignUp64( int64_t bytes ) { return ( bytes + kTableAlign - 1 ) & ~( int64_t( kTableAlign ) - 1 ); } + +// ---- a BYTE BUFFER's node (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// A *bytes or *string slot is a TableRef like every pointer slot, and it names +// a BLOB NODE: this eight-byte header and then the bytes, at offset eight so +// the data is eight-aligned. A *string blob carries one more zero byte after +// its data, so a region hands back a C string with no copy. The node's extent +// is the header plus its bytes, rounded to the arena's alignment like every +// node's; on the wire it is a record whose body is the bytes (§3.1). +struct TableBlob +{ + uint32_t length; + uint32_t zero; +}; + +static const int64_t kTableBlobHeader = 8; // length (u32), then four zero bytes +static const int64_t kTableBlobMaxLength = 0xFFFFFFFF; // a record's length is a u32 (§3.1) + +// the node's storage: the header, the bytes, a string's terminator, rounded +// to the arena's alignment like every node +inline int64_t TableBlobStorage( int64_t length, bool terminated ) +{ + return TableAlignUp64( kTableBlobHeader + length + ( terminated ? 1 : 0 ) ); +} + +// What a read answers: a pointer INTO the region and the length, NULL and +// zero for a null slot. Off a locked region, a loaded one or an opened cook +// the pointer is one add from the slot, and nothing is copied. +struct TableBytesView +{ + const uint8_t * data; + int64_t length; +}; + +struct TableStringView +{ + const char * data; // zero-terminated + int64_t length; +}; + +// What AllocBytes and AllocString hand back: the bytes to write through, the +// length asked for, and the reference to store in the slot — the three +// answers TableSlot gives for a table node. +struct TableBytesSlot +{ + uint8_t * data = NULL; + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +struct TableStringSlot +{ + char * data = NULL; // room for length bytes and the terminator, already zero + int64_t length = 0; + TableRef ref; + bool null() const { return data == NULL; } + operator TableRef() const { return ref; } +}; + +// ---- the arena: segmented, slab-handed, lock-free by ownership ---- +// +// Allocation is thread-local inside a worker's slab — no atomics on the node +// path. A worker takes its next slab with ONE compare-exchange, and a new +// segment is published with one more. Nothing ever moves: a segment, once +// allocated, lives untouched until the arena is torn down, so a T* obtained +// from Alloc stays valid while other workers allocate, and an offset stays +// correct while the arena grows. +// +// The model this DELIBERATELY refuses: one buffer under a lock, grown by +// realloc. A realloc moves the buffer under workers mid-write; offsets fix +// identity but not the raw references already resolved from them, and the +// resulting corruption is invisible until much later. Segments never move, so +// that bug class cannot be written here. +// +// Slack: at most one slab tail per worker plus one slab per segment (a slab +// that will not fit is skipped rather than split), i.e. under 2% of a segment +// plus threads x 64 KiB. That is the price of never synchronizing per node. +struct TableArena +{ + std::atomic segments[ kTableMaxSegments ]; + std::atomic cursor; // (segment << kTableSegmentBits) | bytes handed out + bool locked = false; // MONOTONIC: Lock() is one-way, there is no unlock + // THE ARENA CARRIES ITS OWN, so everything downstream of a builder — + // segments, pack map, numbering, region, node directory — allocates through + // the one pair the caller named, with nothing to thread by hand. + TableAllocator allocator; +}; + +inline void TableArenaInit( TableArena & arena, TableAllocator allocator ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + arena.segments[i].store( NULL, std::memory_order_relaxed ); + } + arena.cursor.store( 0, std::memory_order_relaxed ); + arena.locked = false; + arena.allocator = allocator; +} + +inline void TableArenaShutdown( TableArena & arena ) +{ + for ( uint32_t i = 0; i < kTableMaxSegments; i++ ) + { + uint8_t * segment = arena.segments[i].exchange( NULL, std::memory_order_acq_rel ); + if ( segment != NULL ) { arena.allocator.free( arena.allocator.context, segment ); } + } + arena.cursor.store( 0, std::memory_order_relaxed ); +} + +// one L1 load plus an add: the segment table is 8 KiB and stays hot +inline uint8_t * TableArenaAt( const TableArena & arena, uint32_t offset ) +{ + return arena.segments[ offset >> kTableSegmentBits ].load( std::memory_order_relaxed ) + ( offset & kTableSegmentMask ); +} + +// TableArenaGrabSlab hands one worker its next private slab. Returns +// kTableAllocFailed when the arena's address space or the allocator is +// exhausted — a loud refusal, never a silent smaller slab. +inline uint32_t TableArenaGrabSlab( TableArena & arena ) +{ + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t segment = cursor >> kTableSegmentBits; + uint32_t used = cursor & kTableSegmentMask; + // strictly less: a slab is never split across segments, and the tail + // is the documented slack + if ( used + kTableSlabBytes < kTableSegmentSize ) + { + if ( arena.segments[segment].load( std::memory_order_acquire ) == NULL ) + { + // THE SEGMENT COMES BACK ZEROED, which is the allocator's + // contract and not an extra pass here: Lock copies whole nodes, + // PADDING INCLUDED, so anything uninitialized reaches a packed + // region. Value-initializing a node with placement new zeroes + // its MEMBERS and not its padding, so the zeroing has to happen + // at the segment or not at all. It costs nothing measurable: a + // fresh segment is untouched pages either way, and the default + // pair's calloc has the kernel hand them over zeroed. + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, (int64_t) kTableSegmentSize ); + if ( memory == NULL ) { return kTableAllocFailed; } + uint8_t * expected = NULL; + if ( !arena.segments[segment].compare_exchange_strong( expected, memory, std::memory_order_acq_rel ) ) + { + // another worker published this segment first + arena.allocator.free( arena.allocator.context, memory ); + } + } + if ( arena.cursor.compare_exchange_weak( cursor, cursor + kTableSlabBytes, std::memory_order_acq_rel ) ) + { + return ( segment << kTableSegmentBits ) | used; + } + continue; + } + uint32_t next_segment = segment + 1; + if ( next_segment >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + arena.cursor.compare_exchange_weak( cursor, next_segment << kTableSegmentBits, std::memory_order_acq_rel ); + } +} + +// TableArenaGrabSpan reserves a SPAN of the arena's address space for one node +// larger than a slab — a BYTE BUFFER of any size (docs/SPEC-TABLES.md §2.5) — +// and allocates it as one contiguous block. It takes whole segment indices +// from the cursor, starting at the index after the cursor's so nothing else +// is ever handed out inside the span, and publishes the block under the first +// of them; the indices the span covers past that one stay NULL, which is +// enough, because only a node's START is ever resolved through the segment +// table and a blob's bytes follow its header inside the one allocation. The +// unused tail of the segment the cursor was in is slack, like a slab tail. +// Returns kTableAllocFailed when the address space or the allocator is +// exhausted — a loud refusal, never a smaller blob. +inline uint32_t TableArenaGrabSpan( TableArena & arena, int64_t bytes ) +{ + if ( bytes <= 0 || bytes > ( (int64_t) kTableMaxSegments - 2 ) * (int64_t) kTableSegmentSize ) { return kTableAllocFailed; } + const uint32_t spanned = (uint32_t) ( ( bytes + kTableSegmentSize - 1 ) >> kTableSegmentBits ); + for ( ;; ) + { + uint32_t cursor = arena.cursor.load( std::memory_order_acquire ); + uint32_t start = ( cursor >> kTableSegmentBits ) + 1; + if ( start + spanned >= kTableMaxSegments ) { return kTableAllocFailed; } // 4 GiB: the u32 reference's ceiling + uint32_t next = ( start + spanned ) << kTableSegmentBits; + if ( !arena.cursor.compare_exchange_weak( cursor, next, std::memory_order_acq_rel ) ) { continue; } + // the span is this worker's now: nothing else can publish under its + // first index, so a plain store suffices, and the block comes back + // ZEROED like every segment — the blob's bytes and its tail are zeros + // until written + uint8_t * memory = (uint8_t *) arena.allocator.alloc( arena.allocator.context, bytes ); + if ( memory == NULL ) { return kTableAllocFailed; } + arena.segments[start].store( memory, std::memory_order_release ); + return start << kTableSegmentBits; + } +} + +// ---- TableWorker: one thread's allocation front ---- +// +// The threading contract, stated plainly: +// * Alloc on YOUR OWN worker is safe concurrently with any other worker's. +// No locks, no atomics per node. +// * Writing fields of a node ANOTHER worker allocated is your own +// synchronization problem — this runtime does not arbitrate it. +// * Lock and Save are single-threaded: call them after the workers have +// joined. +struct TableWorker +{ + TableArena * arena = NULL; + uint32_t next = 0; + uint32_t end = 0; + + template TableSlot Alloc() + { + static_assert( alignof( T ) <= kTableAlign, "a table node's alignment must fit the arena's" ); + TableSlot slot; + if ( arena == NULL || arena->locked ) { return slot; } + uint32_t bytes = TableAlignUp( (uint32_t) sizeof( T ) ); + if ( bytes > kTableSlabBytes ) { return slot; } // a node larger than a slab: refused, never split + if ( end == 0 || next + bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return slot; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + uint32_t at = next; + next += bytes; + // A NODE IS BORN IN TWO HALVES: start its lifetime in the raw + // storage, then write the declared defaults ONE MEMBER AT A TIME. + // + // It is "T", not "T{}". Value-initialising the whole aggregate says + // the same thing and costs cl O(BYTES) TO COMPILE — it expands element + // by element in its front end — while both halves here cost + // O(declarations). The slab cap below refuses a large node at RUN + // TIME and bounds nothing at compile time: the cost is paid by + // whatever T a caller instantiates this with. + // Padding is not the difference: value-initialisation zeroes MEMBERS + // and not padding either way, which is why the segment is calloc'd. + // + // TableReset is an OVERLOAD SET, one per closure member, reached from + // this template by argument-dependent lookup on T's own namespace — + // Alloc is a template and cannot spell Reset. + // + // The reset is here because ONE DEFINITION SAYS WHAT THE DECLARED + // DEFAULTS ARE, and it is Reset. Default-initialisation lands on + // the same values today, because a member with a non-zero default + // carries a member initializer that says so — but that is the class + // definition agreeing with Reset, not the arena reading it, and #320's + // fix was itself a pass that MOVED initialisation between the two. + // The arena reads the definition. + slot.ptr = new ( TableArenaAt( *arena, at ) ) T; + TableReset( *slot.ptr ); + slot.ref.value = at; + return slot; + } + + // Alloc a BYTE BUFFER's node of exactly length bytes (docs/SPEC-TABLES.md + // §2.5): the blob header and its bytes, zeroed, in this thread's slab when + // it fits and in a span of the arena's own when it does not. NULL is the + // arena locked, a length below zero or past a record's u32, or the + // allocator refusing. The offset comes back for the reference. + TableBlob * AllocBlob( int64_t length, bool terminated, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( length < 0 || length > kTableBlobMaxLength ) { return NULL; } + const int64_t bytes = TableBlobStorage( length, terminated ); + if ( bytes > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, bytes ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + } + else + { + if ( end == 0 || next + (uint32_t) bytes > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) bytes; + } + TableBlob * blob = (TableBlob *) TableArenaAt( *arena, at ); + blob->length = (uint32_t) length; // the bytes after it are the segment's zeros + blob->zero = 0; + return blob; + } + + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries + // no type id, takes no index and has no Reset, so it goes through the same + // slab and span the blob path uses rather than through Alloc. + uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) + { + at = 0; + if ( arena == NULL || arena->locked ) { return NULL; } + if ( bytes <= 0 || align > (int64_t) kTableAlign ) { return NULL; } + const int64_t rounded = TableAlignUp64( bytes ); + if ( rounded > (int64_t) kTableSlabBytes ) + { + at = TableArenaGrabSpan( *arena, rounded ); + if ( at == kTableAllocFailed ) { at = 0; return NULL; } + return TableArenaAt( *arena, at ); + } + if ( end == 0 || next + (uint32_t) rounded > end ) + { + uint32_t offset = TableArenaGrabSlab( *arena ); + if ( offset == kTableAllocFailed ) { return NULL; } + next = offset; + end = offset + kTableSlabBytes; + if ( next == 0 ) { next = kTableAlign; } // offset 0 is null: the arena's head stays reserved + } + at = next; + next += (uint32_t) rounded; + return TableArenaAt( *arena, at ); // the segment came back zeroed + } + // a *bytes node: the bytes to write through, and the reference to store + TableBytesSlot AllocBytes( int64_t length ) + { + TableBytesSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, false, at ); + if ( blob == NULL ) { return slot; } + slot.data = (uint8_t *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } + + // a *string node: room for length bytes and the zero byte after them + TableStringSlot AllocString( int64_t length ) + { + TableStringSlot slot; + uint32_t at = 0; + TableBlob * blob = AllocBlob( length, true, at ); + if ( blob == NULL ) { return slot; } + slot.data = (char *) ( blob + 1 ); + slot.length = length; + slot.ref.value = at; + return slot; + } +}; + +// ---- TablePackMap: the pack walk's identity map (docs/SPEC-TABLES.md §3.1, §6.2) ---- +// +// ONE ENTRY PER REACHABLE NODE, and that map IS identity: a node must know +// where it landed to be named a second time, so Lock packs a shared node ONCE +// and every later reference resolves to the one body it already has. That is +// the same first-visit numbering the wire uses, so the pack order and the node +// order are one order. +// +// COLOURING AN ENTRY WHILE ITS DESCENT IS OPEN COSTS ONE BIT, and it is what +// makes a data cycle free to refuse: a reference to an entry still open is a +// cycle, and Lock returns failure rather than recursing away. The ROOT's entry +// is open for the whole walk. +// +// The map is proportional to NODES, never to bytes, and it lives on the +// AUTHORING side, where §6.5 licenses allocation. Nothing on the reading path +// ever builds one. +struct TablePackEntry +{ + const void * key; // the node's address in the graph being packed + int64_t offset; // where that node landed in the region + uint8_t open; // its descent is still open: a reference here is a cycle +}; + +struct TablePackMap +{ + TablePackEntry * entries = NULL; + int64_t capacity = 0; // a power of two, or zero while empty + int64_t count = 0; + TableAllocator allocator; // the caller's, carried from the walk that built it +}; + +inline void TablePackMapInit( TablePackMap & map, TableAllocator allocator ) +{ + map.entries = NULL; + map.capacity = 0; + map.count = 0; + map.allocator = allocator; +} + +inline void TablePackMapShutdown( TablePackMap & map ) +{ + map.allocator.free( map.allocator.context, map.entries ); + TablePackMapInit( map, map.allocator ); +} + +// The two walks behind Lock re-derive the SAME map from the same graph — the +// numbering is never carried between them (§3.1) — so the second starts from +// an empty map and keeps the capacity the first paid for. +inline void TablePackMapReset( TablePackMap & map ) +{ + if ( map.entries != NULL ) { memset( map.entries, 0, (size_t) map.capacity * sizeof( TablePackEntry ) ); } + map.count = 0; +} + +// open addressing, linear probing, a multiply-shift hash over the address: a +// node key is a pointer and its low bits are alignment, so the low bits alone +// would collide on every node of one type +inline int64_t TablePackMapSlot( const TablePackMap & map, const void * key ) +{ + uint64_t hash = (uint64_t) (uintptr_t) key; + hash *= 0x9E3779B97F4A7C15ull; + hash ^= hash >> 29; + int64_t mask = map.capacity - 1; + int64_t at = (int64_t) ( hash & (uint64_t) mask ); + while ( map.entries[at].key != NULL && map.entries[at].key != key ) + { + at = ( at + 1 ) & mask; + } + return at; +} + +inline TablePackEntry * TablePackMapFind( TablePackMap & map, const void * key ) +{ + if ( map.capacity == 0 ) { return NULL; } + TablePackEntry * entry = &map.entries[ TablePackMapSlot( map, key ) ]; + return entry->key == key ? entry : NULL; +} + +// QUADRUPLING, not doubling, and the reason is measured: growth rehashes every +// entry, and on a graph of 131,071 nodes the doubling schedule spent 45% of +// Lock in rehashing alone. Quadrupling from 1024 buys 1.35x on that graph and +// keeps the map NODE-proportional (§6.2) — under 128 bytes a node at its +// worst, right after a grow, and about 64 on average. +inline bool TablePackMapGrow( TablePackMap & map ) +{ + TablePackMap grown; + grown.allocator = map.allocator; + grown.capacity = map.capacity != 0 ? map.capacity * 4 : 1024; + grown.entries = (TablePackEntry *) map.allocator.alloc( map.allocator.context, grown.capacity * (int64_t) sizeof( TablePackEntry ) ); + if ( grown.entries == NULL ) { return false; } + for ( int64_t i = 0; i < map.capacity; i++ ) + { + if ( map.entries[i].key == NULL ) { continue; } + grown.entries[ TablePackMapSlot( grown, map.entries[i].key ) ] = map.entries[i]; + grown.count++; + } + map.allocator.free( map.allocator.context, map.entries ); + map = grown; + return true; +} + +// REACH a node: one probe answers both questions the walk has. A true "taken" +// says this is a FIRST visit, and the entry is now the node's, coloured open +// at "offset"; otherwise the entry is the one the node already has, and its +// open bit says cycle or sharing. NULL is an allocation failure, and it is a +// refusal like any other: Lock fails rather than packing a graph it cannot +// track. +// +// It is one call and not a find followed by an insert because the walk asks +// this question twice per node — once to measure, once to pack — and every +// probe is a miss into a table larger than L2. +inline TablePackEntry * TablePackMapReach( TablePackMap & map, const void * key, int64_t offset, bool & taken, int64_t & slot ) +{ + if ( ( map.count + 1 ) * 4 >= map.capacity * 3 ) // keep the load factor under three quarters + { + if ( !TablePackMapGrow( map ) ) { return NULL; } + } + slot = TablePackMapSlot( map, key ); + TablePackEntry * entry = &map.entries[slot]; + taken = entry->key != key; // an empty slot is a first visit; the key is never NULL + if ( taken ) + { + entry->key = key; + entry->offset = offset; + entry->open = 1; + map.count++; + } + return entry; +} + +// The descent finished: the node keeps its entry — identity outlives the +// descent — and stops being a cycle. The "hint" is the slot Reach returned, and it +// is checked against the key rather than trusted, so a rehash between the two +// costs a second probe instead of correctness. +inline void TablePackMapClose( TablePackMap & map, const void * key, int64_t hint ) +{ + if ( hint >= 0 && hint < map.capacity && map.entries[hint].key == key ) + { + map.entries[hint].open = 0; + return; + } + TablePackEntry * entry = TablePackMapFind( map, key ); + if ( entry != NULL ) { entry->open = 0; } +} + +// ---- resolution contexts: which encoding a walk is reading ---- + +struct TableArenaCtx { const TableArena * arena; }; +struct TableRegionCtx {}; + +// ---- a BYTE BUFFER's resolution (docs/SPEC-TABLES.md §2.5, §6.3) ---- +// +// The same two encodings a table pointer has, resolved the same way: a +// self-relative delta in a region — one add, no base — and an arena offset +// while the builder is mutable. The blob is reached through its header, and a +// view is the header plus eight and the header's first word. Nothing here +// allocates and nothing copies: off a locked region, a loaded one or an +// opened cook the view points INTO the region. +inline const TableBlob * TableBlobAt( const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableRegionCtx &, const TableRef & ref ) { return TableBlobAt( ref ); } +inline const TableBlob * TableBlobAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +inline const TableBlob * TableBlobAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const TableBlob *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} + +inline TableBytesView TableBytesViewOf( const TableBlob * blob ) +{ + TableBytesView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const uint8_t *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} +inline TableStringView TableStringViewOf( const TableBlob * blob ) +{ + TableStringView view = { NULL, 0 }; + if ( blob != NULL ) { view.data = (const char *) ( blob + 1 ); view.length = (int64_t) blob->length; } + return view; +} + +// the const form's hot path: one add, no base +inline TableBytesView TableBytesAt( const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ref ) ); } +inline TableStringView TableStringAt( const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ref ) ); } +// and the context forms a walk uses: a region context, an arena context, or +// the arena itself while the builder is mutable +template inline TableBytesView TableBytesAt( const Ctx & ctx, const TableRef & ref ) { return TableBytesViewOf( TableBlobAt( ctx, ref ) ); } +template inline TableStringView TableStringAt( const Ctx & ctx, const TableRef & ref ) { return TableStringViewOf( TableBlobAt( ctx, ref ) ); } + +// allocate a blob in the arena and point the slot at it; the slot holds the +// arena offset, as every slot does while the builder is mutable +inline uint8_t * TableBytesEmplace( TableWorker & worker, TableRef & slot, int64_t length ) +{ + TableBytesSlot allocated = worker.AllocBytes( length ); + slot = allocated.ref; + return allocated.data; +} +// the text is copied in when one is given; a NULL text leaves the zeros for +// the caller to fill +inline char * TableStringEmplace( TableWorker & worker, TableRef & slot, const char * text, int64_t length ) +{ + TableStringSlot allocated = worker.AllocString( length ); + slot = allocated.ref; + if ( allocated.data != NULL && text != NULL && length > 0 ) { memcpy( allocated.data, text, (size_t) length ); } + return allocated.data; +} + +// ---- the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table, and a +// pointer field rides as an INDEX into it under kind 17. The encoding is +// flat: no pointer edge is a nesting level, so a chain's length is not a depth, +// and two references to one node are one node. +// +// THE FIELD RIDES ONCE: an L with sixty-four bits of capability frames a +// numbering of any size, so the whole numbering is one contiguous payload and a +// save's node bodies have no aggregate ceiling. + +static const uint64_t kTableNodeIndexNull = 0; // absence and null are one value +static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts the table + +// The not-materialized sentinel (§6.3): a record whose type id this build could +// not name. Distinct from every real offset including the root's 0, so an index +// resolving through it yields NULL and can never fabricate the root. +static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; + +// ---- the numbering, on the SAVE side ---- +// +// One entry per reachable node in FIRST-VISIT order, so entry k is node index +// k + 2. The two thunks are what let one loop write a table of mixed types: the +// numbering walk knows each target's type STATICALLY at the site it numbers it, +// so it stores the instantiation there and the loop never asks what a node is. +struct TableNumbering; + +struct TableNodeEntry +{ + const void * node; + uint64_t type_id; + // the type id's MESSAGE-FORM SLOT (docs/SPEC-TABLES.md §3.3), stored where + // the numbering walk stores the id itself and for the same reason: the + // target's type is known STATICALLY at the site that numbers it, so a + // form 2 save reads the slot out of the entry instead of looking an id up. + // Every pointer target's type id is an entry of the announcement, which is + // what makes the slot a compile-time fact of a POINTERED message too. + uint64_t type_slot; + int64_t ( * measure )( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ); + bool ( * save )( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ); +}; + +struct TableNumbering +{ + TablePackMap seen; // node -> index; the ROOT is index 1, open for the whole walk + TableNodeEntry * entries = NULL; + int64_t count = 0; + int64_t capacity = 0; +}; + +// The numbering allocates through the map's pair rather than carrying a second +// copy of it: one numbering is one walk, and a walk has one allocator. +inline void TableNumberingInit( TableNumbering & n, TableAllocator allocator ) +{ + TablePackMapInit( n.seen, allocator ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +inline void TableNumberingShutdown( TableNumbering & n ) +{ + TableAllocator allocator = n.seen.allocator; + TablePackMapShutdown( n.seen ); + allocator.free( allocator.context, n.entries ); + n.entries = NULL; + n.count = 0; + n.capacity = 0; +} + +// The index a numbered node was given, for the save that writes it into a +// pointer slot. False means the two walks disagree about the graph, which is a +// refusal and never a guess. +inline bool TableNumberingIndex( const TableNumbering & n, const void * node, uint64_t & index ) +{ + if ( n.seen.capacity == 0 ) { return false; } + const TablePackEntry & entry = n.seen.entries[ TablePackMapSlot( n.seen, node ) ]; + if ( entry.key != node ) { return false; } + index = (uint64_t) entry.offset; + return true; +} + +inline bool TableNumberingAppend( TableNumbering & n, const TableNodeEntry & entry ) +{ + if ( n.count == n.capacity ) + { + // GROW BY COPY, never by realloc: the allocator hook is a PAIR, and a + // game's heap is not required to have a resize primitive at all. The + // schedule quadruples, so the copying is amortized to a constant per + // entry and the growth is the same growth it always was. + int64_t capacity = n.capacity != 0 ? n.capacity * 4 : 256; + TableAllocator allocator = n.seen.allocator; + TableNodeEntry * grown = (TableNodeEntry *) allocator.alloc( allocator.context, capacity * (int64_t) sizeof( TableNodeEntry ) ); + if ( grown == NULL ) { return false; } + if ( n.entries != NULL ) + { + memcpy( grown, n.entries, (size_t) n.count * sizeof( TableNodeEntry ) ); + allocator.free( allocator.context, n.entries ); + } + n.entries = grown; + n.capacity = capacity; + } + n.entries[n.count++] = entry; + return true; +} + +// The thunks the numbering stores. Each resolves to the closure member's own +// MeasureBody / SaveBodyFields through an overload set in the member's DECLARING +// file, reached by argument-dependent lookup at instantiation — the same bridge +// the arena's TableReset uses, and the reason a numbering may span the files of +// one unit without any file naming another's members. +template +inline int64_t TableNodeMeasureThunk( const void * ctx, const TableNumbering & numbering, TableIds & ids, const void * node ) +{ + return TableNodeMeasure( *(const Ctx *) ctx, numbering, ids, *(const T *) node ); +} + +template +inline bool TableNodeSaveThunk( const void * ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const void * node ) +{ + return TableNodeSave( *(const Ctx *) ctx, numbering, w, ids, *(const T *) node ); +} + +// ---- a BYTE BUFFER's record (docs/SPEC-TABLES.md §2.5, §3.1) ---- +// +// A blob rides as a node record under one of two RESERVED type ids — the fold +// a table's name takes, over the keywords "bytes" and "string", which no table +// can be named — with the bytes as its body and nothing framed inside. These +// two thunks are what the numbering stores for a blob, as it stores a +// member's codec for a table: the length, and the bytes verbatim. +static const uint64_t kTableBytesTypeId = 0x2f2ec0474f1c4fe4ull; // fnv1a64( "bytes" ) +static const uint64_t kTableStringTypeId = 0x704be0d8faaffc58ull; // fnv1a64( "string" ) + +template +inline int64_t TableBlobMeasureThunk( const void *, const TableNumbering &, TableIds &, const void * node ) +{ + return (int64_t) ( (const TableBlob *) node )->length; +} + +template +inline bool TableBlobSaveThunk( const void *, const TableNumbering &, TableWriter & w, TableIds &, const void * node ) +{ + const TableBlob * blob = (const TableBlob *) node; + w.raw( (const void *) ( blob + 1 ), (int64_t) blob->length ); + return true; +} + +// TableNodeTableMeasure and TableNodeTableSave are the framing, and they are +// ONE fill rule written twice — measure derives it from the graph and save +// derives the same one, which is what makes measure == save hold across a +// pointer graph (§3.1). +// +// The field rides ONCE, under the reserved id, kind 12: the payload opens with +// the count and then carries the records back to back, each a type id +// REFERENCE, a length and a body. The reserved id is interned BEFORE the +// records, and a record's type id before its body, which is the first-use order +// the trailer is written in (§3). +template +inline int64_t TableNodeTablePayload( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + int64_t payload = TableLebBytes( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + payload += TableLebBytes( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return -1; } + payload += TableLebBytes( (uint64_t) body ) + body; + } + return payload; +} + +template +inline int64_t TableNodeTableMeasure( const Ctx & ctx, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return 0; } // a root that reaches no nodes writes none of them + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return -1; } + return TableLebBytes( ref ) + 1 + TableLebBytes( (uint64_t) payload ) + payload; +} + +template +inline bool TableNodeTableSave( const Ctx & ctx, TableWriter & w, TableIds & ids, const TableNumbering & n ) +{ + if ( n.count == 0 ) { return true; } + const uint64_t ref = ids.ref( kTableNodeTableFieldId, kTableNodeTableFieldSlot ); + const int64_t payload = TableNodeTablePayload( ctx, ids, n ); + if ( payload < 0 ) { return false; } + w.putleb( ref ); + w.put8( 12 ); // kind 12 is the opaque byte payload: a reader that cannot name the id skips by L + w.putleb( (uint64_t) payload ); + w.putleb( (uint64_t) n.count ); + for ( int64_t k = 0; k < n.count; k++ ) + { + w.putleb( ids.ref( n.entries[k].type_id, n.entries[k].type_slot ) ); + const int64_t body = n.entries[k].measure( (const void *) &ctx, n, ids, n.entries[k].node ); + if ( body < 0 ) { return false; } + w.putleb( (uint64_t) body ); + if ( !n.entries[k].save( (const void *) &ctx, n, w, ids, n.entries[k].node ) ) { return false; } + } + return true; +} + +// ---- the numbering, on the LOAD side: a region's NODE DIRECTORY (§6.3) ---- +// +// The wire's numbering made resident: one entry per numbered node, in index +// order, position i describing node index i + 1 — so position 0 is the ROOT at +// offset 0. It is ATTRIBUTION, and attribution is separable: nothing that reads +// a structure touches it, a deref is one add on a self-relative offset, and a +// caller may release it once Load returns. +struct TableNodeDirEntry +{ + uint64_t offset; + uint64_t type_id; +}; + +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; + +// TableNodeMap is what a pointer slot resolves through while a body decodes. +struct TableNodeMap +{ + uint8_t * base = NULL; + const TableNodeDirEntry * entries = NULL; + int64_t count = 0; // the ROOT's entry included, so it is records + 1 + bool good = false; // the node table read whole; a numbering that failed resolves nothing + // WHERE THE NODES LIVE, and therefore what a resolved slot holds: a region + // takes the SELF-RELATIVE delta so a deref is one add, and the tool's + // builder path takes the node's ARENA OFFSET (§6.3). + bool arena = false; + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. + TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; +}; + +// TableNodeResolve places one node index in a pointer slot, and every failure +// is one of §4's events with the pointer left null. The declared TARGET type id +// is checked at every index, the root's included: the root carries no record +// and therefore no wire type id, so the READER'S OWN root type is what the +// claim is checked against. +inline void TableNodeResolve( const TableNodeMap & map, TableRef & slot, uint64_t index, uint64_t target, TableReport * report ) +{ + slot.value = 0; + if ( index == kTableNodeIndexNull || !map.good ) { return; } + if ( index - 1 >= (uint64_t) map.count ) + { + report->malformed = true; // an index above node_count + 1 + return; + } + const TableNodeDirEntry & entry = map.entries[index - 1]; + if ( entry.offset == kTableNodeAbsent ) + { + // a node whose type id this build could not name KEEPS ITS INDEX, and + // every pointer naming it reads null. The unknown was counted once, at + // the node, not once per pointer. + return; + } + if ( entry.type_id != target ) + { + report->kind_mismatch++; + return; + } + slot.value = map.arena ? (int64_t) entry.offset + : (int64_t) ( ( map.base + entry.offset ) - (const uint8_t *) &slot ); +} + +// ---- the record SCAN, and it is the whole of load's bound (§3.1) ---- +// +// Reading follows no reference. The scan walks the root body's top-level fields, +// finds the ONE under the reserved id, and reads records out of its payload in +// order — the field rides once, so nothing is copied to make a body contiguous +// and the generated body decoder never learns the transport exists. +struct TableNodeScan +{ + TableReader fields; // over the ROOT body, skipping past everything else + const uint8_t * payload; // the node-table field's payload + int64_t payload_size; + int64_t payload_offset; + bool opened; // the root body has been walked for the field + uint64_t declared; + int64_t records; + bool present; // the root body carries a node table at all + bool malformed; + const TableIdTable * ids; +}; + +inline TableNodeScan TableNodeScanBegin( const uint8_t * body, int64_t size, TableReport * report, const TableIdTable * ids ) +{ + TableNodeScan s = { TableReader( body, size, report, ids ), NULL, 0, 0, false, 0, 0, false, false, ids }; + return s; +} + +// find the node-table field, or answer false when the root body has none. A +// body carrying an id more than once is legal input and THE LAST OCCURRENCE +// WINS (docs/SPEC-TABLES.md §3), so the walk runs to the terminator and keeps +// the last rather than stopping at the first. +inline bool TableNodeScanOpen( TableNodeScan & s ) +{ + if ( s.opened ) { return false; } + s.opened = true; + for ( ;; ) + { + uint64_t ref = 0; + if ( !s.fields.getleb( ref ) ) { break; } + if ( ref == 0 ) { break; } // the terminator + if ( s.ids == NULL || ref > (uint64_t) s.ids->count ) { break; } + const uint64_t id = s.ids->at( ref ); + if ( !s.fields.has( 1 ) ) { break; } + const uint8_t kind = s.fields.get8(); + if ( id == kTableNodeTableFieldId ) + { + s.present = true; + if ( kind != 12 ) { s.malformed = true; return false; } + uint64_t length = 0; + if ( !s.fields.getleb( length ) || !s.fields.room( length ) ) { s.malformed = true; return false; } + s.payload = s.fields.buffer + s.fields.offset; + s.payload_size = (int64_t) length; + s.fields.offset += (int64_t) length; + continue; + } + if ( !s.fields.skip( kind ) ) { break; } + } + if ( s.payload == NULL ) { return false; } + TableReader head( s.payload, s.payload_size, s.fields.report, s.ids ); + if ( !head.getleb( s.declared ) ) { s.malformed = true; return false; } + s.payload_offset = head.offset; + return true; +} + +// the next record, or false at the end of the table — s.malformed says whether +// the end was the end or the framing giving out +inline bool TableNodeScanNext( TableNodeScan & s, uint64_t & type_id, const uint8_t * & body, int64_t & length ) +{ + if ( !s.opened && !TableNodeScanOpen( s ) ) { return false; } + if ( s.payload == NULL || s.payload_offset >= s.payload_size ) { return false; } + TableReader rec( s.payload, s.payload_size, s.fields.report, s.ids ); + rec.offset = s.payload_offset; + uint64_t ref = 0; + if ( !rec.getleb( ref ) || ref == 0 || s.ids == NULL || ref > (uint64_t) s.ids->count ) + { + s.malformed = true; // a type id reference of 0, or one past the table + return false; + } + type_id = s.ids->at( ref ); + uint64_t declared_length = 0; + if ( !rec.getleb( declared_length ) ) + { + s.malformed = true; // a record whose length is damaged + return false; + } + if ( declared_length > (uint64_t) ( s.payload_size - rec.offset ) ) + { + s.malformed = true; // a record whose length runs past its field + return false; + } + body = s.payload + rec.offset; + length = (int64_t) declared_length; + s.payload_offset = rec.offset + length; + s.records++; + return true; +} + +// The record scan is AUTHORITATIVE: node_count is data from the wire, and a +// count that disagrees with the scan is malformed. Nothing is sized from it +// before the scan has confirmed it. +inline bool TableNodeScanWhole( TableNodeScan & s ) +{ + if ( s.malformed ) { return false; } + if ( !s.present ) { return true; } // no node table at all is not a broken one + return s.declared == (uint64_t) s.records; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_ARENA + +#ifndef LISTDEMO_SCHEMA_TABLE_EXTENT +#define LISTDEMO_SCHEMA_TABLE_EXTENT + +namespace listdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_EXTENT + +#ifndef LISTDEMO_SCHEMA_TABLE_MAP +#define LISTDEMO_SCHEMA_TABLE_MAP + +namespace listdemo { + +// ---- a MAP: a sorted entry array, and the lookup over it (§2.8) ---- +// +// On the wire, in a region and in a cook a map is an array of one generated +// ENTRY table held in ascending key order. What this adds is Find — a binary +// search over that array where it lies — and a builder that inserts, replaces +// and erases by key. Nothing here is stored: a region and a cook carry the +// array and the count, and not one byte about a hash or a probe. + +// entries carved from ONE call to the allocator pair; a new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableMapSegmentEntries = 32; + +// TableDeclRef names a type in an unevaluated context and is never defined — +// what 's declval is for, without the include the generated corpus +// refuses to pay for (the iterator_traits note, §13.9). +template T & TableDeclRef(); + +// THE ORDER IS TOTAL, AND IT IS THE SAME IN NINE LANGUAGES (§2.8). Integers +// compare by VALUE, signed for the signed kinds and unsigned for the unsigned. +// Strings compare by BYTES, unsigned, a shorter string that is a prefix of a +// longer one first: memcmp over the common length, then the lengths. Never a +// locale, never a code point, never a case fold. +inline int TableKeyOrder( uint64_t a, uint64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( int64_t a, int64_t b ) { return a < b ? -1 : ( a > b ? 1 : 0 ); } +inline int TableKeyOrder( const char * a, int32_t a_length, const char * b, int32_t b_length ) +{ + const int32_t common = a_length < b_length ? a_length : b_length; + if ( common > 0 ) + { + const int order = memcmp( (const void *) a, (const void *) b, (size_t) common ); + if ( order != 0 ) { return order < 0 ? -1 : 1; } + } + return a_length < b_length ? -1 : ( a_length > b_length ? 1 : 0 ); +} + +// the length of a NUL-terminated key at a call site, bounded by the storage it +// has to fit: a key one byte longer than the bound is refused, never truncated +inline int32_t TableKeyLength( const char * key, int32_t bound ) +{ + if ( key == NULL ) { return 0; } + for ( int32_t i = 0; i <= bound; i++ ) { if ( key[i] == 0 ) { return i; } } + return bound + 1; // longer than the bound: the caller refuses it +} + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.8, §7.2) ---- +// +// An int64 self-relative reference to the entry array and an int32 count, then +// padding to eight. The reference is a TableRef like a pointer's: in the arena +// it names the builder's HEAD, in a region it is the delta from the slot to +// the first entry, and 0 is the empty map in both. +template struct TableMap +{ + TableRef entries; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Entry * Entries() const + { + return entries.value != 0 ? (const Entry *) ( (const uint8_t *) &entries + entries.value ) : NULL; + } + int32_t size() const { return count; } + + // FIND: floor( log2 n ) + 1 key compares, in place, no allocation. NULL + // when absent, and on a map[K]*T the RESOLVED pointer, which is what a + // pointer field's accessor answers. + template const Entry * FindEntry( Key key ) const + { + const Entry * base = Entries(); + int32_t low = 0, high = count; + while ( low < high ) + { + const int32_t mid = low + ( high - low ) / 2; + const int order = TableEntryOrder( base[mid], key ); + if ( order == 0 ) { return base + mid; } + if ( order < 0 ) { low = mid + 1; } else { high = mid; } + } + return NULL; + } + // the return type is DEDUCED, so it is worked out when a call site + // instantiates Find and not when the holder's record declares the slot — + // which is what lets the entry's own overloads be declared after it + template auto Find( Key key ) const + { + return TableEntryFound( FindEntry( key ) ); + } + + // ---- iteration: ASCENDING key order, the key beside the value ---- + // + // A proxy BY VALUE, the keyed array's shape (§2.4): for ( auto [ key, + // value ] : map ). It carries no iterator_traits, for the reason + // TableKeyed's does not (§13.9). + struct ConstEntry + { + decltype( TableEntryKey( TableDeclRef() ) ) key; + decltype( TableEntryFound( (const Entry *) NULL ) ) value; + }; + + struct ConstIterator + { + const Entry * at; + ConstEntry operator*() const { return ConstEntry{ TableEntryKey( *at ), TableEntryFound( at ) }; } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Entries() }; } + ConstIterator end() const { return ConstIterator{ Entries() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.8, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first entry is inserted. Each +// segment is a fixed number of entries carved from one call to the allocator +// pair. An entry's address is stable for the arena's life, so a value handed +// back by an insert stays valid while other entries arrive. +struct TableMapHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an insert appends into + int32_t live; + int32_t dead; +}; + +template struct TableMapSegment +{ + TableRef next; + int32_t used; // entries carved from this segment + int32_t padding; + uint32_t dead[ ( kTableMapSegmentEntries + 31 ) / 32 ]; // Erase marks one bit, never the entry + Entry entries[ kTableMapSegmentEntries ]; +}; + +inline bool TableMapSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// ---- the ORDERED CURSOR the four writing walks read (§2.8) ---- +// +// Measure, Save, Lock and Cook each write a map's entries in ascending key +// order with no key twice, deriving the order from the builder's entries as +// each walk derives the numbering (§3.1). Nothing passes between them, so +// measure == save over a map is a real check on two sorts agreeing. +// +// A REGION is already sorted, so its cursor is the array in place and +// allocates nothing. The BUILDER's is the sort: an array of entry pointers +// allocated through the pair and released before the walk returns, because +// sorting the segments themselves would move entries whose addresses a caller +// holds. +template struct TableMapCursor +{ + const Entry * const * order = NULL; // the builder's form: sorted pointers + const Entry * entries = NULL; // the region's form: the array in place + int32_t count = 0; + TableAllocator allocator; + bool ok = false; + const Entry * operator[]( int32_t index ) const + { + return order != NULL ? order[index] : entries + index; + } +}; + +// heapsort: O( n log n ) once per map, no recursion, no allocation past the +// pointer array the caller already paid for +template inline void TableMapSort( const Entry ** order, int32_t count ) +{ + for ( int32_t start = count / 2 - 1; start >= 0; start-- ) + { + int32_t root = start; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= count ) { break; } + if ( child + 1 < count && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * swap = order[root]; order[root] = order[child]; order[child] = swap; + root = child; + } + } + for ( int32_t end = count - 1; end > 0; end-- ) + { + const Entry * swap = order[0]; order[0] = order[end]; order[end] = swap; + int32_t root = 0; + for ( ;; ) + { + int32_t child = 2 * root + 1; + if ( child >= end ) { break; } + if ( child + 1 < end && TableEntryOrder( *order[child], *order[child + 1] ) < 0 ) { child++; } + if ( TableEntryOrder( *order[root], *order[child] ) >= 0 ) { break; } + const Entry * hold = order[root]; order[root] = order[child]; order[child] = hold; + root = child; + } + } +} + +// the REGION form: the array is already sorted, so the cursor is the array +template +inline TableMapCursor TableMapOrder( const TableRegionCtx &, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.entries = map.Entries(); + cursor.count = map.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: gather the LIVE entries out of the segment chain in +// insertion order, then sort. A dead entry costs nothing on any wire (§2.8). +template +inline TableMapCursor TableMapOrder( const TableArena & arena, const TableMap & map ) +{ + TableMapCursor cursor; + cursor.allocator = arena.allocator; + cursor.count = map.count; + if ( map.entries.value == 0 || map.count <= 0 ) { cursor.ok = map.count == 0; cursor.count = 0; return cursor; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + if ( head->live != map.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + const Entry ** order = (const Entry **) arena.allocator.alloc( arena.allocator.context, (int64_t) map.count * (int64_t) sizeof( const Entry * ) ); + if ( order == NULL ) { return cursor; } + int32_t at = 0; + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 && at < map.count ) + { + const TableMapSegment * segment = (const TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used && at < map.count; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + order[at++] = segment->entries + i; + } + segment_ref = segment->next; + } + if ( at != map.count ) + { + arena.allocator.free( arena.allocator.context, order ); + return cursor; + } + TableMapSort( order, map.count ); + cursor.order = order; + cursor.ok = true; + return cursor; +} + +template +inline TableMapCursor TableMapOrder( const TableArenaCtx & ctx, const TableMap & map ) +{ + return TableMapOrder( *ctx.arena, map ); +} + +template inline void TableMapRelease( TableMapCursor & cursor ) +{ + if ( cursor.order != NULL ) { cursor.allocator.free( cursor.allocator.context, (void *) cursor.order ); } + cursor.order = NULL; +} + +// ---- the builder's five (§2.8) ---- +// +// Insert APPENDS after one LINEAR SCAN of the live entries for the key it may +// replace, Find is that same scan, and Erase is the scan and one bit. The +// builder builds NO INDEX, and that is a rule: the sort happens once, at Lock, +// Save or Cook, and every lookup that matters runs over the sorted region. + +// the head, allocated when the first entry is inserted +template +inline TableMapHead * TableMapReach( TableWorker & worker, TableMap & map ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( map.entries.value != 0 ) { return (TableMapHead *) TableArenaAt( *worker.arena, (uint32_t) map.entries.value ); } + uint32_t at = 0; + TableMapHead * head = (TableMapHead *) worker.AllocRaw( (int64_t) sizeof( TableMapHead ), (int64_t) alignof( TableMapHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + map.entries.value = (int64_t) at; + return head; +} + +// one entry's storage, appended: the current segment when it has room, a new +// one carved from one call to the pair when it does not +template +inline Entry * TableMapAppend( TableWorker & worker, TableMapHead * head, TableMap & map ) +{ + TableMapSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableMapSegmentEntries ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableMapSegment *) worker.AllocRaw( (int64_t) sizeof( TableMapSegment ), (int64_t) alignof( TableMapSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableMapSegment * previous = (TableMapSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Entry * entry = segment->entries + segment->used; + segment->used++; + head->live++; + map.count++; + return entry; +} + +// the LINEAR SCAN: the live entries in insertion order, O( n ) key compares +template +inline Entry * TableMapScan( const TableArena & arena, const TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) == 0 ) { return segment->entries + i; } + } + segment_ref = segment->next; + } + return NULL; +} + +// ERASE marks the entry DEAD, one bit in the segment's slot and not in the +// entry table, and decrements the live count. Its storage is reclaimed at +// RESET and never reused mid-build, because reusing a slot would make "an +// entry's address is stable" false for exactly one case. +template +inline bool TableMapErase( TableArena & arena, TableMap & map, Key key ) +{ + if ( map.entries.value == 0 ) { return false; } + TableMapHead * head = (TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( TableEntryOrder( segment->entries[i], key ) != 0 ) { continue; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + map.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INSERTION order, live entries only (§2.8) ---- +template struct TableMapEach +{ + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableMapSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableMapSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + auto operator*() const { return TableEntryEach( segment->entries + index ); } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableMapSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableMapEach TableMapEachOf( const TableArena & arena, const TableMap & map ) +{ + TableMapEach each = { &arena, TableRef() }; + if ( map.entries.value != 0 ) + { + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + each.first = head->first; + } + return each; +} + +// ---- the LOAD side: where a decoded entry lands (§2.8) ---- +// +// THE READER TRUSTS NOTHING and spends one compare per entry. Every load path +// applies the same rules and produces one report (§4), so the region load of +// §6.5 and LoadBuilder never disagree about a wire. These two shapes are what +// makes that true with one generated decoder: a REGION carves the entry array +// out of the holder node's own extent, and the TOOL's path appends into the +// builder's arena, and the decoder above them cannot tell which it has. + +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. + +// TableMapFill is one map field being decoded: where the next entry lands, and +// the entry that last LANDED, which is what the ascending check compares +// against. +template struct TableMapFill +{ + TableMap * map = NULL; + Entry * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; +}; + +template +inline TableMapFill TableMapFillBegin( const TableNodeMap & nodes, TableMap & map, uint32_t n ) +{ + TableMapFill fill; + fill.map = ↦ + map.entries.value = 0; + map.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Entry ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Entry ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Entry *) base; + fill.capacity = (int32_t) n; + map.entries.value = (int64_t) ( base - (const uint8_t *) &map.entries ); + fill.ok = true; + return fill; +} + +// the entry that last LANDED — NULL before the first +template inline Entry * TableMapFillLast( TableMapFill & fill ) +{ + if ( fill.map->count <= 0 ) { return NULL; } + if ( fill.array != NULL ) { return fill.array + ( fill.map->count - 1 ); } + return TableMapLive( *fill.worker->arena, *fill.map, fill.map->count - 1 ); +} + +// the next slot, at the entry type's declared defaults +template inline Entry * TableMapFillNext( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + if ( fill.map->count >= fill.capacity ) { return NULL; } + Entry * entry = fill.array + fill.map->count; + TableReset( *entry ); + fill.map->count++; + return entry; + } + TableMapHead * head = TableMapReach( *fill.worker, *fill.map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( *fill.worker, head, *fill.map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// A MAP WITH HALF ITS KEYS IS NOT A MAP (§2.8): at the first entry whose key +// kind disagrees with the reader's declaration the map resets to EMPTY, one +// kind_mismatch is counted for the map, and its remaining bytes are skipped. +template inline void TableMapFillReset( TableMapFill & fill ) +{ + if ( fill.array != NULL ) + { + fill.map->entries.value = 0; + fill.map->count = 0; + return; + } + if ( fill.map->entries.value != 0 ) + { + TableMapHead * head = (TableMapHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.map->entries.value ); + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + } + fill.map->count = 0; +} + +// an EMPTY map's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableMapFillEnd( TableMapFill & fill ) +{ + if ( fill.array != NULL && fill.map->count == 0 ) { fill.map->entries.value = 0; } +} + +// the k-th LIVE entry of a builder map, in insertion order — what the tool +// path's ascending check compares against +template +inline Entry * TableMapLive( const TableArena & arena, const TableMap & map, int32_t index ) +{ + if ( map.entries.value == 0 ) { return NULL; } + const TableMapHead * head = (const TableMapHead *) TableArenaAt( arena, (uint32_t) map.entries.value ); + TableRef segment_ref = head->first; + int32_t at = 0; + while ( segment_ref.value != 0 ) + { + TableMapSegment * segment = (TableMapSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + for ( int32_t i = 0; i < segment->used; i++ ) + { + if ( TableMapSegmentDead( segment->dead, i ) ) { continue; } + if ( at == index ) { return segment->entries + i; } + at++; + } + segment_ref = segment->next; + } + return NULL; +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.8, §6.5) ---- +// +// LoadMeasure's term for a map is N x sizeof( Entry ) rounded to +// alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this +// reads no field: it walks the map's own header and, where an entry's value +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. +// A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its +// own L and the body's terminator, and under this form's variable lengths that +// footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a +// map's L can carry, and therefore what a LoadMeasure may be asked for. +static const int64_t kTableMapEntryFloor = 2; + +inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry + at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); + at += (int64_t) n * entry_size; + if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// ---- the TEXT form's placement (docs/SPEC-TABLES.md §2.8, §16) ---- +// +// The text is a plain JSON object keyed by the KEY, and the generic walk fills +// it through the ENTRY'S OWN descriptor — so all it needs from here is one +// entry at one key, handed back at its defaults. It is the builder's Insert +// with the ENTRY returned rather than its value, because the walk writes the +// value through a field row and not through a typed pointer. +template +inline Entry * TableMapPlace( TableWorker & worker, TableMap & map, Key key ) +{ + if ( worker.arena == NULL ) { return NULL; } + Entry * found = TableMapScan( *worker.arena, map, key ); + if ( found != NULL ) + { + TableResetMapValue( *found ); // a repeated key is LAST-WINS, whole + return found; + } + TableMapHead * head = TableMapReach( worker, map ); + if ( head == NULL ) { return NULL; } + Entry * entry = TableMapAppend( worker, head, map ); + if ( entry != NULL ) { TableReset( *entry ); } + return entry; +} + +// ---- the OPTIONAL RUNTIME INDEX (§2.8) ---- +// +// Open addressing with LINEAR PROBING over the sorted array, built AT LOAD for +// a map large enough that log n compares over a cold array cost more than one +// hash and a probe. IT IS NEVER STORED: the caller measures it, owns its +// storage, builds it in one pass and releases it whenever. +// +// ITS HASH AND ITS LOAD FACTOR ARE NOT A CROSS-PORT CONTRACT, and that is a +// rule. What a port is held to is the CONTRACT of the lookup: the same value +// the sorted array's Find returns for the same key, and no allocation past the +// storage the caller handed in. +struct TableMapIndex +{ + int32_t * slots = NULL; // entry indices, +1; 0 is an empty slot + int32_t capacity = 0; + bool good = false; +}; + +// this runtime's own, and no port reproduces it: fnv1a64 over the key's bytes +inline uint64_t TableMapHash( const void * bytes, int32_t length ) +{ + uint64_t hash = 0xCBF29CE484222325ull; + const uint8_t * at = (const uint8_t *) bytes; + for ( int32_t i = 0; i < length; i++ ) { hash ^= (uint64_t) at[i]; hash *= 0x100000001B3ull; } + return hash; +} +inline uint64_t TableMapHash( uint64_t key ) { return TableMapHash( (const void *) &key, (int32_t) sizeof( key ) ); } + +// this runtime's own load factor, and no port reproduces it either: the next +// power of two at or above twice the count, so a probe run stays short +inline int32_t TableMapIndexSlots( int32_t count ) +{ + int32_t slots = 8; + while ( slots < count * 2 ) { slots *= 2; } + return slots; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_MAP + +#ifndef LISTDEMO_SCHEMA_TABLE_LIST +#define LISTDEMO_SCHEMA_TABLE_LIST + +namespace listdemo { + +// ---- an UNBOUNDED ARRAY: a counted array whose count the data decides (§2.9) ---- +// +// On the wire, in a region and in a cook a list is the kind 14 body a [..N]T +// writes, its elements by-value records inside the holder's node extent. What +// this adds is the slot, a builder that appends into segments that never +// move, and a const surface that indexes and iterates in place. There is no +// sort, no key and no lookup: the order is INSERTION order, and it is +// identity the way position is identity in a fixed array. + +// elements carved from ONE call to the allocator pair. A new segment is +// appended when the current one fills, and nothing ever moves (§6.4) +static const int32_t kTableListSegmentElements = 32; + +// THE ELEMENT STORAGE: T itself, and a TableRef slot for a []*T, whose +// elements are references exactly as a pointer field's slot is (§2.1) +template struct TableListStorage { typedef T Element; }; +template struct TableListStorage { typedef TableRef Element; }; + +// WHAT THE CONST FORM ANSWERS: the element by reference, and on a []*T the +// RESOLVED pointer, one add on the self-relative delta, NULL for a null slot, +// exactly as At answers it (§6.2, §6.3) +template struct TableListConst +{ + typedef const T & Result; + static Result At( const T * element ) { return *element; } +}; +template struct TableListConst +{ + typedef const T * Result; + static Result At( const TableRef * element ) + { + return element->value != 0 ? (const T *) ( (const uint8_t *) element + element->value ) : NULL; + } +}; + +// ---- the storage: SIXTEEN BYTES in the holder's record (§2.9, §7.2) ---- +// +// An int64 self-relative reference to the element array and an int32 count, +// then padding to eight. The reference is a TableRef like a pointer's: in the +// arena it names the builder's HEAD, in a region it is the delta from the slot +// to the first element, and 0 is the empty list in both. It is the map's slot +// exactly, because it is the same two facts. +template struct TableList +{ + typedef typename TableListStorage::Element Element; + + TableRef elements; + int32_t count = 0; // the LIVE count, in both forms + int32_t padding = 0; // named, so the record has no unwritten byte in it + + // ---- the CONST form: a locked region, a loaded one, an opened cook ---- + // + // One surface over one encoding (§6.3). A region reference resolves from + // the slot's own address, so every one of these is a member and needs no + // base and no context. + const Element * Elements() const + { + return elements.value != 0 ? (const Element *) ( (const uint8_t *) &elements + elements.value ) : NULL; + } + int32_t size() const { return count; } + + // INDEXING IS BOUNDS-CHECKED IN EVERY BUILD (§2.4, §2.9): the extent is a + // number that CAME FROM A FILE, so an index past it is not a mistake a + // release build gets to make cheaply. There is no undefined-behavior path + // here in any configuration. The assert carries the message where a + // debugger can read it and NDEBUG removes that. The fatal is what stands + // after it. Both go through the hooks: define schema_assert and + // schema_fatal and this refusal lands in your own handler. + void RefuseIndex( int32_t index ) const + { + if ( (uint32_t) index >= (uint32_t) count ) + { + schema_assert( false && "an unbounded array is indexed inside its count, which came from a file" ); + schema_fatal(); + } + } + typename TableListConst::Result operator[]( int32_t index ) const + { + RefuseIndex( index ); + return TableListConst::At( Elements() + index ); + } + + // ---- iteration: INDEX order, the element and no key ---- + // + // It carries no iterator_traits, for the reason TableKeyed's does not + // (§13.9). + struct ConstIterator + { + const Element * at; + typename TableListConst::Result operator*() const { return TableListConst::At( at ); } + ConstIterator & operator++() { at++; return *this; } + bool operator==( const ConstIterator & other ) const { return at == other.at; } + bool operator!=( const ConstIterator & other ) const { return at != other.at; } + }; + + ConstIterator begin() const { return ConstIterator{ Elements() }; } + ConstIterator end() const { return ConstIterator{ Elements() + count }; } +}; + +// ---- the BUILDER's side: a head, and segments that never move (§2.9, §6.4) ---- +// +// The head is a small node in the arena holding the segment chain, the live +// count and the dead count, allocated when the first element is added. Each +// segment is a fixed number of elements carved from one call to the allocator +// pair. An element's address is stable for the arena's life, so a T * handed +// back by Add stays valid while other elements arrive. +struct TableListHead +{ + TableRef first; // the arena offset of the first segment + TableRef last; // and of the one an Add appends into + int32_t live; + int32_t dead; +}; + +template struct TableListSegment +{ + TableRef next; + int32_t used; // elements carved from this segment + int32_t padding; + uint32_t dead[ ( kTableListSegmentElements + 31 ) / 32 ]; // Erase marks one bit, never the element + Element elements[ kTableListSegmentElements ]; +}; + +inline bool TableListSegmentDead( const uint32_t * dead, int32_t index ) +{ + return ( dead[ index / 32 ] & ( 1u << ( index % 32 ) ) ) != 0; +} + +// the head, allocated when the first element is added +template +inline TableListHead * TableListReach( TableWorker & worker, TableList & list ) +{ + if ( worker.arena == NULL || worker.arena->locked ) { return NULL; } + if ( list.elements.value != 0 ) { return (TableListHead *) TableArenaAt( *worker.arena, (uint32_t) list.elements.value ); } + uint32_t at = 0; + TableListHead * head = (TableListHead *) worker.AllocRaw( (int64_t) sizeof( TableListHead ), (int64_t) alignof( TableListHead ), at ); + if ( head == NULL ) { return NULL; } + head->first.value = 0; + head->last.value = 0; + head->live = 0; + head->dead = 0; + list.elements.value = (int64_t) at; + return head; +} + +// one element's storage, appended: the current segment when it has room, a +// new one carved from one call to the pair when it does not. NULL means NOT +// ADDED: an arena that cannot carve another segment, or a count at the int32 +// cap (§2.2, §2.9). +template +inline typename TableList::Element * TableListAppend( TableWorker & worker, TableListHead * head, TableList & list ) +{ + typedef typename TableList::Element Element; + if ( list.count >= INT32_MAX ) { return NULL; } // the int32 storage cap + TableListSegment * segment = NULL; + if ( head->last.value != 0 ) + { + segment = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + if ( segment->used >= kTableListSegmentElements ) { segment = NULL; } + } + if ( segment == NULL ) + { + uint32_t at = 0; + segment = (TableListSegment *) worker.AllocRaw( (int64_t) sizeof( TableListSegment ), (int64_t) alignof( TableListSegment ), at ); + if ( segment == NULL ) { return NULL; } // the arena could not carve another segment + segment->next.value = 0; + segment->used = 0; + segment->padding = 0; + for ( int32_t i = 0; i < (int32_t) ( sizeof( segment->dead ) / sizeof( segment->dead[0] ) ); i++ ) { segment->dead[i] = 0; } + if ( head->last.value != 0 ) + { + TableListSegment * previous = (TableListSegment *) TableArenaAt( *worker.arena, (uint32_t) head->last.value ); + previous->next.value = (int64_t) at; + } + else + { + head->first.value = (int64_t) at; + } + head->last.value = (int64_t) at; + } + Element * element = segment->elements + segment->used; + segment->used++; + head->live++; + list.count++; + return element; +} + +// ADD, whole: the head, the append, and the element at its declared defaults +// (§2.9). The text form's placement is this same call, because a list has no +// key to place under (§16). +template +inline typename TableList::Element * TableListPlace( TableWorker & worker, TableList & list ) +{ + typedef typename TableList::Element Element; + TableListHead * head = TableListReach( worker, list ); + if ( head == NULL ) { return NULL; } + Element * element = TableListAppend( worker, head, list ); + if ( element == NULL ) { return NULL; } + new ( element ) Element(); // value-init: the declared defaults, and null for a slot + return element; +} + +// ERASE, ADDRESSED BY THE POINTER (§2.9): the element Add handed back is the +// handle, because a list has no key and the address is the one thing the +// builder promises never moves (§6.4). It marks the element DEAD, one bit in +// the segment's slot and not in the element storage, and decrements the live +// count. False when the pointer is not this list's. Its storage is reclaimed +// at RESET and never reused mid-build, the map's rule for the map's reason. +template +inline bool TableListErase( TableArena & arena, TableList & list, const typename TableList::Element * element ) +{ + typedef typename TableList::Element Element; + if ( list.elements.value == 0 || element == NULL ) { return false; } + TableListHead * head = (TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + TableRef segment_ref = head->first; + while ( segment_ref.value != 0 ) + { + TableListSegment * segment = (TableListSegment *) TableArenaAt( arena, (uint32_t) segment_ref.value ); + if ( element >= segment->elements && element < segment->elements + segment->used ) + { + const int32_t i = (int32_t) ( element - segment->elements ); + if ( TableListSegmentDead( segment->dead, i ) ) { return false; } // already erased + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + list.count--; + return true; + } + segment_ref = segment->next; + } + return false; +} + +// ---- iterate on the BUILDER: INDEX order, live elements only (§2.9) ---- +template struct TableListEach +{ + typedef typename TableList::Element Element; + const TableArena * arena; + TableRef first; + + struct Iterator + { + const TableArena * arena; + TableListSegment * segment; + int32_t index; + + void Skip() + { + for ( ;; ) + { + if ( segment == NULL ) { return; } + if ( index >= segment->used ) + { + segment = segment->next.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + index = 0; + continue; + } + if ( TableListSegmentDead( segment->dead, index ) ) { index++; continue; } + return; + } + } + Element * operator*() const { return segment->elements + index; } + Iterator & operator++() { index++; Skip(); return *this; } + bool operator==( const Iterator & other ) const { return segment == other.segment && index == other.index; } + bool operator!=( const Iterator & other ) const { return !( *this == other ); } + }; + + Iterator begin() const + { + Iterator it = { arena, first.value != 0 ? (TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL, 0 }; + it.Skip(); + return it; + } + Iterator end() const { Iterator it = { arena, NULL, 0 }; return it; } +}; + +template +inline TableListEach TableListEachOf( const TableArena & arena, const TableList & list ) +{ + TableListEach each = { &arena, TableRef() }; + if ( list.elements.value != 0 ) + { + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + each.first = head->first; + } + return each; +} + +// ---- the INDEX-ORDER CURSOR the four writing walks read (§2.9) ---- +// +// Measure, Save, Lock and Cook each visit a list's live elements in the order +// they were added, and they allocate nothing to do it: a region's cursor is +// the array in place, and the builder's walks the segment chain. Indexing the +// builder's form is SEQUENTIAL by construction, every walk steps i, i + 1, +// i + 2, so the cursor remembers where the last access landed and moves one +// live slot per step. An access behind the memo restarts from the first +// segment, which no walk here does. +template struct TableListCursor +{ + const Element * elements = NULL; // the region's form: the array in place + const TableArena * arena = NULL; // the builder's form: the segments + TableRef first; + int32_t count = 0; + bool ok = false; + // the memo: the segment and slot the last access landed on, and the live + // index that slot holds + mutable const TableListSegment * segment = NULL; + mutable int32_t within = -1; + mutable int32_t logical = -1; + + const Element * At( int32_t index ) const + { + if ( elements != NULL ) { return elements + index; } + if ( segment == NULL || index < logical ) + { + segment = first.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) first.value ) : NULL; + within = -1; + logical = -1; + } + while ( logical < index ) + { + for ( ;; ) + { + within++; + while ( segment != NULL && within >= segment->used ) + { + segment = segment->next.value != 0 ? (const TableListSegment *) TableArenaAt( *arena, (uint32_t) segment->next.value ) : NULL; + within = 0; + } + if ( segment == NULL ) { return NULL; } // the slot and the head disagree + if ( !TableListSegmentDead( segment->dead, within ) ) { break; } + } + logical++; + } + return segment->elements + within; + } + const Element & operator[]( int32_t index ) const { return *At( index ); } +}; + +// the REGION form: the array is the cursor +template +inline TableListCursor::Element> TableListElements( const TableRegionCtx &, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.elements = list.Elements(); + cursor.count = list.count; + cursor.ok = true; + return cursor; +} + +// the BUILDER's form: the live elements out of the segment chain, in the +// order they were added. A dead element costs nothing on any wire (§2.9). +template +inline TableListCursor::Element> TableListElements( const TableArena & arena, const TableList & list ) +{ + TableListCursor::Element> cursor; + cursor.arena = &arena; + cursor.count = list.count; + if ( list.elements.value == 0 || list.count <= 0 ) { cursor.ok = list.count == 0; cursor.count = 0; return cursor; } + const TableListHead * head = (const TableListHead *) TableArenaAt( arena, (uint32_t) list.elements.value ); + if ( head->live != list.count ) { return cursor; } // the slot and the head disagree: refused, never guessed + cursor.first = head->first; + cursor.ok = true; + return cursor; +} + +template +inline TableListCursor::Element> TableListElements( const TableArenaCtx & ctx, const TableList & list ) +{ + return TableListElements( *ctx.arena, list ); +} + +// ---- the LOAD side: where a decoded element lands (§2.9) ---- +// +// The same two shapes the map's fill takes, because the decoder above them +// cannot tell which it has: a REGION carves the element array out of the +// holder node's own extent, PRE-ORDER, and the TOOL's path appends into the +// builder's arena. +template struct TableListFill +{ + typedef typename TableList::Element Element; + TableList * list = NULL; + Element * array = NULL; // the region path: the carved array + int32_t capacity = 0; + TableWorker * worker = NULL; // the TOOL's path + bool ok = false; + bool refused = false; // a count above the int32 cap on the tool's path: LoadBuilder answers NULL +}; + +template +inline TableListFill TableListFillBegin( const TableNodeMap & nodes, TableList & list, uint64_t n ) +{ + typedef typename TableList::Element Element; + TableListFill fill; + fill.list = &list; + list.elements.value = 0; + list.count = 0; + if ( nodes.carve == NULL ) { return fill; } + if ( n > (uint64_t) INT32_MAX ) + { + // A COUNT ABOVE THE int32 STORAGE CAP (§2.2, §2.9): into a region it was + // refused by LoadMeasure before this ran, and into a builder it is the + // refusal LoadBuilder answers NULL for, moving no counter + fill.refused = nodes.carve->worker != NULL; + return fill; + } + if ( nodes.carve->worker != NULL ) + { + fill.worker = nodes.carve->worker; // the tool's path: the arena carves + fill.ok = true; + return fill; + } + const int64_t align = (int64_t) alignof( Element ); + uint8_t * base = (uint8_t *) ( ( (uintptr_t) nodes.carve->at + (uintptr_t) ( align - 1 ) ) & ~( (uintptr_t) ( align - 1 ) ) ); + const int64_t bytes = (int64_t) n * (int64_t) sizeof( Element ); + const int64_t pad = (int64_t) ( base - nodes.carve->at ); + if ( pad + bytes > nodes.carve->left ) { return fill; } // the measure and the load disagree: refused + nodes.carve->at = base + bytes; + nodes.carve->left -= pad + bytes; + fill.array = (Element *) base; + fill.capacity = (int32_t) n; + list.elements.value = (int64_t) ( base - (const uint8_t *) &list.elements ); + fill.ok = true; + return fill; +} + +// the next slot, at the element's declared defaults. NULL when the arena +// could not carve, which the decoder reports as framing damage +template inline typename TableList::Element * TableListFillNext( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count >= fill.capacity ) { return NULL; } + Element * element = fill.array + fill.list->count; + new ( element ) Element(); + fill.list->count++; + return element; + } + return TableListPlace( *fill.worker, *fill.list ); +} + +// A SLOT WHOSE ELEMENT NEVER LANDED is given back (§2.9, §4): the array keeps +// what it decoded, and an element whose own framing gave out before one byte +// of it decoded was not decoded. The region's form uncounts it, and the builder's +// marks it dead, which is what the storage rule allows mid-build. +template inline void TableListFillDrop( TableListFill & fill ) +{ + typedef typename TableList::Element Element; + if ( fill.array != NULL ) + { + if ( fill.list->count > 0 ) { fill.list->count--; } + return; + } + if ( fill.list->elements.value == 0 ) { return; } + TableListHead * head = (TableListHead *) TableArenaAt( *fill.worker->arena, (uint32_t) fill.list->elements.value ); + if ( head->last.value == 0 ) { return; } + TableListSegment * segment = (TableListSegment *) TableArenaAt( *fill.worker->arena, (uint32_t) head->last.value ); + if ( segment->used <= 0 ) { return; } + const int32_t i = segment->used - 1; + if ( TableListSegmentDead( segment->dead, i ) ) { return; } + segment->dead[ i / 32 ] |= 1u << ( i % 32 ); + head->live--; + head->dead++; + fill.list->count--; +} + +// an EMPTY list's reference is null in both encodings, so a load that placed +// nothing leaves the slot exactly as a Reset does +template inline void TableListFillEnd( TableListFill & fill ) +{ + if ( fill.array != NULL && fill.list->count == 0 ) { fill.list->elements.value = 0; } +} + +// ---- LoadMeasure's term, from the FRAMING alone (§2.9, §6.5) ---- +// +// N x sizeof( T ) rounded to alignof( T ), AT EVERY DEPTH. N is framing and +// not a value, so this reads no field: it walks the list's own header and, +// where a table element holds a list or a map of its own, the elements' +// headers under it. Every -1 carries its REASON (§6.5): the int32 cap first, +// because a count past it cannot fit any body, and then the body's own L. +inline bool TableListWireExtent( const uint8_t * body, int64_t length, int64_t & at, + int64_t elem_size, int64_t elem_align, uint8_t elem_kind, int64_t elem_floor, + TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } // no array header: nothing rides + if ( r.get8() != elem_kind ) { return true; } // another element kind: §4's ordinary kind mismatch, the field reads empty + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } + const int64_t rest = length - r.offset; + if ( n > (uint64_t) ( rest / elem_floor ) ) { reason = count_over_length; return false; } // an N the list's L cannot carry + at = ( at + elem_align - 1 ) & ~( elem_align - 1 ); + at += (int64_t) n * elem_size; + if ( inner == NULL ) { return true; } // nothing below an element: one depth is the whole term + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_LIST + +#ifndef LISTDEMO_SCHEMA_BUILD_VERSION +#define LISTDEMO_SCHEMA_BUILD_VERSION + +namespace listdemo { + +// THE BUILD VERSION (docs/SPEC-TABLES.md §20): one digest over every fact the bytes +// this build produces depend on — the type wire's protocol id, every record's +// layout as the compiler's own C ABI model computes it, and the facts that +// decide what a load PUTS in those slots. It is the number a cook's header +// carries and the number Open compares, and the number a block's prologue +// carries and BlockOpen compares: a build version answers "which build?" and +// not "which form?", and what separates the two forms is their MAGIC. +// +// There are TWO ids in the design and they are not interchangeable: the +// PROTOCOL ID is the type wire's and nothing else, and the BUILD VERSION is +// what everything cooked or blocked is keyed by. A table edit moves this and +// never the protocol id; a type edit moves both. +static const uint64_t BuildVersion = 0x8d7c0edaca4571c7ull; + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_BUILD_VERSION + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK +#define LISTDEMO_SCHEMA_TABLE_COOK + +namespace listdemo { + +// ---- the cooked form (docs/SPEC-TABLES.md §7) ---- +// +// A cooked file is a HEADER, a DATA part and an ATTRIBUTION part, in that +// order. Every word of the header is a u64 written in the byte order the cook +// was produced in, and the header is 64 bytes: +// +// 0 magic 0x4b4f4f434d484353, read BYTEWISE before anything else +// 8 build_version the unit's id (docs/SPEC-TABLES.md §20) +// 16 byte_order 1 little, 2 big — the order that WROTE the file +// 24 data_length the region's bytes, rounded up to alignment +// 32 attribution_length the directory's bytes, or 0 +// 40 alignment the region's alignment, never below eight +// 48 reserved zero +// 56 reserved zero +// +// The DATA part is Lock's region written verbatim (§7.2) — the root at its +// base — and it is what a runtime points at. The ATTRIBUTION part is the node +// directory (§6.3), and NOTHING THAT READS THE STRUCTURE TOUCHES IT: it is +// written beside the data for schema cook-check, so a build that ships no +// tooling need not carry it at all. +static const int64_t kTableCookHeaderBytes = 64; + +// THE MAGIC'S VALUE, and a consumer written from the page needs the constant +// rather than a description of one. It is "SCHMCOOK" read as ASCII in the byte +// order a little-endian store produces — the same shape the block form's +// SCHMABLK takes, so a hex dump of a little-endian cook is legible and the two +// accelerators sit in one vocabulary. +// +// IT IS STORED IN THE PRODUCER'S ORDER, which is what makes it the byte-order +// check as well as the form check: a consumer reads back this build's +// constant, or that constant byte-reversed — which identifies a cook of the +// OTHER order — or something that is not a cook. All three answers but the +// first refuse, and a cook and a BLOCK are separated here too, because a +// form's identity belongs in its magic rather than in a second digest. +static const uint64_t TableCookMagic = 0x4b4f4f434d484353ull; + +// THIS BUILD's byte order, as the header's own word carries it. The magic is +// what REFUSES a foreign order; this word is what RECORDS which order wrote +// the file, so a refusal names the order rather than inferring it and a tool +// dumping a cook reads the fact. A file whose magic matched and whose order +// word did not is corrupt, and there is no reading that recovers it. +// +// The BUILD VERSION cannot do either job: §20.1 digests byteorder as a +// GENERATION input, little for every target schema generates for today, so +// two builds of one schema for two orders emit the same id. +#if defined( __BYTE_ORDER__ ) && defined( __ORDER_BIG_ENDIAN__ ) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__ +static const uint64_t TableCookByteOrder = 2; // big +#else +static const uint64_t TableCookByteOrder = 1; // little +#endif + +// The greatest region alignment a cooked file may name. The DATA part begins +// at align_up( 64, alignment ), which is 64 for every unit this language can +// declare — the largest alignment it has is sixteen — so a word past this cap +// describes a file no build of this schema wrote (docs/SPEC-TABLES.md §7.1). +static const uint64_t TableCookMaxAlign = 64; + +// The header read, BYTEWISE. memcpy is the portable spelling of "these eight +// bytes, in this machine's order"; every compiler this repo builds under folds +// it to one load, and it is the only read in the whole of Open that is not a +// comparison. +inline uint64_t table_cook_read64( const uint8_t * p ) +{ + uint64_t v; + memcpy( &v, p, sizeof( v ) ); + return v; +} + +// TableCookOpen: THE WHOLE CHECK, in one place, because §7 states the +// enumeration once and every generated Open is that one enumeration plus +// its own root's two layout facts. +// +// THE CHECK, in order: the magic read bytewise, the byte order it establishes, +// the build version against this build's own, both RESERVED words zero, the +// region alignment the header names, the two part lengths against the length +// the caller passed — a truncated file and a file with trailing bytes are the +// same refusal — the root's own storage inside the data part, and the +// alignment of the base. +// +// AND THAT IS ALL OF IT. On a match the bytes ARE what this build wrote, in +// this build's layout and this build's byte order, so there is nothing to +// validate and nothing to fix up: the caller gets the root. Nothing per node +// happens here, which is what makes open O(1) in the file's size; a walk of +// any shape would forfeit that, and validating an untrusted file is schema +// cook-check's job and a person's decision (§7.4). +// +// EVERY NUMBER BELOW COMES OUT OF THE FILE, so the arithmetic is unsigned and +// each term is BOUNDED BEFORE IT IS ADDED: a forged length near 2^64 must +// refuse, and an addition that wrapped would be the defect the comparison +// after it was supposed to catch. Nothing past length is read on any path, +// including every refusing one. +inline const uint8_t * TableCookOpen( const void * bytes, uint64_t length, uint64_t root_size, uint64_t root_align ) +{ + if ( bytes == NULL ) { return NULL; } + if ( length < (uint64_t) kTableCookHeaderBytes ) { return NULL; } + const uint8_t * raw = (const uint8_t *) bytes; + // the MAGIC, bytewise and first: it is what establishes the byte order + // every other header word is read in, so nothing else may be read before + // it. A byte-reversed constant is a cook of the other order and refuses + // here, which is why the order never reaches a fix-up pass. + if ( table_cook_read64( raw ) != TableCookMagic ) { return NULL; } + if ( table_cook_read64( raw + 16 ) != TableCookByteOrder ) { return NULL; } + if ( table_cook_read64( raw + 8 ) != BuildVersion ) { return NULL; } + // the RESERVED words: a non-zero one means a writer used a form this build + // does not understand, and Open refuses rather than ignoring it. + if ( table_cook_read64( raw + 48 ) != 0 ) { return NULL; } + if ( table_cook_read64( raw + 56 ) != 0 ) { return NULL; } + const uint64_t data_length = table_cook_read64( raw + 24 ); + const uint64_t attribution_length = table_cook_read64( raw + 32 ); + const uint64_t alignment = table_cook_read64( raw + 40 ); + // THE ALIGNMENT WORD IS DATA, and it is the one header field the rest of + // the check does arithmetic WITH rather than only comparison against. A + // region's alignment is a power of two, never below eight (the floor that + // puts the attribution part on an eight-byte boundary without a second + // padding rule) and never past the cap above; a word that is none of those + // rounds nothing and aligns nothing, so it is refused before it is used. + if ( alignment < 8 || alignment > TableCookMaxAlign ) { return NULL; } + if ( ( alignment & ( alignment - 1 ) ) != 0 ) { return NULL; } + // and it must be an alignment THE ROOT CAN SIT AT, since the root is at + // the region's base: both are powers of two, so "at least the root's" + // is one division. + if ( ( alignment % root_align ) != 0 ) { return NULL; } + // The DATA part begins at align_up( 64, alignment ). It is DERIVED and not + // a header field, because a fact a reader computes is a fact two writers + // cannot disagree about. + const uint64_t data_offset = ( (uint64_t) kTableCookHeaderBytes + alignment - 1 ) & ~( alignment - 1 ); + if ( length < data_offset ) { return NULL; } + // the two part lengths against the length the caller passed. The whole + // file is data_offset + data_length + attribution_length, and a length + // that is not EXACTLY that refuses — truncation and trailing bytes are one + // refusal, and both terms are subtracted rather than added so no sum can + // carry. + if ( data_length > length - data_offset ) { return NULL; } + if ( attribution_length != length - data_offset - data_length ) { return NULL; } + // the ROOT sits at the region's base, so the region has to hold it: a + // shorter data part describes a root partly outside the file, which is the + // one way a match-and-point reader could hand back storage it never + // received. + if ( data_length < root_size ) { return NULL; } + const uint8_t * base = raw + data_offset; + // the alignment of the BASE. The header pads the data part to the region's + // alignment, so a base an allocator or mmap gave you is already aligned — + // mmap gives page alignment for free — and a base that is not is a caller's + // buffer this form cannot be read out of. + if ( ( (uintptr_t) base % (uintptr_t) alignment ) != 0 ) { return NULL; } + return base; +} + +// ---- the cooked form, the WRITE side (docs/SPEC-TABLES.md §7.6) ---- +// +// THE BYTE ORDER IS THE TARGET'S, NOT THE HOST'S. A cook is produced in the +// byte order of the build that will read it (§7), so the fixing happens here — +// offline, once, on the writing side — and never at Open. Passing +// TableByteOrder::Big on a little-endian machine produces a big-endian build's +// file, and nothing about the writing host reaches the bytes. +enum class TableByteOrder +{ + Little = 1, // the header's byte_order word, and the order every scalar is written in + Big = 2, +}; + +// One store, width as an argument. Every call site passes a literal width, so +// the loop folds to a store (and a byte swap on the foreign order); a name per +// width would claim four §11 names to save nothing. +inline void table_cook_put( uint8_t * at, uint64_t value, int32_t width, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * i ) ); } + } + else + { + for ( int32_t i = 0; i < width; i++ ) { at[i] = (uint8_t) ( value >> ( 8 * ( width - 1 - i ) ) ); } + } +} + +// A 128-bit store as two lanes: sixteen bytes, the low lane first in the +// little order and the high lane first — each lane big-endian — in the big +// order, exactly as a u64 is one lane of eight (docs/SPEC-TABLES.md §7.2). +inline void table_cook_put128( uint8_t * at, uint64_t lo, uint64_t hi, TableByteOrder order ) +{ + if ( order == TableByteOrder::Little ) { table_cook_put( at, lo, 8, order ); table_cook_put( at + 8, hi, 8, order ); } + else { table_cook_put( at, hi, 8, order ); table_cook_put( at + 8, lo, 8, order ); } +} + +// A buffer piece: the USED bytes and nothing else. The tail is already zero — +// the whole extent was zeroed before any field was written — so this copies the +// used prefix and leaves the rest, which is what makes a string's unused tail a +// consequence of one memset rather than a rule per buffer. A used length past +// the buffer, or below zero, is a value no reader could have produced and it is +// clamped rather than trusted: this writes inside the caller's buffer on every +// input. +inline void table_cook_bytes( uint8_t * at, const void * source, int64_t used, int64_t capacity ) +{ + if ( used <= 0 ) { return; } + const int64_t n = used < capacity ? used : capacity; + memcpy( at, source, (size_t) n ); +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK + +#ifndef LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE +#define LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// ---- the cooked form's WRITE side for a POINTERED root (docs/SPEC-TABLES.md §7.6) ---- +// +// A pointered root's cook is the region of §7.2: every node the numbering +// reached (§3.1), once, at its own type's alignment, in index order, the root +// at offset zero. This is that region while it is being laid out and written — +// the tool's own Layout and Write, in one struct. +// +// The OFFSETS are one per node, the root's zero at position 0 and node index k +// at position k - 1, which is the directory's own order (§6.3); they are the +// one allocation the write makes beyond the numbering, and they go through the +// same pair. A measure needs no offsets and leaves the pointer NULL. +struct TableCookRegion +{ + const TableNumbering * numbering = NULL; // node -> index, from the walk that placed it + int64_t * offsets = NULL; // index - 1 -> the node's region offset; NULL while measuring + int64_t count = 0; // nodes, the root included + int64_t bytes = 0; // the data part's length, rounded to align + int64_t align = 0; // the region's alignment: the nodes' greatest, never below eight + uint8_t * base = NULL; // where the data part is being written; NULL while measuring +}; + +// A reference slot: the SELF-RELATIVE delta from the slot's own address to the +// node's start (§6.3), and zero for null. The node is found by the address the +// numbering keyed it under, which is the same address the walk resolved through +// the same context — so a reference the numbering does not carry is a slot the +// walk never reached (a counted array's slot past its count, an absent +// optional's value) holding a node the region will not hold, and it is refused +// rather than written as a delta to nowhere. +inline bool table_cook_ref( const TableCookRegion & region, uint8_t * at, const void * pointee, TableByteOrder order ) +{ + if ( pointee == NULL ) { table_cook_put( at, 0, 8, order ); return true; } + uint64_t index = 0; + if ( !TableNumberingIndex( *region.numbering, pointee, index ) ) { return false; } + if ( index == 0 || index > (uint64_t) region.count ) { return false; } + const int64_t delta = region.offsets[index - 1] - (int64_t) ( at - region.base ); + table_cook_put( at, (uint64_t) delta, 8, order ); + return true; +} + +} // namespace listdemo + +#endif // LISTDEMO_SCHEMA_TABLE_COOK_VARIABLE + +namespace listdemo { + +// table Photo — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Photo { + uint32_t width = 0; + uint32_t height = 0; +}; + +// table Album — TABLE-wire storage: relocatable, bounded, defaults in the +// member initializers (docs/SPEC-TABLES.md) +struct Album { + TableList photos; // Photo: the element array, empty until an Add + TableRef cover; // *Photo — null until assigned +}; + +// ---- prefill: the declared defaults, in place (docs/SPEC-TABLES.md) ---- + +inline void PhotoReset( Photo & value ); +inline void AlbumReset( Album & value ); + +inline void PhotoReset( Photo & value ) +{ + value.width = 0; + value.height = 0; +} + +inline void AlbumReset( Album & value ) +{ + value.photos.elements.value = 0; // Photo: empty + value.photos.count = 0; + value.photos.padding = 0; + value.cover.value = 0; // *Photo — null +} + +// ---- the arena's reset hook (docs/SPEC-TABLES.md §6) ---- +// +// TableWorker::Alloc is a template and cannot name a member's Reset, so +// the arena reaches it through this overload set by argument-dependent +// lookup. It is how a node born in raw arena storage comes to hold the +// declared defaults without value-initialising the whole aggregate. + +inline void TableReset( Photo & value ) { PhotoReset( value ); } +inline void TableReset( Album & value ) { AlbumReset( value ); } + +// ---- pointer targets: allocation and resolution (docs/SPEC-TABLES.md §2) ---- +// +// A reference resolves differently in the two forms, and the CONTEXT says +// which: in the arena it is an offset; in a region it is a self-relative +// delta, so the const deref below is one add and needs no base pointer. + +// Photo is a pointer target. +inline const Photo * PhotoAt( const TableRef & ref ) // the const form's hot path: one add, no base +{ + return ref.value != 0 ? (const Photo *) ( (const uint8_t *) &ref + ref.value ) : NULL; +} +inline Photo * PhotoAt( TableRef & ref ) +{ + return ref.value != 0 ? (Photo *) ( (uint8_t *) &ref + ref.value ) : NULL; +} +inline const Photo * PhotoAt( const TableRegionCtx &, const TableRef & ref ) { return PhotoAt( ref ); } +inline const Photo * PhotoAt( const TableArenaCtx & ctx, const TableRef & ref ) +{ + return ref.value != 0 ? (const Photo *) TableArenaAt( *ctx.arena, (uint32_t) ref.value ) : NULL; +} +// while the builder is mutable, resolve against the arena itself +inline Photo * PhotoAt( TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (Photo *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} +inline const Photo * PhotoAt( const TableArena & arena, const TableRef & ref ) +{ + return ref.value != 0 ? (const Photo *) TableArenaAt( arena, (uint32_t) ref.value ) : NULL; +} +// allocate one Photo in the arena; the slot holds the arena offset +inline Photo * PhotoEmplace( TableWorker & worker, TableRef & slot ) +{ + TableSlot allocated = worker.Alloc(); + slot = allocated.ref; + return allocated.ptr; +} + +// ---- codecs: measure/save/load per closure member ---- + +inline int64_t PhotoMeasureBody( TableIds & ids, const Photo & value ); +LISTDEMO_TABLE_INLINE bool PhotoSaveBody( TableWriter & w, TableIds & ids, const Photo & value ); +LISTDEMO_TABLE_INLINE bool PhotoLoadBody( TableReader & r, Photo & value ); +template inline int64_t AlbumMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Album & value ); +template inline bool AlbumSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Album & value ); +template inline bool AlbumSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Album & value ); +inline bool AlbumLoadBody( TableReader & r, const TableNodeMap & nodes, Album & value ); + +// ---- pointer-graph walkers: number (measure/save), pack (Lock) ---- + +template inline bool PhotoNumber( const Ctx & ctx, TableNumbering & numbering, const Photo & value ); +template inline int64_t PhotoPackMeasure( const Ctx & ctx, TablePackMap & seen, const Photo & value ); +template inline bool PhotoPack( const Ctx & ctx, TablePackMap & seen, const Photo & src, Photo & dst, uint8_t * base, int64_t capacity, int64_t & used ); +template inline bool AlbumNumber( const Ctx & ctx, TableNumbering & numbering, const Album & value ); +template inline int64_t AlbumPackMeasure( const Ctx & ctx, TablePackMap & seen, const Album & value ); +template inline bool AlbumPack( const Ctx & ctx, TablePackMap & seen, const Album & src, Album & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +// ---- the numbering's bridge to each member's codec (docs/SPEC-TABLES.md §3.1) ---- + +template inline int64_t TableNodeMeasure( const Ctx &, const TableNumbering &, TableIds & ids, const Photo & value ) { return PhotoMeasureBody( ids, value ); } +template inline bool TableNodeSave( const Ctx &, const TableNumbering &, TableWriter & w, TableIds & ids, const Photo & value ) { return PhotoSaveBody( w, ids, value ); } +template inline int64_t TableNodeMeasure( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Album & value ) { return AlbumMeasureBody( ctx, numbering, ids, value ); } +template inline bool TableNodeSave( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Album & value ) { return AlbumSaveBody( ctx, numbering, w, ids, value ); } + +inline int64_t PhotoMeasureBody( TableIds & ids, const Photo & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + if ( value.width != 0 ) { bytes += TableLebBytes( ids.ref( 0xdbdacd932fd1e9bfull, 17 ) ) + 1 + 4; } // width + if ( value.height != 0 ) { bytes += TableLebBytes( ids.ref( 0x17720bf67d347222ull, 18 ) ) + 1 + 4; } // height + return bytes; +} + +inline int64_t PhotoMeasure( const Photo & value ) +{ + TableIds ids; + const int64_t body = PhotoMeasureBody( ids, value ); + if ( body < 0 || ids.overflow ) { return -1; } + return 1 + body + TableIdsBytes( ids ); +} + +LISTDEMO_TABLE_INLINE bool PhotoSaveBody( TableWriter & w, TableIds & ids, const Photo & value ) +{ + if ( value.width != 0 ) + { + w.putleb( ids.ref( 0xdbdacd932fd1e9bfull, 17 ) ); w.put8( 8 ); // width + w.put32( uint32_t( value.width ) ); + } + if ( value.height != 0 ) + { + w.putleb( ids.ref( 0x17720bf67d347222ull, 18 ) ); w.put8( 8 ); // height + w.put32( uint32_t( value.height ) ); + } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline int64_t PhotoSave( const Photo & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + w.put8( kTableWireForm ); // the FORM BYTE is the whole header (§3) + if ( !PhotoSaveBody( w, ids, value ) || ids.overflow ) { return -1; } + TableIdsWrite( w, ids ); // the ID TABLE is the last thing in the file + if ( w.overflow ) { return -1; } + return w.offset; // == PhotoMeasure( value ) +} + +LISTDEMO_TABLE_INLINE bool PhotoLoadBody( TableReader & r, Photo & value ) +{ + PhotoReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0xdbdacd932fd1e9bfull: // width + { + if ( kind != 8 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + uint32_t decoded_v = uint32_t( r.get32( ) ); + value.width = decoded_v; + break; + } + case 0x17720bf67d347222ull: // height + { + if ( kind != 8 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + if ( !r.has( 4 ) ) { r.report->malformed = true; return false; } + uint32_t decoded_v = uint32_t( r.get32( ) ); + value.height = decoded_v; + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// THE FORM BYTE IS READ FIRST, before the trailer and before any body, so a +// file that is both a newer form and damaged is a REFUSAL and never damage. +// A refusal moves none of the report's five counters, because nothing was +// decoded and there is nothing to count (docs/SPEC-TABLES.md §3). +inline TableOpenVerdict PhotoLoadVerdict( Photo & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + TableIdTable table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( buffer, bytes, table, body_bytes ); + if ( verdict != TableOpenOk ) + { + PhotoReset( value ); + if ( verdict == TableOpenDamaged ) { to->malformed = true; } + else + { + // FORM 2 IS A STREAM FORM AND NEVER A FILE FORM: a message + // stored on its own is not readable, because its table is + // somewhere else, and the refusal says so BY NAME rather than + // merely by form byte (docs/SPEC-TABLES.md §3.3). + to->refused = true; + to->reason = bytes > 0 && buffer[0] == kTableWireMessageForm ? message_form_as_file : newer_form; + } + return verdict; + } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY IS + // MALFORMED, because no field claims it and the two ends of the file + // have met: nothing is decoded and one event is counted (§3). + if ( TableBodyEndsEarly( buffer + 1, body_bytes, table ) ) + { + PhotoReset( value ); + to->malformed = true; + return TableOpenDamaged; + } + TableReader r( buffer + 1, body_bytes, to, &table ); + r.nested = false; // the ROOT body, the one that may carry a node table + if ( !PhotoLoadBody( r, value ) ) { return TableOpenBodyStopped; } + return TableOpenOk; +} + +// The bool is the BODY reaching its own terminator, and it is what it has +// always been: framing damage inside a field keeps what it decoded, flags +// the report and reads on, so this answers true (docs/SPEC-TABLES.md §4). +// False is a wire nothing could be decoded from: a refusal, a table that +// cannot be read whole, or a root body the walk could not finish. +inline bool PhotoLoad( Photo & value, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + return PhotoLoadVerdict( value, buffer, bytes, report ) == TableOpenOk; +} + +// The MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the form byte and the root +// body, and no trailer at all — the connection's announced table is where +// the ids live. Every reference is a compile-time SLOT, so this walk does +// no lookup and a save costs what a save costs. +inline int64_t PhotoMeasureMessage( const Photo & value ) +{ + TableIds ids; + ids.vocabulary = true; + const int64_t body = PhotoMeasureBody( ids, value ); + if ( body < 0 ) { return -1; } + return 1 + body; +} + +inline int64_t PhotoSaveMessage( const Photo & value, uint8_t * buffer, int64_t capacity ) +{ + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = true; + w.put8( kTableWireMessageForm ); + if ( !PhotoSaveBody( w, ids, value ) || w.overflow ) { return -1; } + return w.offset; // == PhotoMeasureMessage( value ) +} + +// A form 2 message with NO TABLE for the connection is REFUSED BY NAME: +// nothing is decoded, the reader says it holds no table, no counter moves +// and malformed does not fire. A reader does not fall back to the file +// form on its own and does not guess a table, because a guessed table +// decodes a body under the wrong names in silence. +inline bool PhotoLoadMessage( Photo & value, const TableVocabulary & vocabulary, const uint8_t * buffer, int64_t bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * to = report != NULL ? report : &ignored; + PhotoReset( value ); + if ( bytes < 1 ) { to->malformed = true; return false; } + if ( buffer[0] != kTableWireMessageForm ) { to->refused = true; to->reason = newer_form; return false; } + if ( !vocabulary.announced ) { to->refused = true; to->reason = no_vocabulary; return false; } + // a form 2 wire has NO TRAILER, so the body runs to the last byte and + // there is no stray-byte rule to apply between one and a first entry + TableReader r( buffer + 1, bytes - 1, to, &vocabulary.table ); + r.nested = false; // the ROOT body, the one that may carry a node table + return PhotoLoadBody( r, value ); +} + +template +inline int64_t AlbumMeasureBody( const Ctx & ctx, const TableNumbering & numbering, TableIds & ids, const Album & value ) +{ + int64_t bytes = 1; // the ZERO REFERENCE that ends the body + { + // photos: a kind 14 array of kind 17 elements, INDEX order (§2.9) + TableListCursor cursor_photos = TableListElements( ctx, value.photos ); + if ( !cursor_photos.ok ) { return -1; } // the slot and the head disagree + if ( cursor_photos.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_photos = ids.ref( 0x40b1d94aff3ab130ull, 2 ); + int64_t body_photos = 0; + body_photos += 1 + TableLebBytes( (uint64_t) ( cursor_photos.count ) ); // the element kind byte and the count + for ( int32_t elem_i_photos = 0; elem_i_photos < cursor_photos.count; elem_i_photos++ ) + { + { + const Photo * slot_pointee_photos = PhotoAt( ctx, cursor_photos[elem_i_photos] ); + uint64_t slot_index_photos = 0; + if ( slot_pointee_photos != NULL && !TableNumberingIndex( numbering, (const void *) slot_pointee_photos, slot_index_photos ) ) { return -1; } + body_photos += TableLebBytes( slot_index_photos ); + } + } + bytes += TableLebBytes( ref_photos ) + 1 + TableLebBytes( (uint64_t) ( body_photos ) ) + ( body_photos ); + } + } + { + const Photo * pointee_cover = PhotoAt( ctx, value.cover ); // *Photo + // A POINTER RIDES AS A NODE INDEX (docs/SPEC-TABLES.md §3.1): the + // header and the index and nothing below it, because the pointee's + // body is in the node table and not here. NULL IS ELIDED — absence + // and null are one value — and a non-null pointer ALWAYS rides, even + // when its node's body is entirely default. + if ( pointee_cover != NULL ) + { + uint64_t index_cover = 0; + if ( !TableNumberingIndex( numbering, (const void *) pointee_cover, index_cover ) ) { return -1; } + bytes += TableLebBytes( ids.ref( 0xaa19a78e404dea20ull, 3 ) ) + 1 + TableLebBytes( index_cover ); + } + } + return bytes; +} + +template +inline bool AlbumSaveBodyFields( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Album & value ) +{ + { + TableListCursor cursor_photos = TableListElements( ctx, value.photos ); // photos + if ( !cursor_photos.ok ) { return false; } + if ( cursor_photos.count > 0 ) // an EMPTY list elides, the by-value rule (§3) + { + const uint64_t ref_photos = ids.ref( 0x40b1d94aff3ab130ull, 2 ); + int64_t body_photos = 0; + body_photos += 1 + TableLebBytes( (uint64_t) ( cursor_photos.count ) ); // the element kind byte and the count + for ( int32_t elem_i_photos = 0; elem_i_photos < cursor_photos.count; elem_i_photos++ ) + { + { + const Photo * slot_pointee_photos = PhotoAt( ctx, cursor_photos[elem_i_photos] ); + uint64_t slot_index_photos = 0; + if ( slot_pointee_photos != NULL && !TableNumberingIndex( numbering, (const void *) slot_pointee_photos, slot_index_photos ) ) { return false; } + body_photos += TableLebBytes( slot_index_photos ); + } + } + w.putleb( ref_photos ); w.put8( 14 ); w.putleb( (uint64_t) body_photos ); // photos + w.put8( 17 ); w.putleb( (uint64_t) ( cursor_photos.count ) ); + for ( int32_t elem_i_photos = 0; elem_i_photos < cursor_photos.count; elem_i_photos++ ) + { + { + const Photo * slot_pointee_photos = PhotoAt( ctx, cursor_photos[elem_i_photos] ); + uint64_t slot_index_photos = 0; + if ( slot_pointee_photos != NULL && !TableNumberingIndex( numbering, (const void *) slot_pointee_photos, slot_index_photos ) ) { return false; } + w.putleb( slot_index_photos ); + } + } + } + } + { + const Photo * pointee_cover = PhotoAt( ctx, value.cover ); // *Photo + if ( pointee_cover != NULL ) + { + uint64_t index_cover = 0; + if ( !TableNumberingIndex( numbering, (const void *) pointee_cover, index_cover ) ) { return false; } + w.putleb( ids.ref( 0xaa19a78e404dea20ull, 3 ) ); w.put8( 17 ); // cover — a NODE INDEX into the flat node table + w.putleb( index_cover ); + } + } + return !w.overflow; +} + +template +inline bool AlbumSaveBody( const Ctx & ctx, const TableNumbering & numbering, TableWriter & w, TableIds & ids, const Album & value ) +{ + if ( !AlbumSaveBodyFields( ctx, numbering, w, ids, value ) ) { return false; } + w.put8( 0 ); // the ZERO REFERENCE that ends the body + return !w.overflow; +} + +inline bool AlbumLoadBody( TableReader & r, const TableNodeMap & nodes, Album & value ) +{ + AlbumReset( value ); // prefill declared defaults in place, then overlay + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { r.report->malformed = true; return false; } + if ( field_ref == 0 ) return true; // the body ENDS AT ITS OWN ZERO REFERENCE + if ( r.ids == NULL || field_ref > (uint64_t) r.ids->count ) { r.report->malformed = true; return false; } // a reference ABOVE the entry count + const uint64_t field_id = r.ids->at( field_ref ); + if ( !r.has( 1 ) ) { r.report->malformed = true; return false; } + uint8_t kind = r.get8(); + if ( ( field_id == kTableNodeTableFieldId && r.nested ) || field_id == kTableBuildVersionFieldId ) + { + // A RESERVED ID IN ANY BODY BUT THE ONE WHOSE TRANSPORT IT IS, + // IS MALFORMED (docs/SPEC-TABLES.md §3.1, §3.3). The node + // table's is the ROOT body's alone, on the numbering's own + // rule — a second numbering cannot exist — and the BUILD + // VERSION's rides in the announcement and nowhere else. That + // body stops and the parent reads on past its L. + r.report->malformed = true; + return false; + } + switch ( field_id ) + { + case 0x40b1d94aff3ab130ull: // photos + { + if ( kind != 14 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + uint64_t body_len = 0; + if ( !r.getleb( body_len ) || !r.room( body_len ) ) { r.report->malformed = true; return false; } + int64_t body_end = r.offset + (int64_t) body_len; + // A BODY TOO SHORT FOR ITS OWN HEADER is INERT (§4): the field keeps + // the value it has, no counter is raised, and the walk continues past L. + if ( body_len >= 2 ) + { + uint8_t elem_kind = r.get8(); + uint64_t count = 0; + const bool counted_ok = r.getleb( count ); + if ( !counted_ok ) { r.report->malformed = true; } + // AN ELEMENT KIND THAT DISAGREES with the reader's declaration is §3's + // element-kind rule: the field reads EMPTY and one kind_mismatch counts + else if ( elem_kind != 17 ) { r.report->kind_mismatch++; r.offset = body_end; break; } + else + { + // THE COUNT IS THE DATA'S (§2.9): there is no bound, so clamped + // cannot fire on it. A count above the int32 storage cap is the + // fill's refusal, and it moves no counter. + TableListFill fill = TableListFillBegin( nodes, value.photos, count ); + if ( fill.refused ) { nodes.refused = true; return false; } + if ( !fill.ok ) { r.report->malformed = true; r.offset = body_end; break; } + // elements are BOUNDED by the field body: a count the length cannot + // cover keeps the decoded prefix, flags malformed, and the parent + // continues at the next field + TableReader sub( r.buffer + r.offset, body_end - r.offset, r.report, r.ids ); + for ( uint64_t i = 0; i < count; i++ ) + { + TableRef * slot = TableListFillNext( fill ); + if ( slot == NULL ) { r.report->malformed = true; break; } // the arena could not carve + bool landed = false; + do + { + { + uint64_t node_index_photos = 0; + if ( !sub.getleb( node_index_photos ) ) { r.report->malformed = true; break; } + TableNodeResolve( nodes, ( *slot ), node_index_photos, 0xf1a78dd2508964c3ull, r.report ); // *Photo + } + landed = true; + } while ( 0 ); + if ( !landed ) { TableListFillDrop( fill ); break; } // the element's own framing gave out before it decoded + } + TableListFillEnd( fill ); + } + } + r.offset = body_end; // excess bytes and slack skip via the length + break; + } + case 0xaa19a78e404dea20ull: // cover + { + if ( kind != 17 ) + { + // AT A POSITION THE READER DOES NAME, a field under + // kind 31 or kind 32 takes this same rule and no other (§3) + r.report->kind_mismatch++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + // A POINTER FIELD'S PAYLOAD IS A NUMBER (docs/SPEC-TABLES.md §3.1): it is + // bounds-checked and resolved through the numbering, never FOLLOWED, so + // there is no traversal here and therefore no traversal bound. + { + uint64_t node_index = 0; + if ( !r.getleb( node_index ) ) { r.report->malformed = true; return false; } + TableNodeResolve( nodes, value.cover, node_index, 0xf1a78dd2508964c3ull, r.report ); // *Photo + } + break; + } + case 0xffffffffffffffffull: + { + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + default: + { + r.report->unknown++; + if ( !r.skip( kind ) ) { r.report->malformed = true; return false; } + break; + } + } + } +} + +// PhotoWireExtent: the extent Photo's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool PhotoWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + (void) body; (void) length; (void) at; (void) ids; (void) reason; // no list or map below this record + return true; +} + +// PhotoExtentAt: the node extent Photo's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as PhotoExtentPack advances it (§2.8, §2.9). +template +inline bool PhotoExtentAt( const Ctx & ctx, const Photo & value, int64_t & at ) +{ + (void) ctx; (void) value; (void) at; // no list or map below this record + return true; +} + +// PhotoExtentPack: carve Photo's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset PhotoExtentAt advances (§2.8, §2.9). +template +inline bool PhotoExtentPack( const Ctx & ctx, const Photo & src, Photo & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no list or map below this record + return true; +} + +// AlbumWireExtent: the extent Album's lists and maps command, from the FRAMING alone. +// It reads no field value, so a caller can refuse a number it did not +// expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). +inline bool AlbumWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; // the scan's framing damage is the LOAD's to report + TableReader r( body, length, &scratch, ids ); + for ( ;; ) + { + uint64_t field_ref = 0; + if ( !r.getleb( field_ref ) ) { return true; } + if ( field_ref == 0 ) { return true; } + if ( ids == NULL || field_ref > (uint64_t) ids->count ) { return true; } + const uint64_t field_id = ids->at( field_ref ); + if ( !r.has( 1 ) ) { return true; } + uint8_t field_kind = r.get8(); + if ( field_id == 0x40b1d94aff3ab130ull && field_kind == 14 ) // photos: an unbounded array + { + uint64_t list_len = 0; + if ( !r.getleb( list_len ) || !r.room( list_len ) ) { return true; } + const uint8_t * list_body = r.buffer + r.offset; + r.offset += (int64_t) list_len; + if ( !TableListWireExtent( list_body, (int64_t) list_len, at, (int64_t) sizeof( TableRef ), (int64_t) alignof( TableRef ), 17, 1, NULL, ids, reason ) ) { return false; } + continue; + } + if ( !r.skip( field_kind ) ) { return true; } + } +} + +// AlbumExtentAt: the node extent Album's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as AlbumExtentPack advances it (§2.8, §2.9). +template +inline bool AlbumExtentAt( const Ctx & ctx, const Album & value, int64_t & at ) +{ + { + TableListCursor cursor = TableListElements( ctx, value.photos ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( TableRef ) - 1 ) & ~( (int64_t) alignof( TableRef ) - 1 ); + at += (int64_t) cursor.count * (int64_t) sizeof( TableRef ); // the whole array FIRST + } + return true; +} + +// the whole extent of one node, from a fresh offset: what a pack reserves +// for it beside the record's own storage. +template +inline int64_t AlbumExtent( const Ctx & ctx, const Album & value ) +{ + int64_t at = 0; + if ( !AlbumExtentAt( ctx, value, at ) ) { return -1; } + return at; +} + +// AlbumExtentPack: carve Album's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset AlbumExtentAt advances (§2.8, §2.9). +template +inline bool AlbumExtentPack( const Ctx & ctx, const Album & src, Album & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +{ + { + TableListCursor cursor = TableListElements( ctx, src.photos ); + if ( !cursor.ok ) { return false; } + at = ( at + (int64_t) alignof( TableRef ) - 1 ) & ~( (int64_t) alignof( TableRef ) - 1 ); + const int64_t bytes = (int64_t) cursor.count * (int64_t) sizeof( TableRef ); + if ( at + bytes > capacity ) { return false; } + TableRef * placed = (TableRef *) ( extent + at ); + at += bytes; + dst.photos.count = cursor.count; + dst.photos.padding = 0; + dst.photos.elements.value = cursor.count > 0 ? (int64_t) ( (uint8_t *) placed - (const uint8_t *) &dst.photos.elements ) : 0; + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + memcpy( (void *) ( placed + i ), (const void *) &cursor[i], sizeof( TableRef ) ); // trivially copyable, by construction + } + } + return true; +} + +// ---- Album.photos: the builder's three (§2.9) ---- + +// ADD: the element is appended and handed back to fill. On a []*T that is +// the SLOT at null, which PhotoEmplace fills as it fills any pointer slot, +// and a second slot may hold the same reference: two slots, one node. +// NULL means NOT ADDED: an arena that cannot carve another segment, or a +// count at the int32 cap. A caller that needs the reason checks size(). +inline TableRef * AlbumPhotosAdd( TableWorker & worker, TableList & list ) +{ + return TableListPlace( worker, list ); +} + +// ERASE, by the element's own pointer: marks it DEAD, one bit in the +// segment's slot and not in the element storage. False when the pointer is +// not this list's. Storage is held until the builder resets. INDICES ARE +// NOT STABLE ACROSS AN ERASE: what was index 3 is index 2 in the next Save. +inline bool AlbumPhotosErase( TableArena & arena, TableList & list, const TableRef * element ) +{ + return TableListErase( arena, list, element ); +} + +// EACH on the builder: INDEX order, live elements only, yielding the +// element Add handed back. +inline TableListEach AlbumPhotosEach( const TableArena & arena, const TableList & list ) +{ + return TableListEachOf( arena, list ); +} + +// PhotoNumber: number everything Photo POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool PhotoNumber( const Ctx & ctx, TableNumbering & numbering, const Photo & value ) +{ + (void) ctx; (void) numbering; (void) value; // no pointers below this node + return true; +} + +// PhotoPackMeasure: the packed region bytes of everything Photo POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t PhotoPackMeasure( const Ctx & ctx, TablePackMap & seen, const Photo & value ) +{ + int64_t bytes = 0; + (void) ctx; (void) seen; (void) value; // no pointers below this node + return bytes; +} + +// PhotoPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool PhotoPackEdges( const Ctx & ctx, TablePackMap & seen, const Photo & src, Photo & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool PhotoPack( const Ctx & ctx, TablePackMap & seen, const Photo & src, Photo & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Photo ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Photo ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !PhotoExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return PhotoPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool PhotoPackEdges( const Ctx & ctx, TablePackMap & seen, const Photo & src, Photo & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + (void) ctx; (void) seen; (void) src; (void) dst; (void) base; (void) capacity; (void) used; + return true; +} + +// AlbumNumber: number everything Album POINTS AT, in first-visit order — +// the fields in declaration order, a by-value edge descended in place. +// A reference to an entry whose descent is still OPEN is a data cycle, +// named here rather than recursed away (docs/SPEC-TABLES.md §3.1). +template +inline bool AlbumNumber( const Ctx & ctx, TableNumbering & numbering, const Album & value ) +{ + { // photos: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_photos = TableListElements( ctx, value.photos ); + if ( !cursor_photos.ok ) { return false; } + for ( int32_t i = 0; i < cursor_photos.count; i++ ) + { + { + const Photo * pointee = PhotoAt( ctx, cursor_photos[i] ); // photos + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( numbering.seen, (const void *) pointee, + (int64_t) ( numbering.count + 2 ), taken, slot ); // its index, if this is its first visit + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + } + else + { + TableNodeEntry node; + node.node = (const void *) pointee; + node.type_id = 0xf1a78dd2508964c3ull; // fnv1a64( "Photo" ) + node.type_slot = 52; // its slot in the unit's vocabulary (§3.3) + node.measure = &TableNodeMeasureThunk; + node.save = &TableNodeSaveThunk; + if ( !TableNumberingAppend( numbering, node ) ) { return false; } + if ( !PhotoNumber( ctx, numbering, *pointee ) ) { return false; } + TablePackMapClose( numbering.seen, (const void *) pointee, slot ); + } + } + } + } + } + { + const Photo * pointee = PhotoAt( ctx, value.cover ); // cover + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( numbering.seen, (const void *) pointee, + (int64_t) ( numbering.count + 2 ), taken, slot ); // its index, if this is its first visit + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + } + else + { + TableNodeEntry node; + node.node = (const void *) pointee; + node.type_id = 0xf1a78dd2508964c3ull; // fnv1a64( "Photo" ) + node.type_slot = 52; // its slot in the unit's vocabulary (§3.3) + node.measure = &TableNodeMeasureThunk; + node.save = &TableNodeSaveThunk; + if ( !TableNumberingAppend( numbering, node ) ) { return false; } + if ( !PhotoNumber( ctx, numbering, *pointee ) ) { return false; } + TablePackMapClose( numbering.seen, (const void *) pointee, slot ); + } + } + } + return true; +} + +// AlbumPackMeasure: the packed region bytes of everything Album POINTS AT. +// ONE VISIT PER NODE: `seen` carries the first-visit numbering (§3.1), so a +// node two references name is measured ONCE and packed once, and a +// reference to a node whose descent is still open is a data cycle, refused. +template +inline int64_t AlbumPackMeasure( const Ctx & ctx, TablePackMap & seen, const Album & value ) +{ + int64_t bytes = 0; + { // photos: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_photos = TableListElements( ctx, value.photos ); + if ( !cursor_photos.ok ) { return -1; } + for ( int32_t i = 0; i < cursor_photos.count; i++ ) + { + { + const Photo * pointee = PhotoAt( ctx, cursor_photos[i] ); // photos + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, 0, taken, slot ); + if ( entry == NULL ) { return -1; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return -1; } // a data cycle + } + else + { + int64_t inner = PhotoPackMeasure( ctx, seen, *pointee ); + if ( inner < 0 ) { return -1; } + TablePackMapClose( seen, (const void *) pointee, slot ); + bytes += TableAlignUp64( (int64_t) sizeof( Photo ) ) + inner; + } + } + } + } + } + { + const Photo * pointee = PhotoAt( ctx, value.cover ); // cover + if ( pointee != NULL ) + { + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, 0, taken, slot ); + if ( entry == NULL ) { return -1; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return -1; } // a data cycle + } + else + { + int64_t inner = PhotoPackMeasure( ctx, seen, *pointee ); + if ( inner < 0 ) { return -1; } + TablePackMapClose( seen, (const void *) pointee, slot ); + bytes += TableAlignUp64( (int64_t) sizeof( Photo ) ) + inner; + } + } + } + return bytes; +} + +// AlbumPack: copy src into dst (already placed), then lay every pointee out +// depth-first behind it, in FIELD ORDER, by bump allocation. +// +// ONE NODE, ONE BODY (§6.2): `seen` holds every node already placed and +// where it landed, so a node's FIRST reference lays it out and every later +// reference points BACK at that one body. A region delta therefore has no +// required sign (§6.3), and sharing and a back-reference are one fact. A +// reference to a node whose descent is still OPEN is a cycle, and this +// refuses it rather than packing one. +template +inline bool AlbumPackEdges( const Ctx & ctx, TablePackMap & seen, const Album & src, Album & dst, uint8_t * base, int64_t capacity, int64_t & used ); + +template +inline bool AlbumPack( const Ctx & ctx, TablePackMap & seen, const Album & src, Album & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + memcpy( (void *) &dst, (const void *) &src, sizeof( Album ) ); // trivially copyable, by construction + int64_t at = 0; + uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Album ) ); + const int64_t room = capacity - ( (int64_t) ( extent - base ) ); + if ( !AlbumExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } + return AlbumPackEdges( ctx, seen, src, dst, base, capacity, used ); +} + +template +inline bool AlbumPackEdges( const Ctx & ctx, TablePackMap & seen, const Album & src, Album & dst, uint8_t * base, int64_t capacity, int64_t & used ) +{ + { // photos: a by-value edge, elements in INDEX order (§2.9, §3.1) + TableListCursor cursor_photos = TableListElements( ctx, src.photos ); + if ( !cursor_photos.ok ) { return false; } + TableRef * placed_photos = (TableRef *) ( dst.photos.elements.value != 0 ? ( (uint8_t *) &dst.photos.elements + dst.photos.elements.value ) : NULL ); + for ( int32_t i = 0; i < cursor_photos.count; i++ ) + { + { + placed_photos[i].value = 0; // photos + const Photo * pointee = PhotoAt( ctx, cursor_photos[i] ); + if ( pointee != NULL ) + { + int64_t at = TableAlignUp64( used ); // where it WOULD land, if this is its first visit + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, at, taken, slot ); + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + placed_photos[i].value = (int64_t) ( ( base + entry->offset ) - (const uint8_t *) &placed_photos[i] ); // the one body it already has + } + else + { + if ( at + (int64_t) sizeof( Photo ) > capacity ) { return false; } + used = at + TableAlignUp64( (int64_t) sizeof( Photo ) ); + Photo * child = new ( base + at ) Photo; // lifetime only: the Pack below memcpy's the whole node over it + placed_photos[i].value = (int64_t) ( ( base + at ) - (const uint8_t *) &placed_photos[i] ); + if ( !PhotoPack( ctx, seen, *pointee, *child, base, capacity, used ) ) { return false; } + TablePackMapClose( seen, (const void *) pointee, slot ); + } + } + } + } + } + { + dst.cover.value = 0; // cover + const Photo * pointee = PhotoAt( ctx, src.cover ); + if ( pointee != NULL ) + { + int64_t at = TableAlignUp64( used ); // where it WOULD land, if this is its first visit + bool taken = false; + int64_t slot = 0; + const TablePackEntry * entry = TablePackMapReach( seen, (const void *) pointee, at, taken, slot ); + if ( entry == NULL ) { return false; } // the map could not grow + if ( !taken ) + { + if ( entry->open != 0 ) { return false; } // a data cycle + dst.cover.value = (int64_t) ( ( base + entry->offset ) - (const uint8_t *) &dst.cover ); // the one body it already has + } + else + { + if ( at + (int64_t) sizeof( Photo ) > capacity ) { return false; } + used = at + TableAlignUp64( (int64_t) sizeof( Photo ) ); + Photo * child = new ( base + at ) Photo; // lifetime only: the Pack below memcpy's the whole node over it + dst.cover.value = (int64_t) ( ( base + at ) - (const uint8_t *) &dst.cover ); + if ( !PhotoPack( ctx, seen, *pointee, *child, base, capacity, used ) ) { return false; } + TablePackMapClose( seen, (const void *) pointee, slot ); + } + } + } + return true; +} + +// ---- Album: the variable-length life (docs/SPEC-TABLES.md §2, §6, §9) ---- +// +// MUTABLE: AlbumBuilder — allocate nodes, wire them together, then Lock. +// CONST: one packed region, root at its base. Lock produces it and Load +// produces it, so a locked structure and a loaded one are the +// SAME representation with one view API. There is no unlock: +// re-editing means loading the const form into a fresh builder. +// Album is never held by value — a file-format-scale structure is a region +// and a root pointer, not a struct you copy. + +struct AlbumBuilder +{ + TableArena arena; + TableWorker main; // the calling thread's allocation front + TableRef root_ref; + uint8_t * region = NULL; // the packed const form, produced by Lock() + int64_t region_bytes = 0; + + // THE ALLOCATOR IS THE BUILDER'S, and everything this structure ever + // allocates goes through it: the arena's segments, Lock's identity map, + // the packed region, the wire walks' numbering, and the tool path's node + // directory. Name your own and a profiler sees every byte under it. + AlbumBuilder( TableAllocator allocator = TableDefaultAllocator() ) + { + TableArenaInit( arena, allocator ); + main.arena = &arena; + TableSlot slot = main.Alloc(); + root_ref = slot.ref; + } + ~AlbumBuilder() { TableArenaShutdown( arena ); arena.allocator.free( arena.allocator.context, region ); } + AlbumBuilder( const AlbumBuilder & ) = delete; + AlbumBuilder & operator=( const AlbumBuilder & ) = delete; + + // Alloc a node in THIS thread's slab: no lock, no atomic per node. + // The result is usable both as the node pointer and as the reference + // to store in a pointer field. + template TableSlot Alloc() { return main.Alloc(); } + // a BYTE BUFFER's node of exactly `length` bytes (docs/SPEC-TABLES.md §2.5): + // the bytes to write through, and the reference to store in a *bytes + // or *string slot; a blob past a slab takes a span of its own + TableBytesSlot AllocBytes( int64_t length ) { return main.AllocBytes( length ); } + TableStringSlot AllocString( int64_t length ) { return main.AllocString( length ); } + // one worker per thread; allocate on your own, and synchronize your own + // writes to nodes another worker allocated + TableWorker Worker() { TableWorker worker; worker.arena = &arena; return worker; } + + // GetRoot/AsConst, not Root/Const: a member function hides the type + // name it shares, and `table Root` is this spec's own canonical + // example. The checker refuses a table named after any member here, + // so the remaining spellings cannot collide either. + Album * GetRoot() { return arena.locked ? NULL : (Album *) TableArenaAt( arena, (uint32_t) root_ref.value ); } + bool Locked() const { return arena.locked; } + const Album * AsConst() const { return (const Album *) region; } + const uint8_t * Region() const { return region; } + int64_t RegionBytes() const { return region_bytes; } + + // Lock is ONE WAY and it is the compaction: the segmented arena becomes + // one exact-packed region with zero slack, references rewritten + // self-relative, and the mutable life released. Single-threaded: call + // it after the workers have joined. + bool Lock(); +}; + +inline bool AlbumBuilder::Lock() +{ + if ( arena.locked ) { return region != NULL; } + if ( root_ref.null() ) { return false; } + TableArenaCtx ctx = { &arena }; + const Album & root = *(const Album *) TableArenaAt( arena, (uint32_t) root_ref.value ); + // The ROOT takes the map's first entry: it is packed at offset 0, and its + // descent is open for the whole walk (docs/SPEC-TABLES.md §3.1). + TablePackMap seen; + TablePackMapInit( seen, arena.allocator ); + bool root_taken = false; + int64_t root_slot = 0; + int64_t below = -1; + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) != NULL ) + { + below = AlbumPackMeasure( ctx, seen, root ); + } + if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it + int64_t root_extent = AlbumExtent( ctx, root ); + if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run + int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ) + below; + // the AUTHORING path may allocate (§6.5), and it does so through the + // builder's own pair. The region comes back ZEROED, which is the + // allocator's contract: a packed region carries node padding. + uint8_t * packed = (uint8_t *) arena.allocator.alloc( arena.allocator.context, total ); + if ( packed == NULL ) { TablePackMapShutdown( seen ); return false; } + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + Album * destination = new ( packed ) Album; // lifetime only: the Pack below memcpy's the whole node over it + // The pack walk RE-DERIVES the same numbering rather than carrying the + // measure's — nothing passes between them, which is what makes + // `used == total` below a real check and not a tautology (§3.1). The + // map keeps the capacity the measure paid for, so the second walk + // rehashes nothing. + TablePackMapReset( seen ); + if ( TablePackMapReach( seen, (const void *) &root, 0, root_taken, root_slot ) == NULL || + !AlbumPack( ctx, seen, root, *destination, packed, total, used ) || used != total ) + { + TablePackMapShutdown( seen ); + arena.allocator.free( arena.allocator.context, packed ); + return false; + } + TablePackMapShutdown( seen ); + region = packed; + region_bytes = total; + arena.locked = true; // MONOTONIC: there is no unlock + TableArenaShutdown( arena ); + return true; +} + +// ---- Album on the wire: the FLAT NODE TABLE (docs/SPEC-TABLES.md §3.1) ---- +// +// A pointered save writes every reachable node ONCE, into a node table under +// the reserved id 0xFFFF, and a pointer field rides as a u32 INDEX into it +// under kind 17. No pointer edge is a nesting level, so a chain's length is +// not a depth and two references to one node are one node. + +// AlbumNodeStorage: the region bytes one record commands, or -1 for a type id +// this build cannot name — which keeps its index and reads null. A BYTE +// BUFFER's record commands its header and its bytes (docs/SPEC-TABLES.md §2.5), +// which is the one answer the record's LENGTH decides. +// A MAP'S ENTRIES RIDE IN THEIR HOLDER'S EXTENT (docs/SPEC-TABLES.md §2.8), +// so a record's storage is its type's PLUS N x sizeof( Entry ) at every +// depth, summed from the FRAMING: N is framing and not a value, and this +// reads no field. kTableNodeRefused is a wire whose N its L cannot carry. +inline int64_t AlbumNodeStorage( uint64_t type_id, int64_t length ) +{ + (void) length; // no byte buffer below this root: every node's storage is its type's + switch ( type_id ) + { + case 0xf1a78dd2508964c3ull: return TableAlignUp64( (int64_t) sizeof( Photo ) ); // Photo + default: break; + } + return -1; +} + +// AlbumNodePlace: start one record's node's lifetime in the storage pass one +// reserved for it, holding exactly the declared defaults — a byte buffer's +// header holds its length, and its bytes come in pass two. +inline void AlbumNodePlace( uint64_t type_id, uint8_t * at, int64_t length ) +{ + (void) length; + switch ( type_id ) + { + case 0xf1a78dd2508964c3ull: { Photo * node = new ( at ) Photo; PhotoReset( *node ); break; } // Photo + default: break; + } +} + +// AlbumNodeRecordBytes: one record's OWN storage, before the extent its maps +// take (docs/SPEC-TABLES.md §2.8) — where a node's extent begins. +inline int64_t AlbumNodeRecordBytes( uint64_t type_id ) +{ + switch ( type_id ) + { + case 0xf1a78dd2508964c3ull: return TableAlignUp64( (int64_t) sizeof( Photo ) ); // Photo + default: break; + } + return 0; +} + +// AlbumNodeAlloc: the TOOL's path — one record's node in the builder's arena. +// Zero is the arena's null, and it is also what a type id this build cannot +// name answers. +inline uint32_t AlbumNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t length ) +{ + (void) length; + switch ( type_id ) + { + case 0xf1a78dd2508964c3ull: return (uint32_t) worker.Alloc().ref.value; // Photo + default: break; + } + return 0; +} + +// AlbumNodeBody: PASS TWO's half — decode one record's body into the storage it +// already owns. +inline void AlbumNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) +{ + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; + carve.worker = nodes.worker; + if ( carve.worker == NULL ) + { + const int64_t storage = AlbumNodeStorage( type_id, r.size ); + const int64_t record = storage > 0 ? AlbumNodeRecordBytes( type_id ) : 0; + carve.at = at + record; + carve.left = storage > record ? storage - record : 0; + } + nodes.carve = &carve; + (void) nodes; // every node this root can name is a FIXED table + switch ( type_id ) + { + case 0xf1a78dd2508964c3ull: PhotoLoadBody( r, *(Photo *) at ); break; // Photo + default: break; + } + nodes.carve = NULL; // the cursor is ONE node's, and this node's body is done +} + +// The numbering both wire walks derive, and NEITHER CARRIES THE OTHER'S: the +// root takes index 1 and its entry stays open for the whole walk, so a +// reference back at it is the cycle it is (§3.1). +template +inline bool AlbumNumberFrom( const Ctx & ctx, TableNumbering & numbering, const Album & root ) +{ + bool taken = false; + int64_t slot = 0; + if ( TablePackMapReach( numbering.seen, (const void *) &root, (int64_t) kTableNodeIndexRoot, taken, slot ) == NULL ) { return false; } + return AlbumNumber( ctx, numbering, root ); +} + +// `message` selects the MESSAGE FORM (docs/SPEC-TABLES.md §3.3): the same +// walk over the same graph, with every reference a compile-time SLOT of +// the connection's announced table and no trailer to write. +template +inline int64_t AlbumMeasureWire( const Ctx & ctx, const Album & root, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + int64_t bytes = -1; + if ( AlbumNumberFrom( ctx, numbering, root ) ) + { + TableIds ids; + ids.vocabulary = message; + bytes = AlbumMeasureBody( ctx, numbering, ids, root ); + if ( bytes >= 0 ) + { + const int64_t table = TableNodeTableMeasure( ctx, ids, numbering ); + // the FORM BYTE, the ROOT BODY — its own fields, the node table + // and the terminator — and the ID TABLE (docs/SPEC-TABLES.md §3) + const int64_t trailer = message ? 0 : TableIdsBytes( ids ); + bytes = table < 0 || ids.overflow ? -1 : 1 + bytes + table + trailer; + } + } + TableNumberingShutdown( numbering ); + return bytes; +} + +template +inline int64_t AlbumSaveWire( const Ctx & ctx, const Album & root, uint8_t * buffer, int64_t capacity, TableAllocator allocator, bool message = false ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + if ( !AlbumNumberFrom( ctx, numbering, root ) ) { TableNumberingShutdown( numbering ); return -1; } + TableWriter w( buffer, capacity ); + TableIds ids; + ids.vocabulary = message; + w.put8( message ? kTableWireMessageForm : kTableWireForm ); // the FORM BYTE is the whole header (§3) + // the root's own fields, then the node table's field, then the + // terminator: a reader that gives up inside the table has already + // decoded the ROOT'S OWN FIELDS (§3.1) + bool ok = AlbumSaveBodyFields( ctx, numbering, w, ids, root ) && TableNodeTableSave( ctx, w, ids, numbering ); + TableNumberingShutdown( numbering ); + if ( !ok || ids.overflow ) { return -1; } + w.put8( 0 ); // the ZERO REFERENCE that ends the root body + // A MESSAGE HAS NO TRAILER: its last byte is the body's terminator, + // because the ids live in the connection's table (§3.3). + if ( !message ) { TableIdsWrite( w, ids ); } + if ( w.overflow ) { return -1; } // the caller's buffer was too small + return w.offset; // == AlbumMeasure( root ) +} + +inline int64_t AlbumMeasure( const Album * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return AlbumMeasureWire( ctx, *root, allocator ); +} + +inline int64_t AlbumSave( const Album * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return AlbumSaveWire( ctx, *root, buffer, capacity, allocator ); +} + +inline int64_t AlbumMeasure( const AlbumBuilder & builder ) +{ + if ( builder.region != NULL ) { return AlbumMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return AlbumMeasureWire( ctx, *(const Album *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline int64_t AlbumSave( const AlbumBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return AlbumSave( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return AlbumSaveWire( ctx, *(const Album *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator ); +} + +inline int64_t AlbumMeasureMessage( const Album * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return AlbumMeasureWire( ctx, *root, allocator, true ); +} + +inline int64_t AlbumSaveMessage( const Album * root, uint8_t * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return AlbumSaveWire( ctx, *root, buffer, capacity, allocator, true ); +} + +inline int64_t AlbumMeasureMessage( const AlbumBuilder & builder ) +{ + if ( builder.region != NULL ) { return AlbumMeasureMessage( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return AlbumMeasureWire( ctx, *(const Album *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator, true ); +} + +inline int64_t AlbumSaveMessage( const AlbumBuilder & builder, uint8_t * buffer, int64_t capacity ) +{ + if ( builder.region != NULL ) { return AlbumSaveMessage( builder.AsConst(), buffer, capacity, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } + TableArenaCtx ctx = { &builder.arena }; + return AlbumSaveWire( ctx, *(const Album *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), buffer, capacity, builder.arena.allocator, true ); +} + +// AlbumLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t AlbumLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + if ( TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ) != TableOpenOk ) { return -1; } + // ANY BYTE BETWEEN THE ROOT'S TERMINATOR AND THE TABLE'S FIRST ENTRY + // IS MALFORMED (docs/SPEC-TABLES.md §3): the two ends of the file have + // met, nothing is decoded, and no region is sized from it. + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) { return -1; } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !AlbumWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = AlbumNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// AlbumLoad: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Album * AlbumLoad( uint8_t * region, int64_t region_bytes, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, then the trailer, and only then a body: + // a file that is both a newer form and damaged is a REFUSAL and never + // damage (docs/SPEC-TABLES.md §3). + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; if ( wire_file_bytes > 0 && wire_file[0] == kTableWireMessageForm ) { out->reason = message_form_as_file; } else { out->reason = newer_form; } } + return NULL; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return NULL; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + if ( region == NULL || region_bytes < (int64_t) sizeof( Album ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !AlbumWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = AlbumNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xd858c2cb7f1514ccull; + Album * root = new ( region ) Album; // lifetime only: LoadBody's first act is AlbumReset + AlbumReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = AlbumNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + AlbumNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + AlbumNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Album ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + AlbumLoadBody( r, nodes, *root ); + return root; +} + +// AlbumLoadMeasure: the exact region bytes a wire buffer will need, and it is +// ONE SCAN — a record's type id gives its storage size, its length gives the +// next record — reading no field value at all, so the caller owns the +// allocation and can refuse a number it did not expect (§6.5). +// +// It reports the DATA bytes and the ATTRIBUTION bytes separately, because the +// attribution is the wire's numbering made resident (§6.3) and a caller may +// release it once Load returns. The answer is their sum. +inline int64_t AlbumLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) +{ + TableReport ignored; + if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; + int64_t root_extent = 0; + if ( !AlbumWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + int64_t records = 0; + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = AlbumNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) + if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( attribution_bytes != NULL ) { *attribution_bytes = attribution; } + return data + attribution; +} + +// AlbumLoadMessage: decode the tolerant wire into the caller's exact-sized region and +// return the root. LOAD IS A SCAN, and that is the whole of its bound: it +// follows no reference, so there is no depth cap, no visited set and no +// ordering rule on the indices. Partial results are kept, as everywhere on +// this wire — the report says what happened. NULL means the CALLER's buffer +// was wrong. +inline const Album * AlbumLoadMessage( uint8_t * region, int64_t region_bytes, const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + // THE FORM BYTE IS READ FIRST, and then the connection's own table: + // a message with no table is REFUSED BY NAME, nothing is decoded, no + // counter moves and malformed does not fire (docs/SPEC-TABLES.md §3.3). + if ( message_bytes < 1 ) { out->malformed = true; return NULL; } + if ( message[0] != kTableWireMessageForm ) { out->refused = true; out->reason = newer_form; return NULL; } + if ( !vocabulary.announced ) { out->refused = true; out->reason = no_vocabulary; return NULL; } + const TableIdTable & ids_table = vocabulary.table; + const uint8_t * const wire = message + 1; + const int64_t wire_bytes = message_bytes - 1; + if ( region == NULL || region_bytes < (int64_t) sizeof( Album ) ) { out->malformed = true; return NULL; } + if ( ( ( (uintptr_t) region ) & ( kTableAlign - 1 ) ) != 0 ) { out->malformed = true; return NULL; } + memset( region, 0, (size_t) region_bytes ); + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + + // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed + int64_t root_extent = 0; + if ( !AlbumWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } + int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + records++; + int64_t storage = AlbumNodeStorage( type_id, length ); + if ( storage == kTableNodeRefused ) { out->malformed = true; return NULL; } + if ( storage > 0 ) { data += storage; } + } + } + int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); + if ( data + attribution > region_bytes ) { out->malformed = true; return NULL; } + + TableNodeMap nodes; + nodes.base = region; + nodes.entries = (const TableNodeDirEntry *) ( region + data ); + nodes.count = records + 1; + TableNodeDirEntry * directory = (TableNodeDirEntry *) ( region + data ); + directory[0].offset = 0; // position 0 is the ROOT, at offset 0 (§6.3) + directory[0].type_id = 0xd858c2cb7f1514ccull; + Album * root = new ( region ) Album; // lifetime only: LoadBody's first act is AlbumReset + AlbumReset( *root ); + + // PASS ONE: fill the numbering from the framing, so that an index + // resolves whichever way it points. It reads no body. + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t used = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Album ) ) + root_extent ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + int64_t storage = AlbumNodeStorage( type_id, length ); + if ( storage <= 0 ) + { + // a record whose type id this build cannot name KEEPS ITS + // INDEX, is counted once here and not once per pointer, and + // every reference to it reads null (§3.1) + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + directory[k + 1].type_id = type_id; + } + else + { + directory[k + 1].offset = (uint64_t) used; + directory[k + 1].type_id = type_id; + AlbumNodePlace( type_id, region + used, length ); + used += storage; + } + k++; + } + nodes.good = TableNodeScanWhole( scan ); + // the table is whole or it is nothing: a scan that failed counts + // malformed and NOT the unknowns it met on the way, because the + // numbering they belonged to does not exist (§3.1) + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + + // PASS TWO: decode each body into its own storage. A forward index + // resolves without scratch, because pass one already placed every node. + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + AlbumNodeBody( type_id, sub, nodes, region + directory[k + 1].offset ); + } + k++; + } + } + + // and the ROOT's own body last, so every index it carries resolves + // against a numbering already known good or already known bad + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Album ) ); + root_carve.left = root_extent; + nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's + AlbumLoadBody( r, nodes, *root ); + return root; +} + +// AlbumLoadBuilder: the TOOL's path — the same tolerant decode into a fresh +// builder, so loaded data can be edited and locked again. The numbering is +// the same one; what differs is where a node lives and therefore what a +// resolved slot holds — an arena offset here, a self-relative delta there. +inline bool AlbumLoadBuilder( AlbumBuilder & builder, const uint8_t * wire_file, int64_t wire_file_bytes, TableReport * report ) +{ + TableReport ignored; + TableReport * out = report != NULL ? report : &ignored; + TableIdTable ids_table; + int64_t body_bytes = 0; + const TableOpenVerdict verdict = TableOpen( wire_file, wire_file_bytes, ids_table, body_bytes ); + if ( verdict != TableOpenOk ) + { + if ( verdict == TableOpenDamaged ) { out->malformed = true; } else { out->refused = true; } + return false; + } + if ( TableBodyEndsEarly( wire_file + 1, body_bytes, ids_table ) ) + { + out->malformed = true; // a byte no field claims, before the table (§3) + return false; + } + const uint8_t * const wire = wire_file + 1; + const int64_t wire_bytes = body_bytes; + Album * root = builder.GetRoot(); + if ( root == NULL ) { out->malformed = true; return false; } + uint64_t type_id = 0; + const uint8_t * body = NULL; + int64_t length = 0; + int64_t records = 0; + { + TableReport counting; + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &counting, &ids_table ); + while ( TableNodeScanNext( scan, type_id, body, length ) ) { records++; } + } + // the AUTHORING side may allocate (§6.5), and this is the tool's path. + // It goes through the builder's own pair, like everything else the + // builder reaches, and the entries come back zeroed. + const TableAllocator allocator = builder.arena.allocator; + TableNodeDirEntry * directory = (TableNodeDirEntry *) allocator.alloc( allocator.context, ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ) ); + if ( directory == NULL ) { out->malformed = true; return false; } + directory[0].offset = (uint64_t) builder.root_ref.value; + directory[0].type_id = 0xd858c2cb7f1514ccull; + TableNodeMap nodes; + nodes.base = NULL; + nodes.entries = directory; + nodes.count = records + 1; + nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + int32_t unknown_records = 0; // counted once the scan is known whole + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + uint32_t at = AlbumNodeAlloc( type_id, builder.main, length ); + if ( at == 0 ) + { + unknown_records++; + directory[k + 1].offset = kTableNodeAbsent; + } + else + { + directory[k + 1].offset = (uint64_t) at; + } + directory[k + 1].type_id = type_id; + k++; + } + nodes.good = TableNodeScanWhole( scan ); + if ( nodes.good ) { out->unknown += unknown_records; } else { out->malformed = true; } + } + if ( nodes.good ) + { + TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); + int64_t k = 0; + while ( TableNodeScanNext( scan, type_id, body, length ) ) + { + if ( directory[k + 1].offset != kTableNodeAbsent ) + { + TableReader sub( body, length, out, &ids_table ); + AlbumNodeBody( type_id, sub, nodes, TableArenaAt( builder.arena, (uint32_t) directory[k + 1].offset ) ); + } + k++; + } + } + TableReader r( wire, wire_bytes, out, &ids_table ); + r.nested = false; // the ROOT body, the one that carries the node table + TableExtentCarve root_carve; + root_carve.worker = &builder.main; + nodes.carve = &root_carve; + bool ok = AlbumLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; + allocator.free( allocator.context, directory ); + return ok; +} + +// ---- the cooked form: point at a cook (docs/SPEC-TABLES.md §7) ---- + +// PhotoOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// Photo IS FIXED-SIZE, so its cook is ONE REGION OF ONE NODE and not a second +// shape (§7): one struct behind the header, at the region's base, which is +// what this returns. There is no graph below it and nothing to resolve. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Photo * PhotoOpen( const void * bytes, uint64_t length ) +{ + return (const Photo *) TableCookOpen( bytes, length, (uint64_t) sizeof( Photo ), (uint64_t) alignof( Photo ) ); +} + +// AlbumOpen: match the header and POINT. On a match the bytes ARE what this +// build wrote, in this build's layout and this build's byte order, so there +// is nothing to validate and nothing to fix up and the root comes back as it +// lies. On ANY refusal it returns NULL and the caller falls back to a wire +// load, which is the path that carries every version. +// +// It is O(1) IN THE FILE'S SIZE — the header and nothing per node — so a one +// megabyte cook and a one gigabyte cook open in the same time, and a mapped +// file's pages are touched only as they are used. That is a property of +// touching nothing at open rather than a separate mechanism. +// +// A REFERENCE INSIDE THE REGION IS DEREFERENCED THROUGH AlbumAt: the slot holds +// the signed self-relative byte delta of §6.3, so a deref is one add and +// needs no base pointer, a whole region relocates by plain memcpy, and a +// delta of zero is null. +// +// There is ONE entry point and no tolerant twin: a build either wrote this +// file or it did not, and the build version is what says which. Validating a +// file whose provenance a person doubts is schema cook-check, offline, +// over the ATTRIBUTION part beside the data — a person's decision, never a +// parameter on a load. +inline const Album * AlbumOpen( const void * bytes, uint64_t length ) +{ + return (const Album *) TableCookOpen( bytes, length, (uint64_t) sizeof( Album ), (uint64_t) alignof( Album ) ); +} + +// ---- the cooked form: WRITE a cook (docs/SPEC-TABLES.md §7.6) ---- +// +// The bytes are `schema cook`'s, and the tool stays the reference: the two +// writers are held to one file, byte for byte, in both byte orders. A cook is +// content-addressed by (asset hash, build version), so two writers of one +// instance produce ONE artifact or the pair means nothing. + +inline void PhotoCookBody( uint8_t * at, const Photo & value, TableByteOrder order ); +template inline bool AlbumCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Album & value, TableByteOrder order ); + +inline void PhotoCookBody( uint8_t * at, const Photo & value, TableByteOrder order ) +{ + table_cook_put( at + 0, (uint64_t) value.width, 4, order ); + table_cook_put( at + 4, (uint64_t) value.height, 4, order ); +} + +template inline bool AlbumCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Album & value, TableByteOrder order ) +{ + table_cook_put( at + 0, 0, 8, order ); // photos: the array's delta, filled by the extent writer + table_cook_put( at + 8, 0, 4, order ); // and its count + if ( !table_cook_ref( region, at + 16, (const void *) PhotoAt( ctx, value.cover ), order ) ) { return false; } // cover + return true; +} + +template inline bool PhotoCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Photo & value, TableByteOrder order ); +template inline bool AlbumCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Album & value, TableByteOrder order ); + +// PhotoCookExtent: Photo's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool PhotoCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Photo & value, TableByteOrder order ) +{ + (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; + return true; // no list or map below this record +} + +// AlbumCookExtent: Album's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool AlbumCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Album & value, TableByteOrder order ) +{ + { // photos: an unbounded array + TableListCursor cursor = TableListElements( ctx, value.photos ); + if ( !cursor.ok ) { return false; } + at = ( at + 7 ) & ~(int64_t) 7; // at alignof( TableRef ) + uint8_t * array = extent + at; + at += (int64_t) cursor.count * 8; // the whole array FIRST + // the SIXTEEN BYTES of the slot: the self-relative delta, then the count + table_cook_put( record + 0, cursor.count > 0 ? (uint64_t) (int64_t) ( array - ( record + 0 ) ) : 0, 8, order ); + table_cook_put( record + 8, (uint64_t) (uint32_t) cursor.count, 4, order ); + for ( int32_t i = 0; i < cursor.count; i++ ) // INDEX order, live elements only + { + if ( !table_cook_ref( region, array + i * 8, (const void *) PhotoAt( ctx, cursor[i] ), order ) ) { return false; } + } + } + return true; +} + +// PhotoCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool PhotoCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Photo & value, TableByteOrder order ) +{ + PhotoCookBody( at, value, order ); + int64_t extent_at = 0; + return PhotoCookExtent( ctx, region, at + 8, extent_at, at, value, order ); +} + +// AlbumCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). +template inline bool AlbumCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Album & value, TableByteOrder order ) +{ + if ( !AlbumCookBody( ctx, region, at, value, order ) ) { return false; } + int64_t extent_at = 0; + return AlbumCookExtent( ctx, region, at + 24, extent_at, at, value, order ); +} + +// PhotoCookMeasure: the whole cooked file's bytes — the header, the data part +// and the attribution part (docs/SPEC-TABLES.md §7.1). It answers in int64_t +// because a cook's part lengths are 64 bits: the scale this form exists for is +// a catalog, and a 32-bit answer would reimpose the ceiling §3.1 removed. +// +// Photo IS FIXED-SIZE, so the answer does not depend on the value: its cook is +// ONE REGION OF ONE NODE (§7) — the record at the region's base, its length +// rounded to the region's alignment, and one directory entry. +inline int64_t PhotoCookMeasure( const Photo & value ) +{ + (void) value; + return 88; // 64 header + 8 data + 16 attribution +} + +// PhotoCook: write one cooked file for the build this code is compiled into, +// in the byte order the caller names. The bytes are `schema cook`'s, byte for +// byte, and the tool is the reference (§7.6). +// +// THE CALLER OWNS THE BUFFER AND NOTHING IS ALLOCATED: measure, then write. +// A capacity short of the measure writes nothing and returns false, which is +// the same contract PhotoMeasure/PhotoSave has on the wire (§6.1). +// +// EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) — interior padding, the record's +// trailing padding, a string's unused tail, the bytes of a union outside its +// set arm, and the slack the rounded data length leaves. It comes from the one +// memset below rather than from a rule per padding site, and it is what makes +// two cooks of one value ONE artifact (§7). +inline bool PhotoCook( const Photo & value, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( out == NULL ) { return false; } + const uint64_t need = (uint64_t) PhotoCookMeasure( value ); + if ( capacity < need ) { return false; } + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); + // the HEADER (§7.1), every word a u64 in the order the file is produced in + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, 8, 8, order ); // data_length, rounded to the region's alignment + table_cook_put( raw + 32, 16, 8, order ); // attribution_length: one entry, one node + table_cook_put( raw + 40, 8, 8, order ); // the region's alignment + // the two RESERVED words are zero, and the memset already wrote them + // the DATA part: the region, which for a fixed root is the record at its base + PhotoCookBody( raw + 64, value, order ); + // the ATTRIBUTION part: the node directory (§6.3), written beside the data + // for `schema cook-check` — one entry, the root at offset zero, and its type + // id is the fnv1a64 of the table's name (§3.1) + table_cook_put( raw + 72, 0, 8, order ); + table_cook_put( raw + 80, 0xf1a78dd2508964c3ull, 8, order ); + return true; +} + +// AlbumCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one +// numbering — the root at zero, then every node in index order at +// align_up( offset, alignof ) for its OWN type, no slack between them, the +// data length rounded to the greatest alignment among them and never below +// eight. The offsets go into the region's table when it has one, and are only +// summed when it does not (a measure). A type id the numbering carries that +// this root cannot name is the two walks disagreeing, and it is refused. +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent +// (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context +// the numbering walked and reads the same arrays that walk read. +template +inline bool AlbumCookLayout( const Ctx & ctx, const Album & root, const TableNumbering & numbering, TableCookRegion & region ) +{ + region.numbering = &numbering; + region.count = numbering.count + 1; + const int64_t root_extent = AlbumExtent( ctx, root ); + if ( root_extent < 0 ) { return false; } + int64_t offset = 24 + root_extent; // the root at zero, its extent behind it + int64_t align = 8; + if ( region.offsets != NULL ) { region.offsets[0] = 0; } + for ( int64_t k = 0; k < numbering.count; k++ ) + { + int64_t size = 0; + int64_t node_align = 0; + switch ( numbering.entries[k].type_id ) + { + case 0xf1a78dd2508964c3ull: size = 8; node_align = 4; break; // Photo + default: return false; + } + offset = ( offset + node_align - 1 ) & ~( node_align - 1 ); + if ( region.offsets != NULL ) { region.offsets[k + 1] = offset; } + offset += size; + if ( node_align > align ) { align = node_align; } + } + region.bytes = ( offset + align - 1 ) & ~( align - 1 ); + region.align = align; + return true; +} + +// AlbumCookMeasureFrom: the whole cooked file's bytes for one graph — the header, +// the data part and the attribution part (§7.1). IT DEPENDS ON THE VALUE, +// because the answer is the numbering: the depth-first walk of §3.1 is run +// here and run again by the write, and neither carries the other's (§7.6). A +// data cycle is refused by the walk and answers -1. +template +inline int64_t AlbumCookMeasureFrom( const Ctx & ctx, const Album & root, TableAllocator allocator ) +{ + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + int64_t bytes = -1; + if ( AlbumNumberFrom( ctx, numbering, root ) && AlbumCookLayout( ctx, root, numbering, region ) ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + bytes = data_offset + region.bytes + region.count * (int64_t) sizeof( TableNodeDirEntry ); + } + TableNumberingShutdown( numbering ); + return bytes; +} + +// AlbumCookFrom: write one cooked file of a pointered graph, in the byte order +// the caller names. The bytes are `schema cook`'s, byte for byte (§7.6). +// +// THE CALLER OWNS THE OUTPUT and nothing is allocated toward it. What is +// allocated is the numbering — the identity map, the entry array and one +// offset per node — through the pair handed in, and released before this +// returns (§6.5, §13.9). A capacity short of the measure writes nothing. +// +// THE HEADER IS WRITTEN LAST. A reference the numbering did not carry is +// found while a body is being written, and a write that refuses there has +// already put bytes in the buffer; with no magic ahead of them, no Open can +// mistake them for a cook. +template +inline bool AlbumCookFrom( const Ctx & ctx, const Album & root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator ) +{ + if ( out == NULL ) { return false; } + TableNumbering numbering; + TableNumberingInit( numbering, allocator ); + TableCookRegion region; + bool ok = AlbumNumberFrom( ctx, numbering, root ); + if ( ok ) + { + region.offsets = (int64_t *) allocator.alloc( allocator.context, ( numbering.count + 1 ) * (int64_t) sizeof( int64_t ) ); + ok = region.offsets != NULL && AlbumCookLayout( ctx, root, numbering, region ); + } + if ( ok ) + { + const int64_t data_offset = ( kTableCookHeaderBytes + region.align - 1 ) & ~( region.align - 1 ); + const int64_t attribution = region.count * (int64_t) sizeof( TableNodeDirEntry ); + const int64_t need = data_offset + region.bytes + attribution; + ok = (uint64_t) need <= capacity; + if ( ok ) + { + uint8_t * raw = (uint8_t *) out; + memset( raw, 0, (size_t) need ); // EVERY BYTE NO FIELD COVERS IS ZERO (§7.2) + region.base = raw + data_offset; + // the DATA part: the root at the region's base, then every numbered + // node at the offset the layout gave it, each through its own writer + ok = AlbumCookNode( ctx, region, region.base, root, order ); + for ( int64_t k = 0; ok && k < numbering.count; k++ ) + { + uint8_t * at = region.base + region.offsets[k + 1]; + const void * node = numbering.entries[k].node; + switch ( numbering.entries[k].type_id ) + { + case 0xf1a78dd2508964c3ull: ok = PhotoCookNode( ctx, region, at, *(const Photo *) node, order ); break; // Photo + default: ok = false; break; + } + } + // the ATTRIBUTION part: the node directory (§6.3), one entry per node + // in index order, for `schema cook-check` + uint8_t * entry = raw + data_offset + region.bytes; + table_cook_put( entry, 0, 8, order ); + table_cook_put( entry + 8, 0xd858c2cb7f1514ccull, 8, order ); // the root: fnv1a64( "Album" ) + for ( int64_t k = 0; k < numbering.count; k++ ) + { + entry += sizeof( TableNodeDirEntry ); + table_cook_put( entry, (uint64_t) region.offsets[k + 1], 8, order ); + table_cook_put( entry + 8, numbering.entries[k].type_id, 8, order ); + } + // and the HEADER (§7.1), every word a u64 in the order the file is + // produced in; the two RESERVED words are the memset's zeros + if ( ok ) + { + table_cook_put( raw + 0, TableCookMagic, 8, order ); + table_cook_put( raw + 8, BuildVersion, 8, order ); + table_cook_put( raw + 16, (uint64_t) ( order == TableByteOrder::Big ? 2 : 1 ), 8, order ); + table_cook_put( raw + 24, (uint64_t) region.bytes, 8, order ); + table_cook_put( raw + 32, (uint64_t) attribution, 8, order ); + table_cook_put( raw + 40, (uint64_t) region.align, 8, order ); + } + } + } + allocator.free( allocator.context, region.offsets ); + TableNumberingShutdown( numbering ); + return ok; +} + +// AlbumCookMeasure / AlbumCook over a REGION root — a locked builder's AsConst, a +// region AlbumLoad produced, or an opened cook — with the pair the numbering +// allocates through as an optional last argument, as the wire's own entries +// take it (§13.9). +inline int64_t AlbumCookMeasure( const Album * root, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return -1; } + TableRegionCtx ctx; + return AlbumCookMeasureFrom( ctx, *root, allocator ); +} + +inline bool AlbumCook( const Album * root, void * out, uint64_t capacity, TableByteOrder order, TableAllocator allocator = TableDefaultAllocator() ) +{ + if ( root == NULL ) { return false; } + TableRegionCtx ctx; + return AlbumCookFrom( ctx, *root, out, capacity, order, allocator ); +} + +// and over a BUILDER, locked or not: the builder's own pair, and the arena +// encoding while it is still mutable (§6.3). +inline int64_t AlbumCookMeasure( const AlbumBuilder & builder ) +{ + if ( builder.region != NULL ) { return AlbumCookMeasure( builder.AsConst(), builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return -1; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return AlbumCookMeasureFrom( ctx, *(const Album *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), builder.arena.allocator ); +} + +inline bool AlbumCook( const AlbumBuilder & builder, void * out, uint64_t capacity, TableByteOrder order ) +{ + if ( builder.region != NULL ) { return AlbumCook( builder.AsConst(), out, capacity, order, builder.arena.allocator ); } + if ( builder.root_ref.null() ) { return false; } // the root allocation failed + TableArenaCtx ctx = { &builder.arena }; + return AlbumCookFrom( ctx, *(const Album *) TableArenaAt( builder.arena, (uint32_t) builder.root_ref.value ), out, capacity, order, builder.arena.allocator ); +} + +// ---- relocatability, enforced: the wire is a pure length-prefixed +// stream AND the decoded storage is pointer-free — every closure type +// must stay trivially copyable and standard-layout, so instances can be +// memcpy'd, mmap'd, shared across processes, and walked through +// descriptor offsets. A failure here means a pointer, virtual or +// non-trivial member crept into generated storage. +// +// They ask the COMPILER ITSELF, which is what every C++ standard library +// answers the same two questions with — and it costs this header no +// include at all. +// A pointer FIELD is a TableRef — eight bytes and no address — so the +// property holds in BOTH forms: a fixed-size table is one relocatable +// struct, and a packed region is one relocatable block whose references +// are self-relative and therefore survive a plain memcpy. +static_assert( __is_trivially_copyable( Photo ), "Photo must stay relocatable" ); +static_assert( __is_standard_layout( Photo ), "Photo must stay standard-layout for offsetof" ); +static_assert( __is_trivially_copyable( Album ), "Album must stay relocatable" ); +static_assert( __is_standard_layout( Album ), "Album must stay standard-layout for offsetof" ); + +// ---- the cook's layout contract (docs/SPEC-TABLES.md §20.3) ---- +// +// The compiler derived every number below from the declaration and folded it +// into the BUILD VERSION; these asserts are this compiler saying whether it +// agrees. The model is not self-evidently right — on 32-bit System V +// alignof(uint64_t) is 4, not 8 — which is precisely why it is asserted +// rather than assumed. +static_assert( sizeof( Photo ) == 8, "Photo's sizeof moved: the build version was taken over 8, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Photo ) == 4, "Photo's alignof moved: the build version was taken over 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Photo, width ) == 0, "Photo's field width moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Photo, height ) == 4, "Photo's field height moved: the build version was taken over offset 4 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( sizeof( Album ) == 24, "Album's sizeof moved: the build version was taken over 24, so a cook of it would not be this build's file (docs/SPEC-TABLES.md §20.3)" ); +static_assert( alignof( Album ) == 8, "Album's alignof moved: the build version was taken over 8 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Album, photos ) == 0, "Album's field photos moved: the build version was taken over offset 0 (docs/SPEC-TABLES.md §20.3)" ); +static_assert( offsetof( Album, cover ) == 16, "Album's field cover moved: the build version was taken over offset 16 (docs/SPEC-TABLES.md §20.3)" ); + +static_assert( alignof( TableRef ) <= kTableAlign, "Album.photos: an unbounded array's element alignment must fit the arena's" ); + +// ---- reflection descriptors (tables only, docs/SPEC-TABLES.md) ---- + +inline const TableTypeInfo * PhotoTableType(); +inline const TableTypeInfo * AlbumTableType(); +// The descriptors are CONSTANT-INITIALISED data, and a field's target is +// the ADDRESS of another descriptor. These declarations are what let a +// self- or mutually-referential graph — Node naming itself through *Node — +// be expressed as constant data instead of a lazy link, which could not +// have been written race-free OR recursion-safe. The whole reflection +// surface is therefore immutable: read it from any thread, any time. +extern const TableTypeInfo PhotoTableInfo; +extern const TableTypeInfo AlbumTableInfo; + +inline const TableFieldInfo PhotoTableFields[] = { + { "width", "width", "uint32", 0xdbdacd932fd1e9bfull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Photo, width ), (uint32_t) sizeof( Photo::width ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "height", "height", "uint32", 0x17720bf67d347222ull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Photo, height ), (uint32_t) sizeof( Photo::height ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo PhotoTableInfo = { "Photo", (uint32_t) sizeof( Photo ), 2, PhotoTableFields, +[]( void * p ) { PhotoReset( *(Photo *) p ); }, false }; +inline const TableTypeInfo * PhotoTableType() { return &PhotoTableInfo; } + +inline const TableFieldInfo AlbumTableFields[] = { + { "photos", "photos", "Photo", 0x40b1d94aff3ab130ull, 17, true, true, []( const void * slot ) -> const void * { return (const void *) PhotoAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) PhotoEmplace( worker, *(TableRef *) slot ); }, true, false, 0, (uint32_t) offsetof( Album, photos ), (uint32_t) sizeof( TableRef ), (uint32_t) offsetof( Album, photos.count ), 0xffffffffu, &PhotoTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t ) -> void * { return (void *) TableListPlace( worker, *(TableList *) slot ); }, "" }, + { "cover", "cover", "Photo", 0xaa19a78e404dea20ull, 17, false, true, []( const void * slot ) -> const void * { return (const void *) PhotoAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) PhotoEmplace( worker, *(TableRef *) slot ); }, false, false, 0, (uint32_t) offsetof( Album, cover ), (uint32_t) sizeof( TableRef ), 0xffffffffu, 0xffffffffu, &PhotoTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, +}; +inline const TableTypeInfo AlbumTableInfo = { "Album", (uint32_t) sizeof( Album ), 2, AlbumTableFields, +[]( void * p ) { AlbumReset( *(Album *) p ); }, true }; +inline const TableTypeInfo * AlbumTableType() { return &AlbumTableInfo; } + +// ---- the text form (docs/SPEC-TABLES.md §16) ---- + +// Photo in and out of a JSON text — one instance, one text, the generic +// walk over this type's descriptors (docs/SPEC-TABLES.md §16). Defined in +// SharedTable.cpp; link it to use them. +bool PhotoFromJson( Photo & value, const char * text, int64_t bytes, TableReport * report ); +int64_t PhotoToJsonMeasure( const Photo & value ); +int64_t PhotoToJson( const Photo & value, char * buffer, int64_t capacity ); + +// Album in and out of a JSON text (docs/SPEC-TABLES.md §16.7): read into a +// builder, written from a region's const root. A node named more than once +// carries `&node` in the text. Defined in SharedTable.cpp; link it to use them. +bool AlbumFromJson( AlbumBuilder & builder, const char * text, int64_t bytes, TableReport * report ); +int64_t AlbumToJsonMeasure( const Album * root, TableAllocator allocator = TableDefaultAllocator() ); +int64_t AlbumToJson( const Album * root, char * buffer, int64_t capacity, TableAllocator allocator = TableDefaultAllocator() ); + +} // namespace listdemo diff --git a/testdata/golden/tables/maps/DepthTable.cpp b/testdata/golden/tables/maps/DepthTable.cpp index 703a26c0b..4265104a5 100644 --- a/testdata/golden/tables/maps/DepthTable.cpp +++ b/testdata/golden/tables/maps/DepthTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2731,14 +2748,33 @@ inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * inf // ---- json graph walk: end ---- +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + // ---- json map walk: begin ---- -inline bool TableJsonIsMap( const TableFieldInfo * f ) { return f->entry != NULL; } +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} // the entry's two rows: fields[0] IS the key and fields[1] IS the value, which // is what makes a user's own table of pairs the same bytes (§2.8) -inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->entry->fields[0]; } -inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->entry->fields[1]; } +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } @@ -2798,16 +2834,17 @@ inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const // A region holds them in that order already, so this is the array in place. inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) { - const int32_t count = f->map_count( slot ); + const int32_t count = TableJsonExtentCount( slot ); if ( count == 0 ) { out.raw( "{}", 2 ); return true; } const TableFieldInfo * key = TableJsonMapKeyField( f ); const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); out.put( '{' ); for ( int32_t i = 0; i < count; i++ ) { if ( i > 0 ) { out.put( ',' ); } out.line( depth + 1 ); - const void * entry = f->map_at( slot, i ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); TableJsonWriteMapKey( out, entry, key ); out.raw( ": ", 2 ); if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } @@ -2896,15 +2933,15 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf } if ( !fits ) { in.report->kind_mismatch++; place = false; } } - const int32_t before = f->map_count( (const void *) slot ); - void * entry = place ? f->map_insert( *graph->worker, slot, token, token_length, key_value ) : NULL; + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; if ( place && entry == NULL ) { // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the // wire's rule, because a clamped key is a merged entry (§2.8). in.report->clamped++; } - else if ( entry != NULL && f->map_count( (const void *) slot ) == before ) + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) { in.report->duplicate++; // last-wins, the object rule inside the map } @@ -2955,6 +2992,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf // ---- json map walk: end ---- +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace mapdemo #endif // MAPDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/maps/DepthTable.h b/testdata/golden/tables/maps/DepthTable.h index 78653da82..ccff4552c 100644 --- a/testdata/golden/tables/maps/DepthTable.h +++ b/testdata/golden/tables/maps/DepthTable.h @@ -110,6 +110,18 @@ struct TableReport TableMessageReason reason = newer_form; }; + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; // ---- reflection (tables only, docs/SPEC-TABLES.md) ---- // // Static field descriptors for every type in the table closure: name, wire @@ -225,18 +237,13 @@ struct TableFieldInfo // a function pointer at compile time; the arms themselves are a static // inside it). NULL for every other kind. const TableUnionInfo * (*arms)(); - // a MAP (docs/SPEC-TABLES.md §2.8): the generated ENTRY's descriptor — - // fields[0] is the key and fields[1] the value — and the three the ONE - // text walk cannot spell for itself, because TableMap is a type - // it has no name for. NULL on every field that is not a map. - const TableTypeInfo * entry; - int32_t ( * map_count )( const void * slot ); - const void * ( * map_at )( const void * slot, int32_t index ); - // place one entry BY KEY and hand back the entry, at its defaults: a - // string key comes in as the bytes and the length, an integer key as - // the value, and NULL is NOT INSERTED — a key past the bound, or an - // arena that could not carve another segment. - void * ( * map_insert )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded }; @@ -1292,8 +1299,8 @@ struct TableWorker return blob; } - // RAW, ZEROED storage of the bytes asked for, at the alignment asked for — a MAP's builder head and its - // entry segments (docs/SPEC-TABLES.md §2.8). It is not a node: it carries + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries // no type id, takes no index and has no Reset, so it goes through the same // slab and span the blob path uses rather than through Alloc. uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) @@ -1577,12 +1584,6 @@ static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts th // resolving through it yields NULL and can never fabricate the root. static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; -// What a node's storage answers when the FRAMING ITSELF is refused rather than -// merely unnameable: a map whose N cannot fit in its L (docs/SPEC-TABLES.md -// §2.8). An unnameable type id commands no storage and keeps its index; this -// one makes the whole measure answer -1 (§7.6). -static const int64_t kTableNodeRefused = -2; - // ---- the numbering, on the SAVE side ---- // // One entry per reachable node in FIRST-VISIT order, so entry k is node index @@ -1780,9 +1781,9 @@ struct TableNodeDirEntry uint64_t type_id; }; -// a map's extent cursor, defined with the map runtime (docs/SPEC-TABLES.md -// §2.8); the node map names it only through a pointer. -struct TableMapCarve; +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; // TableNodeMap is what a pointer slot resolves through while a body decodes. struct TableNodeMap @@ -1795,17 +1796,21 @@ struct TableNodeMap // takes the SELF-RELATIVE delta so a deref is one add, and the tool's // builder path takes the node's ARENA OFFSET (§6.3). bool arena = false; - // WHERE A MAP'S ENTRIES LAND while this node's body decodes - // (docs/SPEC-TABLES.md §2.8): the node's own extent on the region path - // and the builder's arena on the tool's. It is MUTABLE because the - // cursor belongs to ONE node's decode and the dispatch that owns that - // node holds the map by const reference, exactly as it did before maps - // existed — the decoder's signature does not move for a construct it - // may not carry. - mutable TableMapCarve * carve = NULL; - // and the TOOL's path's allocation front, set once: there a map's - // entries are the builder's arena's rather than a node's extent. + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; }; // TableNodeResolve places one node index in a pointer slot, and every failure @@ -1949,6 +1954,90 @@ inline bool TableNodeScanWhole( TableNodeScan & s ) #endif // MAPDEMO_SCHEMA_TABLE_ARENA +#ifndef MAPDEMO_SCHEMA_TABLE_EXTENT +#define MAPDEMO_SCHEMA_TABLE_EXTENT + +namespace mapdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace mapdemo + +#endif // MAPDEMO_SCHEMA_TABLE_EXTENT + #ifndef MAPDEMO_SCHEMA_TABLE_MAP #define MAPDEMO_SCHEMA_TABLE_MAP @@ -2379,12 +2468,6 @@ inline TableMapEach TableMapEachOf( const TableArena & arena, const Table return each; } -// AN UNREACHED SLOT MUST HOLD NO MAP WITH ENTRIES IN IT (§2.8, §7.6). An empty -// map takes no bytes, so a record whose extent measures ZERO is a record whose -// every by-value map is empty; a measure that REFUSED answers non-zero here -// too, and refusing on it is the same answer one level up. -inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } - // ---- the LOAD side: where a decoded entry lands (§2.8) ---- // // THE READER TRUSTS NOTHING and spends one compare per entry. Every load path @@ -2394,16 +2477,9 @@ inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } // out of the holder node's own extent, and the TOOL's path appends into the // builder's arena, and the decoder above them cannot tell which it has. -// TableMapCarve is a node's extent cursor, PRE-ORDER: a map's whole entry -// array first, then, entry by entry in key order, the arrays of any map an -// entry's value holds by value. The cursor is the node map's, because the -// generated decoder is threaded with that and not with a region. -struct TableMapCarve -{ - uint8_t * at = NULL; // the region path: the node's extent, unspent - int64_t left = 0; - TableWorker * worker = NULL; // the TOOL's path: entries come from the arena -}; +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. // TableMapFill is one map field being decoded: where the next entry lands, and // the entry that last LANDED, which is what the ascending check compares @@ -2528,10 +2604,11 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // LoadMeasure's term for a map is N x sizeof( Entry ) rounded to // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value -// holds a map of its own, the entries' headers under it. The caller owns the -// allocation precisely so it can refuse a number it did not expect. -typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ); - +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2539,8 +2616,8 @@ typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, i static const int64_t kTableMapEntryFloor = 2; inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, - int64_t entry_size, int64_t entry_align, TableMapWireExtentFn inner, - const TableIdTable * ids ) + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; TableReader r( body, length, &scratch, ids ); @@ -2548,8 +2625,9 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; - if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { return false; } // an N the map's L cannot carry + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); at += (int64_t) n * entry_size; if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term @@ -2557,49 +2635,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & { uint64_t elem = 0; if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// the same framing walk over an ARRAY OF TABLES that is not a map: its -// elements' own maps are part of this node's extent too -inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each -// length-prefixed element (docs/SPEC-TABLES.md §3.2) -inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t key = 0; - if ( !r.getleb( key ) ) { return true; } - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } r.offset += (int64_t) elem; } return true; @@ -3067,7 +3103,7 @@ inline void SquadRosterEntryReset( SquadRosterEntry & value ) inline void SquadReset( Squad & value ) { - value.roster.entries.value = 0; // map[uint8]Item — empty + value.roster.entries.value = 0; // map[uint8]Item: empty value.roster.count = 0; value.roster.padding = 0; } @@ -3960,10 +3996,10 @@ inline bool DepthLoadBody( TableReader & r, const TableNodeMap & nodes, Depth & } } -// SquadWireExtent: the extent Squad's maps command, from the FRAMING alone. +// SquadWireExtent: the extent Squad's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool SquadWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool SquadWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; // the scan's framing damage is the LOAD's to report TableReader r( body, length, &scratch, ids ); @@ -3982,17 +4018,17 @@ inline bool SquadWireExtent( const uint8_t * body, int64_t length, int64_t & at, if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( SquadRosterEntry ), (int64_t) alignof( SquadRosterEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( SquadRosterEntry ), (int64_t) alignof( SquadRosterEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( !r.skip( field_kind ) ) { return true; } } } -// SquadMapExtentAt: the node extent Squad's maps take, PRE-ORDER, advancing the -// running offset exactly as SquadMapPack advances it (docs/SPEC-TABLES.md §2.8). +// SquadExtentAt: the node extent Squad's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as SquadExtentPack advances it (§2.8, §2.9). template -inline bool SquadMapExtentAt( const Ctx & ctx, const Squad & value, int64_t & at ) +inline bool SquadExtentAt( const Ctx & ctx, const Squad & value, int64_t & at ) { { TableMapCursor cursor = TableMapOrder( ctx, value.roster ); @@ -4007,18 +4043,18 @@ inline bool SquadMapExtentAt( const Ctx & ctx, const Squad & value, int64_t & at // the whole extent of one node, from a fresh offset: what a pack reserves // for it beside the record's own storage. template -inline int64_t SquadMapExtent( const Ctx & ctx, const Squad & value ) +inline int64_t SquadExtent( const Ctx & ctx, const Squad & value ) { int64_t at = 0; - if ( !SquadMapExtentAt( ctx, value, at ) ) { return -1; } + if ( !SquadExtentAt( ctx, value, at ) ) { return -1; } return at; } -// SquadMapPack: carve Squad's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset SquadMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// SquadExtentPack: carve Squad's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset SquadExtentAt advances (§2.8, §2.9). template -inline bool SquadMapPack( const Ctx & ctx, const Squad & src, Squad & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool SquadExtentPack( const Ctx & ctx, const Squad & src, Squad & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { { TableMapCursor cursor = TableMapOrder( ctx, src.roster ); @@ -4040,10 +4076,10 @@ inline bool SquadMapPack( const Ctx & ctx, const Squad & src, Squad & dst, uint8 return true; } -// DepthWireExtent: the extent Depth's maps command, from the FRAMING alone. +// DepthWireExtent: the extent Depth's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool DepthWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool DepthWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; // the scan's framing damage is the LOAD's to report TableReader r( body, length, &scratch, ids ); @@ -4056,34 +4092,34 @@ inline bool DepthWireExtent( const uint8_t * body, int64_t length, int64_t & at, const uint64_t field_id = ids->at( field_ref ); if ( !r.has( 1 ) ) { return true; } uint8_t field_kind = r.get8(); - if ( field_id == 0x1a08aa1921ca5cafull && field_kind == 13 ) // one: a nesting that holds a map + if ( field_id == 0x1a08aa1921ca5cafull && field_kind == 13 ) // one: a nesting that holds a list or a map { uint64_t nested_len = 0; if ( !r.getleb( nested_len ) || !r.room( nested_len ) ) { return true; } const uint8_t * nested_body = r.buffer + r.offset; r.offset += (int64_t) nested_len; - if ( !SquadWireExtent( nested_body, (int64_t) nested_len, at, ids ) ) { return false; } + if ( !SquadWireExtent( nested_body, (int64_t) nested_len, at, ids, reason ) ) { return false; } continue; } - if ( field_id == 0x1f6459a2cea1fc02ull && field_kind == 14 ) // many: a nesting that holds a map + if ( field_id == 0x1f6459a2cea1fc02ull && field_kind == 14 ) // many: a nesting that holds a list or a map { uint64_t nested_len = 0; if ( !r.getleb( nested_len ) || !r.room( nested_len ) ) { return true; } const uint8_t * nested_body = r.buffer + r.offset; r.offset += (int64_t) nested_len; - if ( !TableWireExtentElements( nested_body, (int64_t) nested_len, at, &SquadWireExtent, ids ) ) { return false; } + if ( !TableWireExtentElements( nested_body, (int64_t) nested_len, at, &SquadWireExtent, ids, reason ) ) { return false; } continue; } - if ( field_id == 0x70551ff29550f15dull && field_kind == 16 ) // keyed: a nesting that holds a map + if ( field_id == 0x70551ff29550f15dull && field_kind == 16 ) // keyed: a nesting that holds a list or a map { uint64_t nested_len = 0; if ( !r.getleb( nested_len ) || !r.room( nested_len ) ) { return true; } const uint8_t * nested_body = r.buffer + r.offset; r.offset += (int64_t) nested_len; - if ( !TableWireExtentKeyed( nested_body, (int64_t) nested_len, at, &SquadWireExtent, ids ) ) { return false; } + if ( !TableWireExtentKeyed( nested_body, (int64_t) nested_len, at, &SquadWireExtent, ids, reason ) ) { return false; } continue; } - if ( field_id == 0xe756c0190570ccb5ull && field_kind == 15 ) // arm: a union arm that holds a map + if ( field_id == 0xe756c0190570ccb5ull && field_kind == 15 ) // arm: a union arm that holds a list or a map { uint64_t arm_ref = 0; if ( !r.getleb( arm_ref ) ) { return true; } @@ -4098,7 +4134,7 @@ inline bool DepthWireExtent( const uint8_t * body, int64_t length, int64_t & at, r.offset += (int64_t) arm_len; switch ( arm_id ) { - case 0xd5c2bb95d63e6331ull: if ( !SquadWireExtent( arm_body, (int64_t) arm_len, at, ids ) ) { return false; } break; // squad + case 0xd5c2bb95d63e6331ull: if ( !SquadWireExtent( arm_body, (int64_t) arm_len, at, ids, reason ) ) { return false; } break; // squad default: break; // an arm this reader cannot name reads None } continue; @@ -4107,31 +4143,31 @@ inline bool DepthWireExtent( const uint8_t * body, int64_t length, int64_t & at, } } -// DepthMapExtentAt: the node extent Depth's maps take, PRE-ORDER, advancing the -// running offset exactly as DepthMapPack advances it (docs/SPEC-TABLES.md §2.8). +// DepthExtentAt: the node extent Depth's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as DepthExtentPack advances it (§2.8, §2.9). template -inline bool DepthMapExtentAt( const Ctx & ctx, const Depth & value, int64_t & at ) +inline bool DepthExtentAt( const Ctx & ctx, const Depth & value, int64_t & at ) { { // one (nested by value) - if ( !SquadMapExtentAt( ctx, value.one, at ) ) { return false; } + if ( !SquadExtentAt( ctx, value.one, at ) ) { return false; } } for ( int32_t i = 0; i < value.many_count && i < 3; i++ ) // many { - if ( !SquadMapExtentAt( ctx, value.many[i], at ) ) { return false; } + if ( !SquadExtentAt( ctx, value.many[i], at ) ) { return false; } } for ( int32_t i = value.many_count; i < 3; i++ ) // many: the slots the walk does not reach (§7.6) { - if ( !TableMapUnreachedEmpty( SquadMapExtent( ctx, value.many[i] ) ) ) { return false; } + if ( !TableExtentUnreachedEmpty( SquadExtent( ctx, value.many[i] ) ) ) { return false; } } for ( int32_t i = 0; i < 2; i++ ) // keyed { - if ( !SquadMapExtentAt( ctx, value.keyed.slots[i], at ) ) { return false; } + if ( !SquadExtentAt( ctx, value.keyed.slots[i], at ) ) { return false; } } switch ( value.arm.type ) // arm: the set arm is the edge { case ForceType::Squad: { - if ( !SquadMapExtentAt( ctx, value.arm.squad, at ) ) { return false; } + if ( !SquadExtentAt( ctx, value.arm.squad, at ) ) { return false; } break; } default: break; @@ -4142,39 +4178,39 @@ inline bool DepthMapExtentAt( const Ctx & ctx, const Depth & value, int64_t & at // the whole extent of one node, from a fresh offset: what a pack reserves // for it beside the record's own storage. template -inline int64_t DepthMapExtent( const Ctx & ctx, const Depth & value ) +inline int64_t DepthExtent( const Ctx & ctx, const Depth & value ) { int64_t at = 0; - if ( !DepthMapExtentAt( ctx, value, at ) ) { return -1; } + if ( !DepthExtentAt( ctx, value, at ) ) { return -1; } return at; } -// DepthMapPack: carve Depth's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset DepthMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// DepthExtentPack: carve Depth's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset DepthExtentAt advances (§2.8, §2.9). template -inline bool DepthMapPack( const Ctx & ctx, const Depth & src, Depth & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool DepthExtentPack( const Ctx & ctx, const Depth & src, Depth & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { { // one (nested by value) - if ( !SquadMapPack( ctx, src.one, dst.one, extent, at, capacity ) ) { return false; } + if ( !SquadExtentPack( ctx, src.one, dst.one, extent, at, capacity ) ) { return false; } } for ( int32_t i = 0; i < src.many_count && i < 3; i++ ) // many { - if ( !SquadMapPack( ctx, src.many[i], dst.many[i], extent, at, capacity ) ) { return false; } + if ( !SquadExtentPack( ctx, src.many[i], dst.many[i], extent, at, capacity ) ) { return false; } } for ( int32_t i = src.many_count; i < 3; i++ ) // many: the slots the walk does not reach (§7.6) { - if ( !TableMapUnreachedEmpty( SquadMapExtent( ctx, src.many[i] ) ) ) { return false; } + if ( !TableExtentUnreachedEmpty( SquadExtent( ctx, src.many[i] ) ) ) { return false; } } for ( int32_t i = 0; i < 2; i++ ) // keyed { - if ( !SquadMapPack( ctx, src.keyed.slots[i], dst.keyed.slots[i], extent, at, capacity ) ) { return false; } + if ( !SquadExtentPack( ctx, src.keyed.slots[i], dst.keyed.slots[i], extent, at, capacity ) ) { return false; } } switch ( src.arm.type ) // arm: the set arm is the edge { case ForceType::Squad: { - if ( !SquadMapPack( ctx, src.arm.squad, dst.arm.squad, extent, at, capacity ) ) { return false; } + if ( !SquadExtentPack( ctx, src.arm.squad, dst.arm.squad, extent, at, capacity ) ) { return false; } break; } default: break; @@ -4323,7 +4359,7 @@ inline bool SquadPack( const Ctx & ctx, TablePackMap & seen, const Squad & src, int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Squad ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !SquadMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !SquadExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return SquadPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -4422,7 +4458,7 @@ inline bool DepthPack( const Ctx & ctx, TablePackMap & seen, const Depth & src, int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Depth ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !DepthMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !DepthExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return DepthPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -4533,7 +4569,7 @@ inline bool SquadBuilder::Lock() below = SquadPackMeasure( ctx, seen, root ); } if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it - int64_t root_extent = SquadMapExtent( ctx, root ); + int64_t root_extent = SquadExtent( ctx, root ); if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ) + below; // the AUTHORING path may allocate (§6.5), and it does so through the @@ -4631,10 +4667,11 @@ inline uint32_t SquadNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t // already owns. inline void SquadNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) { - // the node's own EXTENT, where its maps' entry arrays are carved from, - // PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8). The tool's - // path carries a worker instead: there the entries are the arena's. - TableMapCarve carve; + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; carve.worker = nodes.worker; if ( carve.worker == NULL ) { @@ -4783,7 +4820,7 @@ inline int64_t SquadSaveMessage( const SquadBuilder & builder, uint8_t * buffer, // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t SquadLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t SquadLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; TableIdTable ids_table; @@ -4796,8 +4833,9 @@ inline int64_t SquadLoadMeasure( const uint8_t * wire_file, int64_t wire_file_by const uint8_t * const wire = wire_file + 1; const int64_t wire_bytes = body_bytes; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -4807,7 +4845,7 @@ inline int64_t SquadLoadMeasure( const uint8_t * wire_file, int64_t wire_file_by { records++; int64_t storage = SquadNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -4851,8 +4889,9 @@ inline const Squad * SquadLoad( uint8_t * region, int64_t region_bytes, const ui int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); int64_t records = 0; { @@ -4935,7 +4974,7 @@ inline const Squad * SquadLoad( uint8_t * region, int64_t region_bytes, const ui // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Squad ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -4951,7 +4990,7 @@ inline const Squad * SquadLoad( uint8_t * region, int64_t region_bytes, const ui // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t SquadLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t SquadLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } @@ -4959,8 +4998,9 @@ inline int64_t SquadLoadMeasure( const TableVocabulary & vocabulary, const uint8 const uint8_t * const wire = message + 1; const int64_t wire_bytes = message_bytes - 1; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -4970,7 +5010,7 @@ inline int64_t SquadLoadMeasure( const TableVocabulary & vocabulary, const uint8 { records++; int64_t storage = SquadNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -5005,8 +5045,9 @@ inline const Squad * SquadLoadMessage( uint8_t * region, int64_t region_bytes, c int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !SquadWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Squad ) ) + root_extent ); int64_t records = 0; { @@ -5089,7 +5130,7 @@ inline const Squad * SquadLoadMessage( uint8_t * region, int64_t region_bytes, c // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Squad ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -5144,7 +5185,7 @@ inline bool SquadLoadBuilder( SquadBuilder & builder, const uint8_t * wire_file, nodes.entries = directory; nodes.count = records + 1; nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here - nodes.worker = &builder.main; // and a map's entries are the arena's, not a node extent's (§2.8) + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) { TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); int64_t k = 0; @@ -5183,10 +5224,14 @@ inline bool SquadLoadBuilder( SquadBuilder & builder, const uint8_t * wire_file, } TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.worker = &builder.main; nodes.carve = &root_carve; bool ok = SquadLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; allocator.free( allocator.context, directory ); return ok; } @@ -5272,7 +5317,7 @@ inline bool DepthBuilder::Lock() below = DepthPackMeasure( ctx, seen, root ); } if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it - int64_t root_extent = DepthMapExtent( ctx, root ); + int64_t root_extent = DepthExtent( ctx, root ); if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Depth ) ) + root_extent ) + below; // the AUTHORING path may allocate (§6.5), and it does so through the @@ -5370,10 +5415,11 @@ inline uint32_t DepthNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t // already owns. inline void DepthNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) { - // the node's own EXTENT, where its maps' entry arrays are carved from, - // PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8). The tool's - // path carries a worker instead: there the entries are the arena's. - TableMapCarve carve; + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; carve.worker = nodes.worker; if ( carve.worker == NULL ) { @@ -5522,7 +5568,7 @@ inline int64_t DepthSaveMessage( const DepthBuilder & builder, uint8_t * buffer, // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t DepthLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t DepthLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; TableIdTable ids_table; @@ -5535,8 +5581,9 @@ inline int64_t DepthLoadMeasure( const uint8_t * wire_file, int64_t wire_file_by const uint8_t * const wire = wire_file + 1; const int64_t wire_bytes = body_bytes; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Depth ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -5546,7 +5593,7 @@ inline int64_t DepthLoadMeasure( const uint8_t * wire_file, int64_t wire_file_by { records++; int64_t storage = DepthNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -5590,8 +5637,9 @@ inline const Depth * DepthLoad( uint8_t * region, int64_t region_bytes, const ui int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Depth ) ) + root_extent ); int64_t records = 0; { @@ -5674,7 +5722,7 @@ inline const Depth * DepthLoad( uint8_t * region, int64_t region_bytes, const ui // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Depth ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -5690,7 +5738,7 @@ inline const Depth * DepthLoad( uint8_t * region, int64_t region_bytes, const ui // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t DepthLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t DepthLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } @@ -5698,8 +5746,9 @@ inline int64_t DepthLoadMeasure( const TableVocabulary & vocabulary, const uint8 const uint8_t * const wire = message + 1; const int64_t wire_bytes = message_bytes - 1; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Depth ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -5709,7 +5758,7 @@ inline int64_t DepthLoadMeasure( const TableVocabulary & vocabulary, const uint8 { records++; int64_t storage = DepthNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -5744,8 +5793,9 @@ inline const Depth * DepthLoadMessage( uint8_t * region, int64_t region_bytes, c int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !DepthWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Depth ) ) + root_extent ); int64_t records = 0; { @@ -5828,7 +5878,7 @@ inline const Depth * DepthLoadMessage( uint8_t * region, int64_t region_bytes, c // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Depth ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -5883,7 +5933,7 @@ inline bool DepthLoadBuilder( DepthBuilder & builder, const uint8_t * wire_file, nodes.entries = directory; nodes.count = records + 1; nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here - nodes.worker = &builder.main; // and a map's entries are the arena's, not a node extent's (§2.8) + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) { TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); int64_t k = 0; @@ -5922,10 +5972,14 @@ inline bool DepthLoadBuilder( DepthBuilder & builder, const uint8_t * wire_file, } TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.worker = &builder.main; nodes.carve = &root_carve; bool ok = DepthLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; allocator.free( allocator.context, directory ); return ok; } @@ -6003,9 +6057,9 @@ inline void SquadRosterEntryCookBody( uint8_t * at, const SquadRosterEntry & val template inline bool SquadCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Squad & value, TableByteOrder order ) { - (void) ctx; (void) region; // no reference below this node: the class was decided by a pointer elsewhere in its closure + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure (void) value; - table_cook_put( at + 0, 0, 8, order ); // roster: the entry array's delta, filled by the extent writer + table_cook_put( at + 0, 0, 8, order ); // roster: the array's delta, filled by the extent writer table_cook_put( at + 8, 0, 4, order ); // and its count return true; } @@ -6041,23 +6095,25 @@ template inline bool DepthCookBody( const Ctx & ctx, const TableC return true; } -template inline bool SquadRosterEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const SquadRosterEntry & value, TableByteOrder order ); -template inline bool SquadCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Squad & value, TableByteOrder order ); -template inline bool DepthCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Depth & value, TableByteOrder order ); +template inline bool SquadRosterEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const SquadRosterEntry & value, TableByteOrder order ); +template inline bool SquadCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Squad & value, TableByteOrder order ); +template inline bool DepthCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Depth & value, TableByteOrder order ); -// SquadRosterEntryCookMaps: SquadRosterEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool SquadRosterEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const SquadRosterEntry & value, TableByteOrder order ) +// SquadRosterEntryCookExtent: SquadRosterEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SquadRosterEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const SquadRosterEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// SquadCookMaps: Squad's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool SquadCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Squad & value, TableByteOrder order ) +// SquadCookExtent: Squad's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool SquadCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Squad & value, TableByteOrder order ) { - (void) region; // a map's entries carry their own references through their own bodies + (void) region; // a table element's and an entry's references resolve through their own bodies { // roster TableMapCursor cursor = TableMapOrder( ctx, value.roster ); if ( !cursor.ok ) { return false; } @@ -6076,45 +6132,46 @@ template inline bool SquadCookMaps( const Ctx & ctx, const TableC return true; } -// DepthCookMaps: Depth's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool DepthCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Depth & value, TableByteOrder order ) +// DepthCookExtent: Depth's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool DepthCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Depth & value, TableByteOrder order ) { - (void) region; // a map's entries carry their own references through their own bodies - if ( !SquadCookMaps( ctx, region, extent, at, record + 0, value.one, order ) ) { return false; } // one + (void) region; // a table element's and an entry's references resolve through their own bodies + if ( !SquadCookExtent( ctx, region, extent, at, record + 0, value.one, order ) ) { return false; } // one for ( int32_t i = 0; i < ( value.many_count < 3 ? value.many_count : 3 ); i++ ) // many { - if ( !SquadCookMaps( ctx, region, extent, at, record + 16 + i * 16, value.many[i], order ) ) { return false; } + if ( !SquadCookExtent( ctx, region, extent, at, record + 16 + i * 16, value.many[i], order ) ) { return false; } } for ( int32_t i = 0; i < 2; i++ ) // keyed { - if ( !SquadCookMaps( ctx, region, extent, at, record + 72 + i * 16, value.keyed.slots[i], order ) ) { return false; } + if ( !SquadCookExtent( ctx, region, extent, at, record + 72 + i * 16, value.keyed.slots[i], order ) ) { return false; } } return true; } -// SquadRosterEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// SquadRosterEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool SquadRosterEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const SquadRosterEntry & value, TableByteOrder order ) { SquadRosterEntryCookBody( at, value, order ); int64_t extent_at = 0; - return SquadRosterEntryCookMaps( ctx, region, at + 8, extent_at, at, value, order ); + return SquadRosterEntryCookExtent( ctx, region, at + 8, extent_at, at, value, order ); } -// SquadCookNode: one node — the record, then the extent its maps take (§2.8). +// SquadCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool SquadCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Squad & value, TableByteOrder order ) { if ( !SquadCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return SquadCookMaps( ctx, region, at + 16, extent_at, at, value, order ); + return SquadCookExtent( ctx, region, at + 16, extent_at, at, value, order ); } -// DepthCookNode: one node — the record, then the extent its maps take (§2.8). +// DepthCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool DepthCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Depth & value, TableByteOrder order ) { if ( !DepthCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return DepthCookMaps( ctx, region, at + 136, extent_at, at, value, order ); + return DepthCookExtent( ctx, region, at + 136, extent_at, at, value, order ); } // SquadCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one @@ -6124,15 +6181,15 @@ template inline bool DepthCookNode( const Ctx & ctx, const TableC // eight. The offsets go into the region's table when it has one, and are only // summed when it does not (a measure). A type id the numbering carries that // this root cannot name is the two walks disagreeing, and it is refused. -// A NODE'S SIZE DEPENDS ON ITS VALUE where a map rides in its extent +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent // (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context -// the numbering walked and reads the same maps that walk read. +// the numbering walked and reads the same arrays that walk read. template inline bool SquadCookLayout( const Ctx & ctx, const Squad & root, const TableNumbering & numbering, TableCookRegion & region ) { region.numbering = &numbering; region.count = numbering.count + 1; - const int64_t root_extent = SquadMapExtent( ctx, root ); + const int64_t root_extent = SquadExtent( ctx, root ); if ( root_extent < 0 ) { return false; } int64_t offset = 16 + root_extent; // the root at zero, its extent behind it int64_t align = 8; @@ -6287,15 +6344,15 @@ inline bool SquadCook( const SquadBuilder & builder, void * out, uint64_t capaci // eight. The offsets go into the region's table when it has one, and are only // summed when it does not (a measure). A type id the numbering carries that // this root cannot name is the two walks disagreeing, and it is refused. -// A NODE'S SIZE DEPENDS ON ITS VALUE where a map rides in its extent +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent // (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context -// the numbering walked and reads the same maps that walk read. +// the numbering walked and reads the same arrays that walk read. template inline bool DepthCookLayout( const Ctx & ctx, const Depth & root, const TableNumbering & numbering, TableCookRegion & region ) { region.numbering = &numbering; region.count = numbering.count + 1; - const int64_t root_extent = DepthMapExtent( ctx, root ); + const int64_t root_extent = DepthExtent( ctx, root ); if ( root_extent < 0 ) { return false; } int64_t offset = 136 + root_extent; // the root at zero, its extent behind it int64_t align = 8; @@ -6502,24 +6559,24 @@ extern const TableTypeInfo SquadTableInfo; extern const TableTypeInfo DepthTableInfo; inline const TableFieldInfo SquadRosterEntryTableFields[] = { - { "key", "key", "uint8", 0x3dc94a19365b10ecull, 6, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( SquadRosterEntry, key ), (uint32_t) sizeof( SquadRosterEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( SquadRosterEntry, value ), (uint32_t) sizeof( SquadRosterEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "uint8", 0x3dc94a19365b10ecull, 6, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( SquadRosterEntry, key ), (uint32_t) sizeof( SquadRosterEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( SquadRosterEntry, value ), (uint32_t) sizeof( SquadRosterEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo SquadRosterEntryTableInfo = { "SquadRosterEntry", (uint32_t) sizeof( SquadRosterEntry ), 2, SquadRosterEntryTableFields, +[]( void * p ) { SquadRosterEntryReset( *(SquadRosterEntry *) p ); }, false }; inline const TableTypeInfo * SquadRosterEntryTableType() { return &SquadRosterEntryTableInfo; } inline const TableFieldInfo SquadTableFields[] = { - { "roster", "roster", "map[uint8]Item", 0x1c84390d304f4f42ull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Squad, roster ), (uint32_t) sizeof( Squad::roster ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &SquadRosterEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { SquadRosterEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint8_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint8_t) key_value ); } return (void *) placed; }, "" }, + { "roster", "roster", "map[uint8]Item", 0x1c84390d304f4f42ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Squad, roster ), (uint32_t) sizeof( SquadRosterEntry ), (uint32_t) offsetof( Squad, roster.count ), 0xffffffffu, &SquadRosterEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { SquadRosterEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint8_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint8_t) key_value ); } return (void *) placed; }, "" }, }; inline const TableTypeInfo SquadTableInfo = { "Squad", (uint32_t) sizeof( Squad ), 1, SquadTableFields, +[]( void * p ) { SquadReset( *(Squad *) p ); }, true }; inline const TableTypeInfo * SquadTableType() { return &SquadTableInfo; } inline const TableFieldInfo DepthTableFields[] = { - { "one", "one", "Squad", 0x1a08aa1921ca5cafull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Depth, one ), (uint32_t) sizeof( Depth::one ), 0xffffffffu, 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "many", "many", "Squad", 0x1f6459a2cea1fc02ull, 13, true, false, NULL, NULL, true, false, 3, (uint32_t) offsetof( Depth, many ), (uint32_t) sizeof( Depth::many[0] ), (uint32_t) offsetof( Depth, many_count ), 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "keyed", "keyed", "Squad", 0x70551ff29550f15dull, 13, true, false, NULL, NULL, false, false, (int32_t) Slot::Max, (uint32_t) offsetof( Depth, keyed ), (uint32_t) sizeof( Depth::keyed.slots[0] ), 0xffffffffu, 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, "Slot", +[]( uint64_t v ) { return EnumName( Slot( v ) ); }, +[]( uint64_t v ) -> uint64_t { uint64_t id = 0; TableEnumId( Slot( v ), id ); return id; }, NULL, NULL, NULL, NULL, NULL, "" }, - { "arm", "arm", "Force", 0xe756c0190570ccb5ull, 15, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Depth, arm ), (uint32_t) sizeof( Depth::arm ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, 2, +[]( uint64_t v ) -> const char * { switch ( v ) { case 0: return "None"; case 1: return "squad"; case 2: return "plain"; default: return "???"; } }, +[]( uint64_t v ) -> uint64_t { switch ( v ) { case 0: return 0; case 1: return 0xd5c2bb95d63e6331ull; case 2: return 0xfd4d194e1652b207ull; default: return 0; } }, NULL, NULL, NULL, +[]() -> const TableUnionInfo * { static const TableFieldInfo arm_fields_Force[] = { { "plain", "plain", "int32", 0xfd4d194e1652b207ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Force, plain ), (uint32_t) sizeof( Force::plain ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; static const TableUnionArmInfo arms[] = { { 0, NULL, NULL, 0 }, { (uint32_t) offsetof( Force, squad ), &SquadTableInfo, NULL, 16 }, { (uint32_t) offsetof( Force, plain ), NULL, &arm_fields_Force[0], 4 }, }; static const TableUnionInfo info = { (uint32_t) offsetof( Force, type ), (uint32_t) sizeof( Force::type ), arms }; return &info; }, NULL, NULL, NULL, NULL, "" }, - { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Depth, after ), (uint32_t) sizeof( Depth::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "one", "one", "Squad", 0x1a08aa1921ca5cafull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Depth, one ), (uint32_t) sizeof( Depth::one ), 0xffffffffu, 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "many", "many", "Squad", 0x1f6459a2cea1fc02ull, 13, true, false, NULL, NULL, true, false, 3, (uint32_t) offsetof( Depth, many ), (uint32_t) sizeof( Depth::many[0] ), (uint32_t) offsetof( Depth, many_count ), 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "keyed", "keyed", "Squad", 0x70551ff29550f15dull, 13, true, false, NULL, NULL, false, false, (int32_t) Slot::Max, (uint32_t) offsetof( Depth, keyed ), (uint32_t) sizeof( Depth::keyed.slots[0] ), 0xffffffffu, 0xffffffffu, &SquadTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, "Slot", +[]( uint64_t v ) { return EnumName( Slot( v ) ); }, +[]( uint64_t v ) -> uint64_t { uint64_t id = 0; TableEnumId( Slot( v ), id ); return id; }, NULL, NULL, "" }, + { "arm", "arm", "Force", 0xe756c0190570ccb5ull, 15, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Depth, arm ), (uint32_t) sizeof( Depth::arm ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, 2, +[]( uint64_t v ) -> const char * { switch ( v ) { case 0: return "None"; case 1: return "squad"; case 2: return "plain"; default: return "???"; } }, +[]( uint64_t v ) -> uint64_t { switch ( v ) { case 0: return 0; case 1: return 0xd5c2bb95d63e6331ull; case 2: return 0xfd4d194e1652b207ull; default: return 0; } }, NULL, NULL, NULL, +[]() -> const TableUnionInfo * { static const TableFieldInfo arm_fields_Force[] = { { "plain", "plain", "int32", 0xfd4d194e1652b207ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Force, plain ), (uint32_t) sizeof( Force::plain ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; static const TableUnionArmInfo arms[] = { { 0, NULL, NULL, 0 }, { (uint32_t) offsetof( Force, squad ), &SquadTableInfo, NULL, 16 }, { (uint32_t) offsetof( Force, plain ), NULL, &arm_fields_Force[0], 4 }, }; static const TableUnionInfo info = { (uint32_t) offsetof( Force, type ), (uint32_t) sizeof( Force::type ), arms }; return &info; }, NULL, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Depth, after ), (uint32_t) sizeof( Depth::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo DepthTableInfo = { "Depth", (uint32_t) sizeof( Depth ), 5, DepthTableFields, +[]( void * p ) { DepthReset( *(Depth *) p ); }, true }; inline const TableTypeInfo * DepthTableType() { return &DepthTableInfo; } diff --git a/testdata/golden/tables/maps/FleetTable.cpp b/testdata/golden/tables/maps/FleetTable.cpp index 8fea2bfe8..6e774adde 100644 --- a/testdata/golden/tables/maps/FleetTable.cpp +++ b/testdata/golden/tables/maps/FleetTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2731,14 +2748,33 @@ inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * inf // ---- json graph walk: end ---- +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + // ---- json map walk: begin ---- -inline bool TableJsonIsMap( const TableFieldInfo * f ) { return f->entry != NULL; } +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} // the entry's two rows: fields[0] IS the key and fields[1] IS the value, which // is what makes a user's own table of pairs the same bytes (§2.8) -inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->entry->fields[0]; } -inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->entry->fields[1]; } +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } @@ -2798,16 +2834,17 @@ inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const // A region holds them in that order already, so this is the array in place. inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) { - const int32_t count = f->map_count( slot ); + const int32_t count = TableJsonExtentCount( slot ); if ( count == 0 ) { out.raw( "{}", 2 ); return true; } const TableFieldInfo * key = TableJsonMapKeyField( f ); const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); out.put( '{' ); for ( int32_t i = 0; i < count; i++ ) { if ( i > 0 ) { out.put( ',' ); } out.line( depth + 1 ); - const void * entry = f->map_at( slot, i ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); TableJsonWriteMapKey( out, entry, key ); out.raw( ": ", 2 ); if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } @@ -2896,15 +2933,15 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf } if ( !fits ) { in.report->kind_mismatch++; place = false; } } - const int32_t before = f->map_count( (const void *) slot ); - void * entry = place ? f->map_insert( *graph->worker, slot, token, token_length, key_value ) : NULL; + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; if ( place && entry == NULL ) { // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the // wire's rule, because a clamped key is a merged entry (§2.8). in.report->clamped++; } - else if ( entry != NULL && f->map_count( (const void *) slot ) == before ) + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) { in.report->duplicate++; // last-wins, the object rule inside the map } @@ -2955,6 +2992,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf // ---- json map walk: end ---- +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace mapdemo #endif // MAPDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/maps/FleetTable.h b/testdata/golden/tables/maps/FleetTable.h index 3ef7f2542..057861ccb 100644 --- a/testdata/golden/tables/maps/FleetTable.h +++ b/testdata/golden/tables/maps/FleetTable.h @@ -109,6 +109,18 @@ struct TableReport TableMessageReason reason = newer_form; }; + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; // ---- reflection (tables only, docs/SPEC-TABLES.md) ---- // // Static field descriptors for every type in the table closure: name, wire @@ -224,18 +236,13 @@ struct TableFieldInfo // a function pointer at compile time; the arms themselves are a static // inside it). NULL for every other kind. const TableUnionInfo * (*arms)(); - // a MAP (docs/SPEC-TABLES.md §2.8): the generated ENTRY's descriptor — - // fields[0] is the key and fields[1] the value — and the three the ONE - // text walk cannot spell for itself, because TableMap is a type - // it has no name for. NULL on every field that is not a map. - const TableTypeInfo * entry; - int32_t ( * map_count )( const void * slot ); - const void * ( * map_at )( const void * slot, int32_t index ); - // place one entry BY KEY and hand back the entry, at its defaults: a - // string key comes in as the bytes and the length, an integer key as - // the value, and NULL is NOT INSERTED — a key past the bound, or an - // arena that could not carve another segment. - void * ( * map_insert )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded }; @@ -1291,8 +1298,8 @@ struct TableWorker return blob; } - // RAW, ZEROED storage of the bytes asked for, at the alignment asked for — a MAP's builder head and its - // entry segments (docs/SPEC-TABLES.md §2.8). It is not a node: it carries + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries // no type id, takes no index and has no Reset, so it goes through the same // slab and span the blob path uses rather than through Alloc. uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) @@ -1576,12 +1583,6 @@ static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts th // resolving through it yields NULL and can never fabricate the root. static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; -// What a node's storage answers when the FRAMING ITSELF is refused rather than -// merely unnameable: a map whose N cannot fit in its L (docs/SPEC-TABLES.md -// §2.8). An unnameable type id commands no storage and keeps its index; this -// one makes the whole measure answer -1 (§7.6). -static const int64_t kTableNodeRefused = -2; - // ---- the numbering, on the SAVE side ---- // // One entry per reachable node in FIRST-VISIT order, so entry k is node index @@ -1779,9 +1780,9 @@ struct TableNodeDirEntry uint64_t type_id; }; -// a map's extent cursor, defined with the map runtime (docs/SPEC-TABLES.md -// §2.8); the node map names it only through a pointer. -struct TableMapCarve; +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; // TableNodeMap is what a pointer slot resolves through while a body decodes. struct TableNodeMap @@ -1794,17 +1795,21 @@ struct TableNodeMap // takes the SELF-RELATIVE delta so a deref is one add, and the tool's // builder path takes the node's ARENA OFFSET (§6.3). bool arena = false; - // WHERE A MAP'S ENTRIES LAND while this node's body decodes - // (docs/SPEC-TABLES.md §2.8): the node's own extent on the region path - // and the builder's arena on the tool's. It is MUTABLE because the - // cursor belongs to ONE node's decode and the dispatch that owns that - // node holds the map by const reference, exactly as it did before maps - // existed — the decoder's signature does not move for a construct it - // may not carry. - mutable TableMapCarve * carve = NULL; - // and the TOOL's path's allocation front, set once: there a map's - // entries are the builder's arena's rather than a node's extent. + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; }; // TableNodeResolve places one node index in a pointer slot, and every failure @@ -1948,6 +1953,90 @@ inline bool TableNodeScanWhole( TableNodeScan & s ) #endif // MAPDEMO_SCHEMA_TABLE_ARENA +#ifndef MAPDEMO_SCHEMA_TABLE_EXTENT +#define MAPDEMO_SCHEMA_TABLE_EXTENT + +namespace mapdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace mapdemo + +#endif // MAPDEMO_SCHEMA_TABLE_EXTENT + #ifndef MAPDEMO_SCHEMA_TABLE_MAP #define MAPDEMO_SCHEMA_TABLE_MAP @@ -2378,12 +2467,6 @@ inline TableMapEach TableMapEachOf( const TableArena & arena, const Table return each; } -// AN UNREACHED SLOT MUST HOLD NO MAP WITH ENTRIES IN IT (§2.8, §7.6). An empty -// map takes no bytes, so a record whose extent measures ZERO is a record whose -// every by-value map is empty; a measure that REFUSED answers non-zero here -// too, and refusing on it is the same answer one level up. -inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } - // ---- the LOAD side: where a decoded entry lands (§2.8) ---- // // THE READER TRUSTS NOTHING and spends one compare per entry. Every load path @@ -2393,16 +2476,9 @@ inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } // out of the holder node's own extent, and the TOOL's path appends into the // builder's arena, and the decoder above them cannot tell which it has. -// TableMapCarve is a node's extent cursor, PRE-ORDER: a map's whole entry -// array first, then, entry by entry in key order, the arrays of any map an -// entry's value holds by value. The cursor is the node map's, because the -// generated decoder is threaded with that and not with a region. -struct TableMapCarve -{ - uint8_t * at = NULL; // the region path: the node's extent, unspent - int64_t left = 0; - TableWorker * worker = NULL; // the TOOL's path: entries come from the arena -}; +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. // TableMapFill is one map field being decoded: where the next entry lands, and // the entry that last LANDED, which is what the ascending check compares @@ -2527,10 +2603,11 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // LoadMeasure's term for a map is N x sizeof( Entry ) rounded to // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value -// holds a map of its own, the entries' headers under it. The caller owns the -// allocation precisely so it can refuse a number it did not expect. -typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ); - +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2538,8 +2615,8 @@ typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, i static const int64_t kTableMapEntryFloor = 2; inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, - int64_t entry_size, int64_t entry_align, TableMapWireExtentFn inner, - const TableIdTable * ids ) + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; TableReader r( body, length, &scratch, ids ); @@ -2547,8 +2624,9 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; - if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { return false; } // an N the map's L cannot carry + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); at += (int64_t) n * entry_size; if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term @@ -2556,49 +2634,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & { uint64_t elem = 0; if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// the same framing walk over an ARRAY OF TABLES that is not a map: its -// elements' own maps are part of this node's extent too -inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each -// length-prefixed element (docs/SPEC-TABLES.md §3.2) -inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t key = 0; - if ( !r.getleb( key ) ) { return true; } - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } r.offset += (int64_t) elem; } return true; @@ -3057,7 +3093,7 @@ inline void FleetLoadoutsEntryReset( FleetLoadoutsEntry & value ) { memset( value.key, 0, sizeof( value.key ) ); value.key_length = 0; - value.value.entries.value = 0; // map[uint8]Item — empty + value.value.entries.value = 0; // map[uint8]Item: empty value.value.count = 0; value.value.padding = 0; } @@ -3070,17 +3106,17 @@ inline void FleetTiersEntryReset( FleetTiersEntry & value ) inline void FleetReset( Fleet & value ) { - value.ships.entries.value = 0; // map[string(32)]ShipConfig — empty + value.ships.entries.value = 0; // map[string(32)]ShipConfig: empty value.ships.count = 0; value.ships.padding = 0; - value.by_id.entries.value = 0; // map[uint32]*ShipConfig — empty + value.by_id.entries.value = 0; // map[uint32]*ShipConfig: empty value.by_id.count = 0; value.by_id.padding = 0; value.flagship.value = 0; // *ShipConfig — null - value.loadouts.entries.value = 0; // map[string(16)]map[uint8]Item — empty + value.loadouts.entries.value = 0; // map[string(16)]map[uint8]Item: empty value.loadouts.count = 0; value.loadouts.padding = 0; - value.tiers.entries.value = 0; // map[int16]Item — empty + value.tiers.entries.value = 0; // map[int16]Item: empty value.tiers.count = 0; value.tiers.padding = 0; } @@ -3436,7 +3472,7 @@ struct FleetLoadoutsEntryEach { const char * key; decltype( TableEntryValue( (Fl inline FleetLoadoutsEntryEach TableEntryEach( FleetLoadoutsEntry * entry ) { return FleetLoadoutsEntryEach{ TableEntryKey( *entry ), TableEntryValue( entry ) }; } inline void TableResetMapValue( FleetLoadoutsEntry & value ) { - value.value.entries.value = 0; // map[uint8]Item — empty + value.value.entries.value = 0; // map[uint8]Item: empty value.value.count = 0; value.value.padding = 0; } @@ -5249,66 +5285,66 @@ inline bool FleetLoadBody( TableReader & r, const TableNodeMap & nodes, Fleet & } } -// ShipConfigWireExtent: the extent ShipConfig's maps command, from the FRAMING alone. +// ShipConfigWireExtent: the extent ShipConfig's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool ShipConfigWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool ShipConfigWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { - (void) body; (void) length; (void) at; (void) ids; // no map below this record + (void) body; (void) length; (void) at; (void) ids; (void) reason; // no list or map below this record return true; } -// ShipConfigMapExtentAt: the node extent ShipConfig's maps take, PRE-ORDER, advancing the -// running offset exactly as ShipConfigMapPack advances it (docs/SPEC-TABLES.md §2.8). +// ShipConfigExtentAt: the node extent ShipConfig's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as ShipConfigExtentPack advances it (§2.8, §2.9). template -inline bool ShipConfigMapExtentAt( const Ctx & ctx, const ShipConfig & value, int64_t & at ) +inline bool ShipConfigExtentAt( const Ctx & ctx, const ShipConfig & value, int64_t & at ) { - (void) ctx; (void) value; (void) at; // no map below this record + (void) ctx; (void) value; (void) at; // no list or map below this record return true; } -// ShipConfigMapPack: carve ShipConfig's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset ShipConfigMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// ShipConfigExtentPack: carve ShipConfig's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset ShipConfigExtentAt advances (§2.8, §2.9). template -inline bool ShipConfigMapPack( const Ctx & ctx, const ShipConfig & src, ShipConfig & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool ShipConfigExtentPack( const Ctx & ctx, const ShipConfig & src, ShipConfig & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { - (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no map below this record + (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no list or map below this record return true; } -// FleetByIdEntryWireExtent: the extent FleetByIdEntry's maps command, from the FRAMING alone. +// FleetByIdEntryWireExtent: the extent FleetByIdEntry's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool FleetByIdEntryWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool FleetByIdEntryWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { - (void) body; (void) length; (void) at; (void) ids; // no map below this record + (void) body; (void) length; (void) at; (void) ids; (void) reason; // no list or map below this record return true; } -// FleetByIdEntryMapExtentAt: the node extent FleetByIdEntry's maps take, PRE-ORDER, advancing the -// running offset exactly as FleetByIdEntryMapPack advances it (docs/SPEC-TABLES.md §2.8). +// FleetByIdEntryExtentAt: the node extent FleetByIdEntry's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as FleetByIdEntryExtentPack advances it (§2.8, §2.9). template -inline bool FleetByIdEntryMapExtentAt( const Ctx & ctx, const FleetByIdEntry & value, int64_t & at ) +inline bool FleetByIdEntryExtentAt( const Ctx & ctx, const FleetByIdEntry & value, int64_t & at ) { - (void) ctx; (void) value; (void) at; // no map below this record + (void) ctx; (void) value; (void) at; // no list or map below this record return true; } -// FleetByIdEntryMapPack: carve FleetByIdEntry's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset FleetByIdEntryMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// FleetByIdEntryExtentPack: carve FleetByIdEntry's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset FleetByIdEntryExtentAt advances (§2.8, §2.9). template -inline bool FleetByIdEntryMapPack( const Ctx & ctx, const FleetByIdEntry & src, FleetByIdEntry & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool FleetByIdEntryExtentPack( const Ctx & ctx, const FleetByIdEntry & src, FleetByIdEntry & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { - (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no map below this record + (void) ctx; (void) src; (void) dst; (void) extent; (void) at; (void) capacity; // no list or map below this record return true; } -// FleetLoadoutsEntryWireExtent: the extent FleetLoadoutsEntry's maps command, from the FRAMING alone. +// FleetLoadoutsEntryWireExtent: the extent FleetLoadoutsEntry's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool FleetLoadoutsEntryWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool FleetLoadoutsEntryWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; // the scan's framing damage is the LOAD's to report TableReader r( body, length, &scratch, ids ); @@ -5327,17 +5363,17 @@ inline bool FleetLoadoutsEntryWireExtent( const uint8_t * body, int64_t length, if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetLoadoutsEntryValueEntry ), (int64_t) alignof( FleetLoadoutsEntryValueEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetLoadoutsEntryValueEntry ), (int64_t) alignof( FleetLoadoutsEntryValueEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( !r.skip( field_kind ) ) { return true; } } } -// FleetLoadoutsEntryMapExtentAt: the node extent FleetLoadoutsEntry's maps take, PRE-ORDER, advancing the -// running offset exactly as FleetLoadoutsEntryMapPack advances it (docs/SPEC-TABLES.md §2.8). +// FleetLoadoutsEntryExtentAt: the node extent FleetLoadoutsEntry's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as FleetLoadoutsEntryExtentPack advances it (§2.8, §2.9). template -inline bool FleetLoadoutsEntryMapExtentAt( const Ctx & ctx, const FleetLoadoutsEntry & value, int64_t & at ) +inline bool FleetLoadoutsEntryExtentAt( const Ctx & ctx, const FleetLoadoutsEntry & value, int64_t & at ) { { TableMapCursor cursor = TableMapOrder( ctx, value.value ); @@ -5352,18 +5388,18 @@ inline bool FleetLoadoutsEntryMapExtentAt( const Ctx & ctx, const FleetLoadoutsE // the whole extent of one node, from a fresh offset: what a pack reserves // for it beside the record's own storage. template -inline int64_t FleetLoadoutsEntryMapExtent( const Ctx & ctx, const FleetLoadoutsEntry & value ) +inline int64_t FleetLoadoutsEntryExtent( const Ctx & ctx, const FleetLoadoutsEntry & value ) { int64_t at = 0; - if ( !FleetLoadoutsEntryMapExtentAt( ctx, value, at ) ) { return -1; } + if ( !FleetLoadoutsEntryExtentAt( ctx, value, at ) ) { return -1; } return at; } -// FleetLoadoutsEntryMapPack: carve FleetLoadoutsEntry's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset FleetLoadoutsEntryMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// FleetLoadoutsEntryExtentPack: carve FleetLoadoutsEntry's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset FleetLoadoutsEntryExtentAt advances (§2.8, §2.9). template -inline bool FleetLoadoutsEntryMapPack( const Ctx & ctx, const FleetLoadoutsEntry & src, FleetLoadoutsEntry & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool FleetLoadoutsEntryExtentPack( const Ctx & ctx, const FleetLoadoutsEntry & src, FleetLoadoutsEntry & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { { TableMapCursor cursor = TableMapOrder( ctx, src.value ); @@ -5385,10 +5421,10 @@ inline bool FleetLoadoutsEntryMapPack( const Ctx & ctx, const FleetLoadoutsEntry return true; } -// FleetWireExtent: the extent Fleet's maps command, from the FRAMING alone. +// FleetWireExtent: the extent Fleet's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool FleetWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool FleetWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; // the scan's framing damage is the LOAD's to report TableReader r( body, length, &scratch, ids ); @@ -5407,7 +5443,7 @@ inline bool FleetWireExtent( const uint8_t * body, int64_t length, int64_t & at, if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetShipsEntry ), (int64_t) alignof( FleetShipsEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetShipsEntry ), (int64_t) alignof( FleetShipsEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( field_id == 0x7b024c46e98d3404ull && field_kind == 14 ) // by_id @@ -5416,7 +5452,7 @@ inline bool FleetWireExtent( const uint8_t * body, int64_t length, int64_t & at, if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetByIdEntry ), (int64_t) alignof( FleetByIdEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetByIdEntry ), (int64_t) alignof( FleetByIdEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( field_id == 0x294fa1b3f0f5f070ull && field_kind == 14 ) // loadouts @@ -5425,7 +5461,7 @@ inline bool FleetWireExtent( const uint8_t * body, int64_t length, int64_t & at, if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetLoadoutsEntry ), (int64_t) alignof( FleetLoadoutsEntry ), &FleetLoadoutsEntryWireExtent, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetLoadoutsEntry ), (int64_t) alignof( FleetLoadoutsEntry ), &FleetLoadoutsEntryWireExtent, ids, reason ) ) { return false; } continue; } if ( field_id == 0x6dd8dc6c5fdae3ceull && field_kind == 14 ) // tiers @@ -5434,17 +5470,17 @@ inline bool FleetWireExtent( const uint8_t * body, int64_t length, int64_t & at, if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetTiersEntry ), (int64_t) alignof( FleetTiersEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( FleetTiersEntry ), (int64_t) alignof( FleetTiersEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( !r.skip( field_kind ) ) { return true; } } } -// FleetMapExtentAt: the node extent Fleet's maps take, PRE-ORDER, advancing the -// running offset exactly as FleetMapPack advances it (docs/SPEC-TABLES.md §2.8). +// FleetExtentAt: the node extent Fleet's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as FleetExtentPack advances it (§2.8, §2.9). template -inline bool FleetMapExtentAt( const Ctx & ctx, const Fleet & value, int64_t & at ) +inline bool FleetExtentAt( const Ctx & ctx, const Fleet & value, int64_t & at ) { { TableMapCursor cursor = TableMapOrder( ctx, value.ships ); @@ -5460,7 +5496,7 @@ inline bool FleetMapExtentAt( const Ctx & ctx, const Fleet & value, int64_t & at at += (int64_t) cursor.count * (int64_t) sizeof( FleetByIdEntry ); // the whole array FIRST for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order { - if ( !FleetByIdEntryMapExtentAt( ctx, *cursor[i], at ) ) { TableMapRelease( cursor ); return false; } + if ( !FleetByIdEntryExtentAt( ctx, *cursor[i], at ) ) { TableMapRelease( cursor ); return false; } } TableMapRelease( cursor ); } @@ -5471,7 +5507,7 @@ inline bool FleetMapExtentAt( const Ctx & ctx, const Fleet & value, int64_t & at at += (int64_t) cursor.count * (int64_t) sizeof( FleetLoadoutsEntry ); // the whole array FIRST for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order { - if ( !FleetLoadoutsEntryMapExtentAt( ctx, *cursor[i], at ) ) { TableMapRelease( cursor ); return false; } + if ( !FleetLoadoutsEntryExtentAt( ctx, *cursor[i], at ) ) { TableMapRelease( cursor ); return false; } } TableMapRelease( cursor ); } @@ -5488,18 +5524,18 @@ inline bool FleetMapExtentAt( const Ctx & ctx, const Fleet & value, int64_t & at // the whole extent of one node, from a fresh offset: what a pack reserves // for it beside the record's own storage. template -inline int64_t FleetMapExtent( const Ctx & ctx, const Fleet & value ) +inline int64_t FleetExtent( const Ctx & ctx, const Fleet & value ) { int64_t at = 0; - if ( !FleetMapExtentAt( ctx, value, at ) ) { return -1; } + if ( !FleetExtentAt( ctx, value, at ) ) { return -1; } return at; } -// FleetMapPack: carve Fleet's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset FleetMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// FleetExtentPack: carve Fleet's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset FleetExtentAt advances (§2.8, §2.9). template -inline bool FleetMapPack( const Ctx & ctx, const Fleet & src, Fleet & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool FleetExtentPack( const Ctx & ctx, const Fleet & src, Fleet & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { { TableMapCursor cursor = TableMapOrder( ctx, src.ships ); @@ -5535,7 +5571,7 @@ inline bool FleetMapPack( const Ctx & ctx, const Fleet & src, Fleet & dst, uint8 } for ( int32_t i = 0; i < cursor.count; i++ ) { - if ( !FleetByIdEntryMapPack( ctx, *cursor[i], placed[i], extent, at, capacity ) ) { TableMapRelease( cursor ); return false; } + if ( !FleetByIdEntryExtentPack( ctx, *cursor[i], placed[i], extent, at, capacity ) ) { TableMapRelease( cursor ); return false; } } TableMapRelease( cursor ); } @@ -5556,7 +5592,7 @@ inline bool FleetMapPack( const Ctx & ctx, const Fleet & src, Fleet & dst, uint8 } for ( int32_t i = 0; i < cursor.count; i++ ) { - if ( !FleetLoadoutsEntryMapPack( ctx, *cursor[i], placed[i], extent, at, capacity ) ) { TableMapRelease( cursor ); return false; } + if ( !FleetLoadoutsEntryExtentPack( ctx, *cursor[i], placed[i], extent, at, capacity ) ) { TableMapRelease( cursor ); return false; } } TableMapRelease( cursor ); } @@ -6121,7 +6157,7 @@ inline bool ShipConfigPack( const Ctx & ctx, TablePackMap & seen, const ShipConf int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( ShipConfig ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !ShipConfigMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !ShipConfigExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return ShipConfigPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -6220,7 +6256,7 @@ inline bool FleetByIdEntryPack( const Ctx & ctx, TablePackMap & seen, const Flee int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( FleetByIdEntry ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !FleetByIdEntryMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !FleetByIdEntryExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return FleetByIdEntryPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -6298,7 +6334,7 @@ inline bool FleetLoadoutsEntryPack( const Ctx & ctx, TablePackMap & seen, const int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( FleetLoadoutsEntry ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !FleetLoadoutsEntryMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !FleetLoadoutsEntryExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return FleetLoadoutsEntryPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -6417,7 +6453,7 @@ inline bool FleetPack( const Ctx & ctx, TablePackMap & seen, const Fleet & src, int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Fleet ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !FleetMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !FleetExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return FleetPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -6544,7 +6580,7 @@ inline bool FleetBuilder::Lock() below = FleetPackMeasure( ctx, seen, root ); } if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it - int64_t root_extent = FleetMapExtent( ctx, root ); + int64_t root_extent = FleetExtent( ctx, root ); if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Fleet ) ) + root_extent ) + below; // the AUTHORING path may allocate (§6.5), and it does so through the @@ -6644,10 +6680,11 @@ inline uint32_t FleetNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t // already owns. inline void FleetNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) { - // the node's own EXTENT, where its maps' entry arrays are carved from, - // PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8). The tool's - // path carries a worker instead: there the entries are the arena's. - TableMapCarve carve; + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; carve.worker = nodes.worker; if ( carve.worker == NULL ) { @@ -6797,7 +6834,7 @@ inline int64_t FleetSaveMessage( const FleetBuilder & builder, uint8_t * buffer, // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t FleetLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t FleetLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; TableIdTable ids_table; @@ -6810,8 +6847,9 @@ inline int64_t FleetLoadMeasure( const uint8_t * wire_file, int64_t wire_file_by const uint8_t * const wire = wire_file + 1; const int64_t wire_bytes = body_bytes; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Fleet ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -6821,7 +6859,7 @@ inline int64_t FleetLoadMeasure( const uint8_t * wire_file, int64_t wire_file_by { records++; int64_t storage = FleetNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -6865,8 +6903,9 @@ inline const Fleet * FleetLoad( uint8_t * region, int64_t region_bytes, const ui int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Fleet ) ) + root_extent ); int64_t records = 0; { @@ -6949,7 +6988,7 @@ inline const Fleet * FleetLoad( uint8_t * region, int64_t region_bytes, const ui // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Fleet ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -6965,7 +7004,7 @@ inline const Fleet * FleetLoad( uint8_t * region, int64_t region_bytes, const ui // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t FleetLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t FleetLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } @@ -6973,8 +7012,9 @@ inline int64_t FleetLoadMeasure( const TableVocabulary & vocabulary, const uint8 const uint8_t * const wire = message + 1; const int64_t wire_bytes = message_bytes - 1; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Fleet ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -6984,7 +7024,7 @@ inline int64_t FleetLoadMeasure( const TableVocabulary & vocabulary, const uint8 { records++; int64_t storage = FleetNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -7019,8 +7059,9 @@ inline const Fleet * FleetLoadMessage( uint8_t * region, int64_t region_bytes, c int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !FleetWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Fleet ) ) + root_extent ); int64_t records = 0; { @@ -7103,7 +7144,7 @@ inline const Fleet * FleetLoadMessage( uint8_t * region, int64_t region_bytes, c // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Fleet ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -7158,7 +7199,7 @@ inline bool FleetLoadBuilder( FleetBuilder & builder, const uint8_t * wire_file, nodes.entries = directory; nodes.count = records + 1; nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here - nodes.worker = &builder.main; // and a map's entries are the arena's, not a node extent's (§2.8) + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) { TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); int64_t k = 0; @@ -7197,10 +7238,14 @@ inline bool FleetLoadBuilder( FleetBuilder & builder, const uint8_t * wire_file, } TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.worker = &builder.main; nodes.carve = &root_carve; bool ok = FleetLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; allocator.free( allocator.context, directory ); return ok; } @@ -7333,10 +7378,10 @@ inline void FleetLoadoutsEntryValueEntryCookBody( uint8_t * at, const FleetLoado template inline bool FleetLoadoutsEntryCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const FleetLoadoutsEntry & value, TableByteOrder order ) { - (void) ctx; (void) region; // no reference below this node: the class was decided by a pointer elsewhere in its closure + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure table_cook_bytes( at + 0, value.key, value.key_length, 17 ); table_cook_put( at + 20, (uint64_t) (uint32_t) value.key_length, 4, order ); - table_cook_put( at + 24, 0, 8, order ); // value: the entry array's delta, filled by the extent writer + table_cook_put( at + 24, 0, 8, order ); // value: the array's delta, filled by the extent writer table_cook_put( at + 32, 0, 4, order ); // and its count return true; } @@ -7349,72 +7394,78 @@ inline void FleetTiersEntryCookBody( uint8_t * at, const FleetTiersEntry & value template inline bool FleetCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Fleet & value, TableByteOrder order ) { - table_cook_put( at + 0, 0, 8, order ); // ships: the entry array's delta, filled by the extent writer + table_cook_put( at + 0, 0, 8, order ); // ships: the array's delta, filled by the extent writer table_cook_put( at + 8, 0, 4, order ); // and its count - table_cook_put( at + 16, 0, 8, order ); // by_id: the entry array's delta, filled by the extent writer + table_cook_put( at + 16, 0, 8, order ); // by_id: the array's delta, filled by the extent writer table_cook_put( at + 24, 0, 4, order ); // and its count if ( !table_cook_ref( region, at + 32, (const void *) ShipConfigAt( ctx, value.flagship ), order ) ) { return false; } // flagship - table_cook_put( at + 40, 0, 8, order ); // loadouts: the entry array's delta, filled by the extent writer + table_cook_put( at + 40, 0, 8, order ); // loadouts: the array's delta, filled by the extent writer table_cook_put( at + 48, 0, 4, order ); // and its count - table_cook_put( at + 56, 0, 8, order ); // tiers: the entry array's delta, filled by the extent writer + table_cook_put( at + 56, 0, 8, order ); // tiers: the array's delta, filled by the extent writer table_cook_put( at + 64, 0, 4, order ); // and its count return true; } -template inline bool ShipConfigCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const ShipConfig & value, TableByteOrder order ); -template inline bool ItemCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Item & value, TableByteOrder order ); -template inline bool FleetShipsEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetShipsEntry & value, TableByteOrder order ); -template inline bool FleetByIdEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetByIdEntry & value, TableByteOrder order ); -template inline bool FleetLoadoutsEntryValueEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntryValueEntry & value, TableByteOrder order ); -template inline bool FleetLoadoutsEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntry & value, TableByteOrder order ); -template inline bool FleetTiersEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetTiersEntry & value, TableByteOrder order ); -template inline bool FleetCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Fleet & value, TableByteOrder order ); +template inline bool ShipConfigCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const ShipConfig & value, TableByteOrder order ); +template inline bool ItemCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Item & value, TableByteOrder order ); +template inline bool FleetShipsEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetShipsEntry & value, TableByteOrder order ); +template inline bool FleetByIdEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetByIdEntry & value, TableByteOrder order ); +template inline bool FleetLoadoutsEntryValueEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntryValueEntry & value, TableByteOrder order ); +template inline bool FleetLoadoutsEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntry & value, TableByteOrder order ); +template inline bool FleetTiersEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetTiersEntry & value, TableByteOrder order ); +template inline bool FleetCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Fleet & value, TableByteOrder order ); -// ShipConfigCookMaps: ShipConfig's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool ShipConfigCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const ShipConfig & value, TableByteOrder order ) +// ShipConfigCookExtent: ShipConfig's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool ShipConfigCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const ShipConfig & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// ItemCookMaps: Item's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool ItemCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Item & value, TableByteOrder order ) +// ItemCookExtent: Item's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool ItemCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Item & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// FleetShipsEntryCookMaps: FleetShipsEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool FleetShipsEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetShipsEntry & value, TableByteOrder order ) +// FleetShipsEntryCookExtent: FleetShipsEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FleetShipsEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetShipsEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// FleetByIdEntryCookMaps: FleetByIdEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool FleetByIdEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetByIdEntry & value, TableByteOrder order ) +// FleetByIdEntryCookExtent: FleetByIdEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FleetByIdEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetByIdEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// FleetLoadoutsEntryValueEntryCookMaps: FleetLoadoutsEntryValueEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool FleetLoadoutsEntryValueEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntryValueEntry & value, TableByteOrder order ) +// FleetLoadoutsEntryValueEntryCookExtent: FleetLoadoutsEntryValueEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FleetLoadoutsEntryValueEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntryValueEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// FleetLoadoutsEntryCookMaps: FleetLoadoutsEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool FleetLoadoutsEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntry & value, TableByteOrder order ) +// FleetLoadoutsEntryCookExtent: FleetLoadoutsEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FleetLoadoutsEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetLoadoutsEntry & value, TableByteOrder order ) { - (void) region; // a map's entries carry their own references through their own bodies + (void) region; // a table element's and an entry's references resolve through their own bodies { // value TableMapCursor cursor = TableMapOrder( ctx, value.value ); if ( !cursor.ok ) { return false; } @@ -7433,19 +7484,21 @@ template inline bool FleetLoadoutsEntryCookMaps( const Ctx & ctx, return true; } -// FleetTiersEntryCookMaps: FleetTiersEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool FleetTiersEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetTiersEntry & value, TableByteOrder order ) +// FleetTiersEntryCookExtent: FleetTiersEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FleetTiersEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const FleetTiersEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// FleetCookMaps: Fleet's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool FleetCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Fleet & value, TableByteOrder order ) +// FleetCookExtent: Fleet's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool FleetCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Fleet & value, TableByteOrder order ) { - (void) region; // a map's entries carry their own references through their own bodies + (void) region; // a table element's and an entry's references resolve through their own bodies { // ships TableMapCursor cursor = TableMapOrder( ctx, value.ships ); if ( !cursor.ok ) { return false; } @@ -7491,7 +7544,7 @@ template inline bool FleetCookMaps( const Ctx & ctx, const TableC } for ( int32_t i = 0; i < cursor.count; i++ ) // then, entry by entry in key order { - if ( !FleetLoadoutsEntryCookMaps( ctx, region, extent, at, array + i * 40, *cursor[i], order ) ) { TableMapRelease( cursor ); return false; } + if ( !FleetLoadoutsEntryCookExtent( ctx, region, extent, at, array + i * 40, *cursor[i], order ) ) { TableMapRelease( cursor ); return false; } } TableMapRelease( cursor ); } @@ -7513,68 +7566,68 @@ template inline bool FleetCookMaps( const Ctx & ctx, const TableC return true; } -// ShipConfigCookNode: one node — the record, then the extent its maps take (§2.8). +// ShipConfigCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool ShipConfigCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const ShipConfig & value, TableByteOrder order ) { ShipConfigCookBody( at, value, order ); int64_t extent_at = 0; - return ShipConfigCookMaps( ctx, region, at + 80, extent_at, at, value, order ); + return ShipConfigCookExtent( ctx, region, at + 80, extent_at, at, value, order ); } -// ItemCookNode: one node — the record, then the extent its maps take (§2.8). +// ItemCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool ItemCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Item & value, TableByteOrder order ) { ItemCookBody( at, value, order ); int64_t extent_at = 0; - return ItemCookMaps( ctx, region, at + 8, extent_at, at, value, order ); + return ItemCookExtent( ctx, region, at + 8, extent_at, at, value, order ); } -// FleetShipsEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// FleetShipsEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool FleetShipsEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const FleetShipsEntry & value, TableByteOrder order ) { FleetShipsEntryCookBody( at, value, order ); int64_t extent_at = 0; - return FleetShipsEntryCookMaps( ctx, region, at + 120, extent_at, at, value, order ); + return FleetShipsEntryCookExtent( ctx, region, at + 120, extent_at, at, value, order ); } -// FleetByIdEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// FleetByIdEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool FleetByIdEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const FleetByIdEntry & value, TableByteOrder order ) { if ( !FleetByIdEntryCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return FleetByIdEntryCookMaps( ctx, region, at + 16, extent_at, at, value, order ); + return FleetByIdEntryCookExtent( ctx, region, at + 16, extent_at, at, value, order ); } -// FleetLoadoutsEntryValueEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// FleetLoadoutsEntryValueEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool FleetLoadoutsEntryValueEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const FleetLoadoutsEntryValueEntry & value, TableByteOrder order ) { FleetLoadoutsEntryValueEntryCookBody( at, value, order ); int64_t extent_at = 0; - return FleetLoadoutsEntryValueEntryCookMaps( ctx, region, at + 8, extent_at, at, value, order ); + return FleetLoadoutsEntryValueEntryCookExtent( ctx, region, at + 8, extent_at, at, value, order ); } -// FleetLoadoutsEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// FleetLoadoutsEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool FleetLoadoutsEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const FleetLoadoutsEntry & value, TableByteOrder order ) { if ( !FleetLoadoutsEntryCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return FleetLoadoutsEntryCookMaps( ctx, region, at + 40, extent_at, at, value, order ); + return FleetLoadoutsEntryCookExtent( ctx, region, at + 40, extent_at, at, value, order ); } -// FleetTiersEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// FleetTiersEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool FleetTiersEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const FleetTiersEntry & value, TableByteOrder order ) { FleetTiersEntryCookBody( at, value, order ); int64_t extent_at = 0; - return FleetTiersEntryCookMaps( ctx, region, at + 8, extent_at, at, value, order ); + return FleetTiersEntryCookExtent( ctx, region, at + 8, extent_at, at, value, order ); } -// FleetCookNode: one node — the record, then the extent its maps take (§2.8). +// FleetCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool FleetCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Fleet & value, TableByteOrder order ) { if ( !FleetCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return FleetCookMaps( ctx, region, at + 72, extent_at, at, value, order ); + return FleetCookExtent( ctx, region, at + 72, extent_at, at, value, order ); } // ShipConfigCookMeasure: the whole cooked file's bytes — the header, the data part @@ -7688,15 +7741,15 @@ inline bool ItemCook( const Item & value, void * out, uint64_t capacity, TableBy // eight. The offsets go into the region's table when it has one, and are only // summed when it does not (a measure). A type id the numbering carries that // this root cannot name is the two walks disagreeing, and it is refused. -// A NODE'S SIZE DEPENDS ON ITS VALUE where a map rides in its extent +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent // (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context -// the numbering walked and reads the same maps that walk read. +// the numbering walked and reads the same arrays that walk read. template inline bool FleetCookLayout( const Ctx & ctx, const Fleet & root, const TableNumbering & numbering, TableCookRegion & region ) { region.numbering = &numbering; region.count = numbering.count + 1; - const int64_t root_extent = FleetMapExtent( ctx, root ); + const int64_t root_extent = FleetExtent( ctx, root ); if ( root_extent < 0 ) { return false; } int64_t offset = 72 + root_extent; // the root at zero, its extent behind it int64_t align = 8; @@ -7954,59 +8007,59 @@ extern const TableTypeInfo FleetTiersEntryTableInfo; extern const TableTypeInfo FleetTableInfo; inline const TableFieldInfo ShipConfigTableFields[] = { - { "name", "name", "string", 0xc4bcadba8e631b86ull, 12, false, false, NULL, NULL, true, false, 64, (uint32_t) offsetof( ShipConfig, name ), (uint32_t) sizeof( ShipConfig::name ), (uint32_t) offsetof( ShipConfig, name_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "health", "health", "int32", 0x7f69d4b5288ba9cfull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( ShipConfig, health ), (uint32_t) sizeof( ShipConfig::health ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "name", "name", "string", 0xc4bcadba8e631b86ull, 12, false, false, NULL, NULL, true, false, 64, (uint32_t) offsetof( ShipConfig, name ), (uint32_t) sizeof( ShipConfig::name ), (uint32_t) offsetof( ShipConfig, name_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "health", "health", "int32", 0x7f69d4b5288ba9cfull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( ShipConfig, health ), (uint32_t) sizeof( ShipConfig::health ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo ShipConfigTableInfo = { "ShipConfig", (uint32_t) sizeof( ShipConfig ), 2, ShipConfigTableFields, +[]( void * p ) { ShipConfigReset( *(ShipConfig *) p ); }, false }; inline const TableTypeInfo * ShipConfigTableType() { return &ShipConfigTableInfo; } inline const TableFieldInfo ItemTableFields[] = { - { "count", "count", "int32", 0xb1e5e28e4479a274ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Item, count ), (uint32_t) sizeof( Item::count ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "count", "count", "int32", 0xb1e5e28e4479a274ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Item, count ), (uint32_t) sizeof( Item::count ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo ItemTableInfo = { "Item", (uint32_t) sizeof( Item ), 1, ItemTableFields, +[]( void * p ) { ItemReset( *(Item *) p ); }, false }; inline const TableTypeInfo * ItemTableType() { return &ItemTableInfo; } inline const TableFieldInfo FleetShipsEntryTableFields[] = { - { "key", "key", "string", 0x3dc94a19365b10ecull, 12, false, false, NULL, NULL, true, false, 32, (uint32_t) offsetof( FleetShipsEntry, key ), (uint32_t) sizeof( FleetShipsEntry::key ), (uint32_t) offsetof( FleetShipsEntry, key_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "ShipConfig", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetShipsEntry, value ), (uint32_t) sizeof( FleetShipsEntry::value ), 0xffffffffu, 0xffffffffu, &ShipConfigTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "string", 0x3dc94a19365b10ecull, 12, false, false, NULL, NULL, true, false, 32, (uint32_t) offsetof( FleetShipsEntry, key ), (uint32_t) sizeof( FleetShipsEntry::key ), (uint32_t) offsetof( FleetShipsEntry, key_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "ShipConfig", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetShipsEntry, value ), (uint32_t) sizeof( FleetShipsEntry::value ), 0xffffffffu, 0xffffffffu, &ShipConfigTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo FleetShipsEntryTableInfo = { "FleetShipsEntry", (uint32_t) sizeof( FleetShipsEntry ), 2, FleetShipsEntryTableFields, +[]( void * p ) { FleetShipsEntryReset( *(FleetShipsEntry *) p ); }, false }; inline const TableTypeInfo * FleetShipsEntryTableType() { return &FleetShipsEntryTableInfo; } inline const TableFieldInfo FleetByIdEntryTableFields[] = { - { "key", "key", "uint32", 0x3dc94a19365b10ecull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetByIdEntry, key ), (uint32_t) sizeof( FleetByIdEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "ShipConfig", 0x7ce4fd9430e80ceaull, 17, false, true, []( const void * slot ) -> const void * { return (const void *) ShipConfigAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) ShipConfigEmplace( worker, *(TableRef *) slot ); }, false, false, 0, (uint32_t) offsetof( FleetByIdEntry, value ), (uint32_t) sizeof( TableRef ), 0xffffffffu, 0xffffffffu, &ShipConfigTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "uint32", 0x3dc94a19365b10ecull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetByIdEntry, key ), (uint32_t) sizeof( FleetByIdEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "ShipConfig", 0x7ce4fd9430e80ceaull, 17, false, true, []( const void * slot ) -> const void * { return (const void *) ShipConfigAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) ShipConfigEmplace( worker, *(TableRef *) slot ); }, false, false, 0, (uint32_t) offsetof( FleetByIdEntry, value ), (uint32_t) sizeof( TableRef ), 0xffffffffu, 0xffffffffu, &ShipConfigTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo FleetByIdEntryTableInfo = { "FleetByIdEntry", (uint32_t) sizeof( FleetByIdEntry ), 2, FleetByIdEntryTableFields, +[]( void * p ) { FleetByIdEntryReset( *(FleetByIdEntry *) p ); }, true }; inline const TableTypeInfo * FleetByIdEntryTableType() { return &FleetByIdEntryTableInfo; } inline const TableFieldInfo FleetLoadoutsEntryValueEntryTableFields[] = { - { "key", "key", "uint8", 0x3dc94a19365b10ecull, 6, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetLoadoutsEntryValueEntry, key ), (uint32_t) sizeof( FleetLoadoutsEntryValueEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetLoadoutsEntryValueEntry, value ), (uint32_t) sizeof( FleetLoadoutsEntryValueEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "uint8", 0x3dc94a19365b10ecull, 6, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetLoadoutsEntryValueEntry, key ), (uint32_t) sizeof( FleetLoadoutsEntryValueEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetLoadoutsEntryValueEntry, value ), (uint32_t) sizeof( FleetLoadoutsEntryValueEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo FleetLoadoutsEntryValueEntryTableInfo = { "FleetLoadoutsEntryValueEntry", (uint32_t) sizeof( FleetLoadoutsEntryValueEntry ), 2, FleetLoadoutsEntryValueEntryTableFields, +[]( void * p ) { FleetLoadoutsEntryValueEntryReset( *(FleetLoadoutsEntryValueEntry *) p ); }, false }; inline const TableTypeInfo * FleetLoadoutsEntryValueEntryTableType() { return &FleetLoadoutsEntryValueEntryTableInfo; } inline const TableFieldInfo FleetLoadoutsEntryTableFields[] = { - { "key", "key", "string", 0x3dc94a19365b10ecull, 12, false, false, NULL, NULL, true, false, 16, (uint32_t) offsetof( FleetLoadoutsEntry, key ), (uint32_t) sizeof( FleetLoadoutsEntry::key ), (uint32_t) offsetof( FleetLoadoutsEntry, key_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "map[uint8]Item", 0x7ce4fd9430e80ceaull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetLoadoutsEntry, value ), (uint32_t) sizeof( FleetLoadoutsEntry::value ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &FleetLoadoutsEntryValueEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { FleetLoadoutsEntryValueEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint8_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint8_t) key_value ); } return (void *) placed; }, "" }, + { "key", "key", "string", 0x3dc94a19365b10ecull, 12, false, false, NULL, NULL, true, false, 16, (uint32_t) offsetof( FleetLoadoutsEntry, key ), (uint32_t) sizeof( FleetLoadoutsEntry::key ), (uint32_t) offsetof( FleetLoadoutsEntry, key_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "map[uint8]Item", 0x7ce4fd9430e80ceaull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( FleetLoadoutsEntry, value ), (uint32_t) sizeof( FleetLoadoutsEntryValueEntry ), (uint32_t) offsetof( FleetLoadoutsEntry, value.count ), 0xffffffffu, &FleetLoadoutsEntryValueEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { FleetLoadoutsEntryValueEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint8_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint8_t) key_value ); } return (void *) placed; }, "" }, }; inline const TableTypeInfo FleetLoadoutsEntryTableInfo = { "FleetLoadoutsEntry", (uint32_t) sizeof( FleetLoadoutsEntry ), 2, FleetLoadoutsEntryTableFields, +[]( void * p ) { FleetLoadoutsEntryReset( *(FleetLoadoutsEntry *) p ); }, true }; inline const TableTypeInfo * FleetLoadoutsEntryTableType() { return &FleetLoadoutsEntryTableInfo; } inline const TableFieldInfo FleetTiersEntryTableFields[] = { - { "key", "key", "int16", 0x3dc94a19365b10ecull, 3, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetTiersEntry, key ), (uint32_t) sizeof( FleetTiersEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetTiersEntry, value ), (uint32_t) sizeof( FleetTiersEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "int16", 0x3dc94a19365b10ecull, 3, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetTiersEntry, key ), (uint32_t) sizeof( FleetTiersEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( FleetTiersEntry, value ), (uint32_t) sizeof( FleetTiersEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo FleetTiersEntryTableInfo = { "FleetTiersEntry", (uint32_t) sizeof( FleetTiersEntry ), 2, FleetTiersEntryTableFields, +[]( void * p ) { FleetTiersEntryReset( *(FleetTiersEntry *) p ); }, false }; inline const TableTypeInfo * FleetTiersEntryTableType() { return &FleetTiersEntryTableInfo; } inline const TableFieldInfo FleetTableFields[] = { - { "ships", "ships", "map[string(32)]ShipConfig", 0x294a5c4913e1ad44ull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Fleet, ships ), (uint32_t) sizeof( Fleet::ships ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &FleetShipsEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * { if ( key == NULL || key_length > kFleetShipsEntryKeyBound ) { return NULL; } FleetShipsEntry * placed = TableMapPlace( worker, *(TableMap *) slot, key ); if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }, "" }, - { "by_id", "by_id", "map[uint32]*ShipConfig", 0x7b024c46e98d3404ull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Fleet, by_id ), (uint32_t) sizeof( Fleet::by_id ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &FleetByIdEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { FleetByIdEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint32_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint32_t) key_value ); } return (void *) placed; }, "" }, - { "flagship", "flagship", "ShipConfig", 0x63dfa0c4a4b3815dull, 17, false, true, []( const void * slot ) -> const void * { return (const void *) ShipConfigAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) ShipConfigEmplace( worker, *(TableRef *) slot ); }, false, false, 0, (uint32_t) offsetof( Fleet, flagship ), (uint32_t) sizeof( TableRef ), 0xffffffffu, 0xffffffffu, &ShipConfigTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "loadouts", "loadouts", "map[string(16)]map[uint8]Item", 0x294fa1b3f0f5f070ull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Fleet, loadouts ), (uint32_t) sizeof( Fleet::loadouts ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &FleetLoadoutsEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * { if ( key == NULL || key_length > kFleetLoadoutsEntryKeyBound ) { return NULL; } FleetLoadoutsEntry * placed = TableMapPlace( worker, *(TableMap *) slot, key ); if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }, "" }, - { "tiers", "tiers", "map[int16]Item", 0x6dd8dc6c5fdae3ceull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Fleet, tiers ), (uint32_t) sizeof( Fleet::tiers ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &FleetTiersEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { FleetTiersEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (int16_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (int16_t) key_value ); } return (void *) placed; }, "" }, + { "ships", "ships", "map[string(32)]ShipConfig", 0x294a5c4913e1ad44ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Fleet, ships ), (uint32_t) sizeof( FleetShipsEntry ), (uint32_t) offsetof( Fleet, ships.count ), 0xffffffffu, &FleetShipsEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * { if ( key == NULL || key_length > kFleetShipsEntryKeyBound ) { return NULL; } FleetShipsEntry * placed = TableMapPlace( worker, *(TableMap *) slot, key ); if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }, "" }, + { "by_id", "by_id", "map[uint32]*ShipConfig", 0x7b024c46e98d3404ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Fleet, by_id ), (uint32_t) sizeof( FleetByIdEntry ), (uint32_t) offsetof( Fleet, by_id.count ), 0xffffffffu, &FleetByIdEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { FleetByIdEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint32_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint32_t) key_value ); } return (void *) placed; }, "" }, + { "flagship", "flagship", "ShipConfig", 0x63dfa0c4a4b3815dull, 17, false, true, []( const void * slot ) -> const void * { return (const void *) ShipConfigAt( *(const TableRef *) slot ); }, []( TableWorker & worker, void * slot ) -> void * { return (void *) ShipConfigEmplace( worker, *(TableRef *) slot ); }, false, false, 0, (uint32_t) offsetof( Fleet, flagship ), (uint32_t) sizeof( TableRef ), 0xffffffffu, 0xffffffffu, &ShipConfigTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "loadouts", "loadouts", "map[string(16)]map[uint8]Item", 0x294fa1b3f0f5f070ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Fleet, loadouts ), (uint32_t) sizeof( FleetLoadoutsEntry ), (uint32_t) offsetof( Fleet, loadouts.count ), 0xffffffffu, &FleetLoadoutsEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * { if ( key == NULL || key_length > kFleetLoadoutsEntryKeyBound ) { return NULL; } FleetLoadoutsEntry * placed = TableMapPlace( worker, *(TableMap *) slot, key ); if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }, "" }, + { "tiers", "tiers", "map[int16]Item", 0x6dd8dc6c5fdae3ceull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Fleet, tiers ), (uint32_t) sizeof( FleetTiersEntry ), (uint32_t) offsetof( Fleet, tiers.count ), 0xffffffffu, &FleetTiersEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { FleetTiersEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (int16_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (int16_t) key_value ); } return (void *) placed; }, "" }, }; inline const TableTypeInfo FleetTableInfo = { "Fleet", (uint32_t) sizeof( Fleet ), 5, FleetTableFields, +[]( void * p ) { FleetReset( *(Fleet *) p ); }, true }; inline const TableTypeInfo * FleetTableType() { return &FleetTableInfo; } diff --git a/testdata/golden/tables/maps/RowsTable.cpp b/testdata/golden/tables/maps/RowsTable.cpp index 5d6de21fa..537449dfd 100644 --- a/testdata/golden/tables/maps/RowsTable.cpp +++ b/testdata/golden/tables/maps/RowsTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2731,14 +2748,33 @@ inline int64_t TableJsonWriteGraph( const void * root, const TableTypeInfo * inf // ---- json graph walk: end ---- +// ---- the out-of-line array's slot (docs/SPEC-TABLES.md §8.1) ---- + +inline int32_t TableJsonExtentCount( const void * slot ) +{ + int32_t count = 0; + memcpy( &count, (const uint8_t *) slot + 8, sizeof( count ) ); + return count < 0 ? 0 : count; +} + +inline const uint8_t * TableJsonExtentElements( const void * slot ) +{ + int64_t delta = 0; + memcpy( &delta, slot, sizeof( delta ) ); + return delta != 0 ? (const uint8_t *) slot + delta : NULL; +} + // ---- json map walk: begin ---- -inline bool TableJsonIsMap( const TableFieldInfo * f ) { return f->entry != NULL; } +inline bool TableJsonIsMap( const TableFieldInfo * f ) +{ + return f->is_array && f->array_bound == 0 && strncmp( f->type_name, "map[", 4 ) == 0; +} // the entry's two rows: fields[0] IS the key and fields[1] IS the value, which // is what makes a user's own table of pairs the same bytes (§2.8) -inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->entry->fields[0]; } -inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->entry->fields[1]; } +inline const TableFieldInfo * TableJsonMapKeyField( const TableFieldInfo * f ) { return &f->table->fields[0]; } +inline const TableFieldInfo * TableJsonMapValueField( const TableFieldInfo * f ) { return &f->table->fields[1]; } inline bool TableJsonMapKeyIsString( const TableFieldInfo * key ) { return key->kind == 12; } inline bool TableJsonMapKeySigned( const TableFieldInfo * key ) { return key->kind >= 2 && key->kind <= 5; } @@ -2798,16 +2834,17 @@ inline void TableJsonWriteMapKey( TableJsonOut & out, const void * entry, const // A region holds them in that order already, so this is the array in place. inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ) { - const int32_t count = f->map_count( slot ); + const int32_t count = TableJsonExtentCount( slot ); if ( count == 0 ) { out.raw( "{}", 2 ); return true; } const TableFieldInfo * key = TableJsonMapKeyField( f ); const TableFieldInfo * value = TableJsonMapValueField( f ); + const uint8_t * entries = TableJsonExtentElements( slot ); out.put( '{' ); for ( int32_t i = 0; i < count; i++ ) { if ( i > 0 ) { out.put( ',' ); } out.line( depth + 1 ); - const void * entry = f->map_at( slot, i ); + const void * entry = (const void *) ( entries + (int64_t) i * f->elem_size ); TableJsonWriteMapKey( out, entry, key ); out.raw( ": ", 2 ); if ( !TableJsonWriteField( out, entry, value, depth + 1 ) ) { return false; } @@ -2896,15 +2933,15 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf } if ( !fits ) { in.report->kind_mismatch++; place = false; } } - const int32_t before = f->map_count( (const void *) slot ); - void * entry = place ? f->map_insert( *graph->worker, slot, token, token_length, key_value ) : NULL; + const int32_t before = TableJsonExtentCount( (const void *) slot ); + void * entry = place ? f->place( *graph->worker, slot, token, token_length, key_value ) : NULL; if ( place && entry == NULL ) { // A KEY LONGER THAN N DROPS ITS ENTRY AND COUNTS clamped, the // wire's rule, because a clamped key is a merged entry (§2.8). in.report->clamped++; } - else if ( entry != NULL && f->map_count( (const void *) slot ) == before ) + else if ( entry != NULL && TableJsonExtentCount( (const void *) slot ) == before ) { in.report->duplicate++; // last-wins, the object rule inside the map } @@ -2955,6 +2992,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInf // ---- json map walk: end ---- +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace mapdemo #endif // MAPDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/maps/RowsTable.h b/testdata/golden/tables/maps/RowsTable.h index 394e6212b..f538e3f9b 100644 --- a/testdata/golden/tables/maps/RowsTable.h +++ b/testdata/golden/tables/maps/RowsTable.h @@ -110,6 +110,18 @@ struct TableReport TableMessageReason reason = newer_form; }; + +// WHY A MEASURE WAS REFUSED, by name (docs/SPEC-TABLES.md §6.5): a -1 from +// LoadMeasure carries one of these as an out-parameter. A REFUSAL moves no +// counter: nothing was decoded, so there is nothing to report, and the reason +// is where the answer lives. The two values here are the ones a map's and an +// unbounded array's framing can raise. The rest of §6.5's vocabulary, the +// accelerators' refusals, is owed with them (schema#523). +enum TableRefuseReason +{ + count_over_length, // an array or map count whose elements cannot fit the field's own L (§2.8, §2.9) + count_over_extent_cap // a count above the int32 extent cap (§2.2), which no region can hold whatever its size +}; // ---- reflection (tables only, docs/SPEC-TABLES.md) ---- // // Static field descriptors for every type in the table closure: name, wire @@ -225,18 +237,13 @@ struct TableFieldInfo // a function pointer at compile time; the arms themselves are a static // inside it). NULL for every other kind. const TableUnionInfo * (*arms)(); - // a MAP (docs/SPEC-TABLES.md §2.8): the generated ENTRY's descriptor — - // fields[0] is the key and fields[1] the value — and the three the ONE - // text walk cannot spell for itself, because TableMap is a type - // it has no name for. NULL on every field that is not a map. - const TableTypeInfo * entry; - int32_t ( * map_count )( const void * slot ); - const void * ( * map_at )( const void * slot, int32_t index ); - // place one entry BY KEY and hand back the entry, at its defaults: a - // string key comes in as the bytes and the length, an integer key as - // the value, and NULL is NOT INSERTED — a key past the bound, or an - // arena that could not carve another segment. - void * ( * map_insert )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); + // an OUT-OF-LINE array (docs/SPEC-TABLES.md §8.1): place one element and + // hand it back at its defaults. A MAP places BY KEY, a string key comes + // in as the bytes and the length, an integer key as the value, and NULL + // is NOT INSERTED: a key past the bound, or an arena that could not carve + // another segment. A LIST ignores the key and APPENDS, NULL at the arena + // or the int32 cap. NULL on every field that is neither. + void * ( * place )( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t key_value ); const char * guard; // branch guard, e.g. "at_rest" or "!at_rest"; "" if unguarded }; @@ -1292,8 +1299,8 @@ struct TableWorker return blob; } - // RAW, ZEROED storage of the bytes asked for, at the alignment asked for — a MAP's builder head and its - // entry segments (docs/SPEC-TABLES.md §2.8). It is not a node: it carries + // RAW, ZEROED storage of the bytes asked for, at the alignment asked for: a MAP's or a LIST's builder + // head and its segments (docs/SPEC-TABLES.md §2.8, §2.9). It is not a node: it carries // no type id, takes no index and has no Reset, so it goes through the same // slab and span the blob path uses rather than through Alloc. uint8_t * AllocRaw( int64_t bytes, int64_t align, uint32_t & at ) @@ -1577,12 +1584,6 @@ static const uint64_t kTableNodeIndexRoot = 1; // the body that hosts th // resolving through it yields NULL and can never fabricate the root. static const uint64_t kTableNodeAbsent = 0xFFFFFFFFFFFFFFFFull; -// What a node's storage answers when the FRAMING ITSELF is refused rather than -// merely unnameable: a map whose N cannot fit in its L (docs/SPEC-TABLES.md -// §2.8). An unnameable type id commands no storage and keeps its index; this -// one makes the whole measure answer -1 (§7.6). -static const int64_t kTableNodeRefused = -2; - // ---- the numbering, on the SAVE side ---- // // One entry per reachable node in FIRST-VISIT order, so entry k is node index @@ -1780,9 +1781,9 @@ struct TableNodeDirEntry uint64_t type_id; }; -// a map's extent cursor, defined with the map runtime (docs/SPEC-TABLES.md -// §2.8); the node map names it only through a pointer. -struct TableMapCarve; +// the node's extent cursor, defined with the extent runtime (docs/SPEC-TABLES.md +// §2.8, §2.9); the node map names it only through a pointer. +struct TableExtentCarve; // TableNodeMap is what a pointer slot resolves through while a body decodes. struct TableNodeMap @@ -1795,17 +1796,21 @@ struct TableNodeMap // takes the SELF-RELATIVE delta so a deref is one add, and the tool's // builder path takes the node's ARENA OFFSET (§6.3). bool arena = false; - // WHERE A MAP'S ENTRIES LAND while this node's body decodes - // (docs/SPEC-TABLES.md §2.8): the node's own extent on the region path - // and the builder's arena on the tool's. It is MUTABLE because the - // cursor belongs to ONE node's decode and the dispatch that owns that - // node holds the map by const reference, exactly as it did before maps - // existed — the decoder's signature does not move for a construct it - // may not carry. - mutable TableMapCarve * carve = NULL; - // and the TOOL's path's allocation front, set once: there a map's - // entries are the builder's arena's rather than a node's extent. + // WHERE A MAP'S ENTRIES AND A LIST'S ELEMENTS LAND while this node's body + // decodes (docs/SPEC-TABLES.md §2.8, §2.9): the node's own extent on the + // region path and the builder's arena on the tool's. It is MUTABLE + // because the cursor belongs to ONE node's decode and the dispatch that + // owns that node holds the map by const reference, exactly as it did + // before either construct existed. The decoder's signature does not + // move for a construct it may not carry. + mutable TableExtentCarve * carve = NULL; + // and the TOOL's path's allocation front, set once: there the arrays + // are the builder's arena's rather than a node's extent. TableWorker * worker = NULL; + // THE TOOL PATH'S REFUSAL (docs/SPEC-TABLES.md §2.9): a count above the + // int32 cap met while a body decoded. LoadBuilder answers NULL for it + // and moves no counter; mutable for the reason the cursor is. + mutable bool refused = false; }; // TableNodeResolve places one node index in a pointer slot, and every failure @@ -1949,6 +1954,90 @@ inline bool TableNodeScanWhole( TableNodeScan & s ) #endif // MAPDEMO_SCHEMA_TABLE_ARENA +#ifndef MAPDEMO_SCHEMA_TABLE_EXTENT +#define MAPDEMO_SCHEMA_TABLE_EXTENT + +namespace mapdemo { + +// ---- the NODE EXTENT: where a map's entries and a list's elements live (§2.8, §2.9) ---- + +// What a node's storage answers when the FRAMING ITSELF is refused rather than +// merely unnameable: a count its L cannot carry, or one above the int32 cap +// (docs/SPEC-TABLES.md §6.5). An unnameable type id commands no storage and +// keeps its index. This one makes the whole measure answer -1 with its reason. +static const int64_t kTableNodeRefused = -2; + +// TableExtentCarve is a node's extent cursor, PRE-ORDER: a container's whole +// array first, then, element by element in the container's own order, the +// arrays of any list or map an element holds by value. The cursor is the node +// map's, because the generated decoder is threaded with that and not with a +// region. +struct TableExtentCarve +{ + uint8_t * at = NULL; // the region path: the node's extent, unspent + int64_t left = 0; + TableWorker * worker = NULL; // the TOOL's path: the arrays come from the arena +}; + +// AN UNREACHED SLOT MUST HOLD NO LIST OR MAP WITH ELEMENTS IN IT (§2.8, §2.9, +// §7.6). An empty one takes no bytes, so a record whose extent measures ZERO is +// a record whose every by-value list and map is empty. A measure that REFUSED +// answers non-zero here too, and refusing on it is the same answer one level up. +inline bool TableExtentUnreachedEmpty( int64_t extent ) { return extent == 0; } + +// ---- LoadMeasure's framing walk (§6.5) ---- +// +// The measure reads no field value: it walks each record's field headers, +// skipping every payload by its framing, to reach each N at every depth. A +// false is a REFUSAL, and it carries its reason (§6.5). +typedef bool ( * TableWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ); + +// the framing walk over an ARRAY OF TABLES held by value: its elements' own +// lists and maps are part of this node's extent too +inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each +// length-prefixed element (docs/SPEC-TABLES.md §3.2) +inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableWireExtentFn inner, const TableIdTable * ids, TableRefuseReason & reason ) +{ + TableReport scratch; + TableReader r( body, length, &scratch, ids ); + if ( length < 2 ) { return true; } + if ( r.get8() != 13 ) { return true; } + uint64_t n = 0; + if ( !r.getleb( n ) ) { return true; } + for ( uint64_t i = 0; i < n; i++ ) + { + uint64_t key = 0; + if ( !r.getleb( key ) ) { return true; } + uint64_t elem = 0; + if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } + r.offset += (int64_t) elem; + } + return true; +} + +} // namespace mapdemo + +#endif // MAPDEMO_SCHEMA_TABLE_EXTENT + #ifndef MAPDEMO_SCHEMA_TABLE_MAP #define MAPDEMO_SCHEMA_TABLE_MAP @@ -2379,12 +2468,6 @@ inline TableMapEach TableMapEachOf( const TableArena & arena, const Table return each; } -// AN UNREACHED SLOT MUST HOLD NO MAP WITH ENTRIES IN IT (§2.8, §7.6). An empty -// map takes no bytes, so a record whose extent measures ZERO is a record whose -// every by-value map is empty; a measure that REFUSED answers non-zero here -// too, and refusing on it is the same answer one level up. -inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } - // ---- the LOAD side: where a decoded entry lands (§2.8) ---- // // THE READER TRUSTS NOTHING and spends one compare per entry. Every load path @@ -2394,16 +2477,9 @@ inline bool TableMapUnreachedEmpty( int64_t extent ) { return extent == 0; } // out of the holder node's own extent, and the TOOL's path appends into the // builder's arena, and the decoder above them cannot tell which it has. -// TableMapCarve is a node's extent cursor, PRE-ORDER: a map's whole entry -// array first, then, entry by entry in key order, the arrays of any map an -// entry's value holds by value. The cursor is the node map's, because the -// generated decoder is threaded with that and not with a region. -struct TableMapCarve -{ - uint8_t * at = NULL; // the region path: the node's extent, unspent - int64_t left = 0; - TableWorker * worker = NULL; // the TOOL's path: entries come from the arena -}; +// The node's extent cursor is TableExtentCarve, the extent runtime's (§2.8, +// §2.9): a map's whole entry array is carved first, then, entry by entry in +// key order, the arrays of any list or map an entry's value holds by value. // TableMapFill is one map field being decoded: where the next entry lands, and // the entry that last LANDED, which is what the ascending check compares @@ -2528,10 +2604,11 @@ inline Entry * TableMapLive( const TableArena & arena, const TableMap & m // LoadMeasure's term for a map is N x sizeof( Entry ) rounded to // alignof( Entry ), AT EVERY DEPTH. N is framing and not a value, so this // reads no field: it walks the map's own header and, where an entry's value -// holds a map of its own, the entries' headers under it. The caller owns the -// allocation precisely so it can refuse a number it did not expect. -typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ); - +// holds a map or a list of its own, the entries' headers under it. The caller +// owns the allocation precisely so it can refuse a number it did not expect. +// Every -1 carries its REASON (§6.5): the int32 cap first, because a count +// past it cannot fit any body, and then the body's own L, the one rule a +// list's term answers by. // A MAP ENTRY'S SMALLEST WIRE FOOTPRINT that commands one storage unit is its // own L and the body's terminator, and under this form's variable lengths that // footprint is TWO BYTES (docs/SPEC-TABLES.md §4.2). It is what bounds the N a @@ -2539,8 +2616,8 @@ typedef bool ( * TableMapWireExtentFn )( const uint8_t * body, int64_t length, i static const int64_t kTableMapEntryFloor = 2; inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & at, - int64_t entry_size, int64_t entry_align, TableMapWireExtentFn inner, - const TableIdTable * ids ) + int64_t entry_size, int64_t entry_align, TableWireExtentFn inner, + const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; TableReader r( body, length, &scratch, ids ); @@ -2548,8 +2625,9 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & if ( r.get8() != 13 ) { return true; } // not an array of tables: §4's ordinary kind mismatch uint64_t n = 0; if ( !r.getleb( n ) ) { return true; } + if ( n > (uint64_t) INT32_MAX ) { reason = count_over_extent_cap; return false; } const int64_t rest = length - r.offset; - if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { return false; } // an N the map's L cannot carry + if ( n > (uint64_t) ( rest / kTableMapEntryFloor ) ) { reason = count_over_length; return false; } // an N the map's L cannot carry at = ( at + entry_align - 1 ) & ~( entry_align - 1 ); at += (int64_t) n * entry_size; if ( inner == NULL ) { return true; } // no map below an entry: one depth is the whole term @@ -2557,49 +2635,7 @@ inline bool TableMapWireExtent( const uint8_t * body, int64_t length, int64_t & { uint64_t elem = 0; if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } // framing damage: the load reports it - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// the same framing walk over an ARRAY OF TABLES that is not a map: its -// elements' own maps are part of this node's extent too -inline bool TableWireExtentElements( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } - r.offset += (int64_t) elem; - } - return true; -} - -// and over an ENUM-KEYED array, whose triples carry a key REFERENCE before each -// length-prefixed element (docs/SPEC-TABLES.md §3.2) -inline bool TableWireExtentKeyed( const uint8_t * body, int64_t length, int64_t & at, TableMapWireExtentFn inner, const TableIdTable * ids ) -{ - TableReport scratch; - TableReader r( body, length, &scratch, ids ); - if ( length < 2 ) { return true; } - if ( r.get8() != 13 ) { return true; } - uint64_t n = 0; - if ( !r.getleb( n ) ) { return true; } - for ( uint64_t i = 0; i < n; i++ ) - { - uint64_t key = 0; - if ( !r.getleb( key ) ) { return true; } - uint64_t elem = 0; - if ( !r.getleb( elem ) || !r.room( elem ) ) { return true; } - if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids ) ) { return false; } + if ( !inner( r.buffer + r.offset, (int64_t) elem, at, ids, reason ) ) { return false; } r.offset += (int64_t) elem; } return true; @@ -2996,7 +3032,7 @@ inline void RowEntriesEntryReset( RowEntriesEntry & value ) inline void RowReset( Row & value ) { - value.entries.entries.value = 0; // map[string(8)]Item — empty + value.entries.entries.value = 0; // map[string(8)]Item: empty value.entries.count = 0; value.entries.padding = 0; value.after = 0; @@ -3010,7 +3046,7 @@ inline void WideRowEntriesEntryReset( WideRowEntriesEntry & value ) inline void WideRowReset( WideRow & value ) { - value.entries.entries.value = 0; // map[uint32]Item — empty + value.entries.entries.value = 0; // map[uint32]Item: empty value.entries.count = 0; value.entries.padding = 0; value.after = 0; @@ -3868,10 +3904,10 @@ inline bool WideRowLoadBody( TableReader & r, const TableNodeMap & nodes, WideRo } } -// RowWireExtent: the extent Row's maps command, from the FRAMING alone. +// RowWireExtent: the extent Row's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool RowWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool RowWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; // the scan's framing damage is the LOAD's to report TableReader r( body, length, &scratch, ids ); @@ -3890,17 +3926,17 @@ inline bool RowWireExtent( const uint8_t * body, int64_t length, int64_t & at, c if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( RowEntriesEntry ), (int64_t) alignof( RowEntriesEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( RowEntriesEntry ), (int64_t) alignof( RowEntriesEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( !r.skip( field_kind ) ) { return true; } } } -// RowMapExtentAt: the node extent Row's maps take, PRE-ORDER, advancing the -// running offset exactly as RowMapPack advances it (docs/SPEC-TABLES.md §2.8). +// RowExtentAt: the node extent Row's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as RowExtentPack advances it (§2.8, §2.9). template -inline bool RowMapExtentAt( const Ctx & ctx, const Row & value, int64_t & at ) +inline bool RowExtentAt( const Ctx & ctx, const Row & value, int64_t & at ) { { TableMapCursor cursor = TableMapOrder( ctx, value.entries ); @@ -3915,18 +3951,18 @@ inline bool RowMapExtentAt( const Ctx & ctx, const Row & value, int64_t & at ) // the whole extent of one node, from a fresh offset: what a pack reserves // for it beside the record's own storage. template -inline int64_t RowMapExtent( const Ctx & ctx, const Row & value ) +inline int64_t RowExtent( const Ctx & ctx, const Row & value ) { int64_t at = 0; - if ( !RowMapExtentAt( ctx, value, at ) ) { return -1; } + if ( !RowExtentAt( ctx, value, at ) ) { return -1; } return at; } -// RowMapPack: carve Row's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset RowMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// RowExtentPack: carve Row's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset RowExtentAt advances (§2.8, §2.9). template -inline bool RowMapPack( const Ctx & ctx, const Row & src, Row & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool RowExtentPack( const Ctx & ctx, const Row & src, Row & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { { TableMapCursor cursor = TableMapOrder( ctx, src.entries ); @@ -3948,10 +3984,10 @@ inline bool RowMapPack( const Ctx & ctx, const Row & src, Row & dst, uint8_t * e return true; } -// WideRowWireExtent: the extent WideRow's maps command, from the FRAMING alone. +// WideRowWireExtent: the extent WideRow's lists and maps command, from the FRAMING alone. // It reads no field value, so a caller can refuse a number it did not // expect before one byte is allocated (docs/SPEC-TABLES.md §6.5). -inline bool WideRowWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids ) +inline bool WideRowWireExtent( const uint8_t * body, int64_t length, int64_t & at, const TableIdTable * ids, TableRefuseReason & reason ) { TableReport scratch; // the scan's framing damage is the LOAD's to report TableReader r( body, length, &scratch, ids ); @@ -3970,17 +4006,17 @@ inline bool WideRowWireExtent( const uint8_t * body, int64_t length, int64_t & a if ( !r.getleb( map_len ) || !r.room( map_len ) ) { return true; } const uint8_t * map_body = r.buffer + r.offset; r.offset += (int64_t) map_len; - if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( WideRowEntriesEntry ), (int64_t) alignof( WideRowEntriesEntry ), NULL, ids ) ) { return false; } + if ( !TableMapWireExtent( map_body, (int64_t) map_len, at, (int64_t) sizeof( WideRowEntriesEntry ), (int64_t) alignof( WideRowEntriesEntry ), NULL, ids, reason ) ) { return false; } continue; } if ( !r.skip( field_kind ) ) { return true; } } } -// WideRowMapExtentAt: the node extent WideRow's maps take, PRE-ORDER, advancing the -// running offset exactly as WideRowMapPack advances it (docs/SPEC-TABLES.md §2.8). +// WideRowExtentAt: the node extent WideRow's lists and maps take, PRE-ORDER, advancing +// the running offset exactly as WideRowExtentPack advances it (§2.8, §2.9). template -inline bool WideRowMapExtentAt( const Ctx & ctx, const WideRow & value, int64_t & at ) +inline bool WideRowExtentAt( const Ctx & ctx, const WideRow & value, int64_t & at ) { { TableMapCursor cursor = TableMapOrder( ctx, value.entries ); @@ -3995,18 +4031,18 @@ inline bool WideRowMapExtentAt( const Ctx & ctx, const WideRow & value, int64_t // the whole extent of one node, from a fresh offset: what a pack reserves // for it beside the record's own storage. template -inline int64_t WideRowMapExtent( const Ctx & ctx, const WideRow & value ) +inline int64_t WideRowExtent( const Ctx & ctx, const WideRow & value ) { int64_t at = 0; - if ( !WideRowMapExtentAt( ctx, value, at ) ) { return -1; } + if ( !WideRowExtentAt( ctx, value, at ) ) { return -1; } return at; } -// WideRowMapPack: carve WideRow's map arrays out of the node's extent and copy the -// entries in ASCENDING key order, PRE-ORDER, advancing the same running -// offset WideRowMapExtentAt advances (docs/SPEC-TABLES.md §2.8). +// WideRowExtentPack: carve WideRow's arrays out of the node's extent and copy the +// entries in ASCENDING key order and the elements in INDEX order, PRE-ORDER, +// advancing the same running offset WideRowExtentAt advances (§2.8, §2.9). template -inline bool WideRowMapPack( const Ctx & ctx, const WideRow & src, WideRow & dst, uint8_t * extent, int64_t & at, int64_t capacity ) +inline bool WideRowExtentPack( const Ctx & ctx, const WideRow & src, WideRow & dst, uint8_t * extent, int64_t & at, int64_t capacity ) { { TableMapCursor cursor = TableMapOrder( ctx, src.entries ); @@ -4270,7 +4306,7 @@ inline bool RowPack( const Ctx & ctx, TablePackMap & seen, const Row & src, Row int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( Row ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !RowMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !RowExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return RowPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -4323,7 +4359,7 @@ inline bool WideRowPack( const Ctx & ctx, TablePackMap & seen, const WideRow & s int64_t at = 0; uint8_t * extent = (uint8_t *) &dst + TableAlignUp64( (int64_t) sizeof( WideRow ) ); const int64_t room = capacity - ( (int64_t) ( extent - base ) ); - if ( !WideRowMapPack( ctx, src, dst, extent, at, room ) ) { return false; } + if ( !WideRowExtentPack( ctx, src, dst, extent, at, room ) ) { return false; } return WideRowPackEdges( ctx, seen, src, dst, base, capacity, used ); } @@ -4415,7 +4451,7 @@ inline bool RowBuilder::Lock() below = RowPackMeasure( ctx, seen, root ); } if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it - int64_t root_extent = RowMapExtent( ctx, root ); + int64_t root_extent = RowExtent( ctx, root ); if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ) + below; // the AUTHORING path may allocate (§6.5), and it does so through the @@ -4513,10 +4549,11 @@ inline uint32_t RowNodeAlloc( uint64_t type_id, TableWorker & worker, int64_t le // already owns. inline void RowNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) { - // the node's own EXTENT, where its maps' entry arrays are carved from, - // PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8). The tool's - // path carries a worker instead: there the entries are the arena's. - TableMapCarve carve; + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; carve.worker = nodes.worker; if ( carve.worker == NULL ) { @@ -4665,7 +4702,7 @@ inline int64_t RowSaveMessage( const RowBuilder & builder, uint8_t * buffer, int // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t RowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t RowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; TableIdTable ids_table; @@ -4678,8 +4715,9 @@ inline int64_t RowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_byte const uint8_t * const wire = wire_file + 1; const int64_t wire_bytes = body_bytes; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -4689,7 +4727,7 @@ inline int64_t RowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_byte { records++; int64_t storage = RowNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -4733,8 +4771,9 @@ inline const Row * RowLoad( uint8_t * region, int64_t region_bytes, const uint8_ int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); int64_t records = 0; { @@ -4817,7 +4856,7 @@ inline const Row * RowLoad( uint8_t * region, int64_t region_bytes, const uint8_ // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Row ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -4833,7 +4872,7 @@ inline const Row * RowLoad( uint8_t * region, int64_t region_bytes, const uint8_ // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t RowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t RowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } @@ -4841,8 +4880,9 @@ inline int64_t RowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t const uint8_t * const wire = message + 1; const int64_t wire_bytes = message_bytes - 1; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -4852,7 +4892,7 @@ inline int64_t RowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t { records++; int64_t storage = RowNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -4887,8 +4927,9 @@ inline const Row * RowLoadMessage( uint8_t * region, int64_t region_bytes, const int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !RowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( Row ) ) + root_extent ); int64_t records = 0; { @@ -4971,7 +5012,7 @@ inline const Row * RowLoadMessage( uint8_t * region, int64_t region_bytes, const // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( Row ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -5026,7 +5067,7 @@ inline bool RowLoadBuilder( RowBuilder & builder, const uint8_t * wire_file, int nodes.entries = directory; nodes.count = records + 1; nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here - nodes.worker = &builder.main; // and a map's entries are the arena's, not a node extent's (§2.8) + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) { TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); int64_t k = 0; @@ -5065,10 +5106,14 @@ inline bool RowLoadBuilder( RowBuilder & builder, const uint8_t * wire_file, int } TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.worker = &builder.main; nodes.carve = &root_carve; bool ok = RowLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; allocator.free( allocator.context, directory ); return ok; } @@ -5154,7 +5199,7 @@ inline bool WideRowBuilder::Lock() below = WideRowPackMeasure( ctx, seen, root ); } if ( below < 0 ) { TablePackMapShutdown( seen ); return false; } // a data cycle, named at the reference that closes it - int64_t root_extent = WideRowMapExtent( ctx, root ); + int64_t root_extent = WideRowExtent( ctx, root ); if ( root_extent < 0 ) { TablePackMapShutdown( seen ); return false; } // the sort could not run int64_t total = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( WideRow ) ) + root_extent ) + below; // the AUTHORING path may allocate (§6.5), and it does so through the @@ -5252,10 +5297,11 @@ inline uint32_t WideRowNodeAlloc( uint64_t type_id, TableWorker & worker, int64_ // already owns. inline void WideRowNodeBody( uint64_t type_id, TableReader & r, const TableNodeMap & nodes, uint8_t * at ) { - // the node's own EXTENT, where its maps' entry arrays are carved from, - // PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8). The tool's - // path carries a worker instead: there the entries are the arena's. - TableMapCarve carve; + // the node's own EXTENT, where its lists' and maps' arrays are carved + // from, PRE-ORDER as the bodies decode (docs/SPEC-TABLES.md §2.8, §2.9). + // The tool's path carries a worker instead: there the arrays are the + // arena's. + TableExtentCarve carve; carve.worker = nodes.worker; if ( carve.worker == NULL ) { @@ -5404,7 +5450,7 @@ inline int64_t WideRowSaveMessage( const WideRowBuilder & builder, uint8_t * buf // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t WideRowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t WideRowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; TableIdTable ids_table; @@ -5417,8 +5463,9 @@ inline int64_t WideRowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_ const uint8_t * const wire = wire_file + 1; const int64_t wire_bytes = body_bytes; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( WideRow ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -5428,7 +5475,7 @@ inline int64_t WideRowLoadMeasure( const uint8_t * wire_file, int64_t wire_file_ { records++; int64_t storage = WideRowNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -5472,8 +5519,9 @@ inline const WideRow * WideRowLoad( uint8_t * region, int64_t region_bytes, cons int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( WideRow ) ) + root_extent ); int64_t records = 0; { @@ -5556,7 +5604,7 @@ inline const WideRow * WideRowLoad( uint8_t * region, int64_t region_bytes, cons // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( WideRow ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -5572,7 +5620,7 @@ inline const WideRow * WideRowLoad( uint8_t * region, int64_t region_bytes, cons // It reports the DATA bytes and the ATTRIBUTION bytes separately, because the // attribution is the wire's numbering made resident (§6.3) and a caller may // release it once Load returns. The answer is their sum. -inline int64_t WideRowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL ) +inline int64_t WideRowLoadMeasure( const TableVocabulary & vocabulary, const uint8_t * message, int64_t message_bytes, int64_t * attribution_bytes = NULL, TableRefuseReason * reason_out = NULL ) { TableReport ignored; if ( message_bytes < 1 || message[0] != kTableWireMessageForm || !vocabulary.announced ) { return -1; } @@ -5580,8 +5628,9 @@ inline int64_t WideRowLoadMeasure( const TableVocabulary & vocabulary, const uin const uint8_t * const wire = message + 1; const int64_t wire_bytes = message_bytes - 1; TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, &ignored, &ids_table ); + TableRefuseReason reason = count_over_length; int64_t root_extent = 0; - if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { return -1; } + if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( WideRow ) ) + root_extent ); int64_t records = 0; uint64_t type_id = 0; @@ -5591,7 +5640,7 @@ inline int64_t WideRowLoadMeasure( const TableVocabulary & vocabulary, const uin { records++; int64_t storage = WideRowNodeStorage( type_id, length ); - if ( storage == kTableNodeRefused ) { return -1; } // an N the record's framing cannot carry (§2.8) + if ( storage == kTableNodeRefused ) { if ( reason_out != NULL ) { *reason_out = reason; } return -1; } // an N the record's framing cannot carry (§2.8, §2.9) if ( storage > 0 ) { data += storage; } // a type id this build cannot name commands none } int64_t attribution = ( records + 1 ) * (int64_t) sizeof( TableNodeDirEntry ); @@ -5626,8 +5675,9 @@ inline const WideRow * WideRowLoadMessage( uint8_t * region, int64_t region_byte int64_t length = 0; // the record count and the data bytes, from the FRAMING alone + TableRefuseReason reason = count_over_length; // LoadMeasure is where a caller reads it; a Load past a refusal is malformed int64_t root_extent = 0; - if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table ) ) { out->malformed = true; return NULL; } + if ( !WideRowWireExtent( wire, wire_bytes, root_extent, &ids_table, reason ) ) { out->malformed = true; return NULL; } int64_t data = TableAlignUp64( TableAlignUp64( (int64_t) sizeof( WideRow ) ) + root_extent ); int64_t records = 0; { @@ -5710,7 +5760,7 @@ inline const WideRow * WideRowLoadMessage( uint8_t * region, int64_t region_byte // against a numbering already known good or already known bad TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.at = region + TableAlignUp64( (int64_t) sizeof( WideRow ) ); root_carve.left = root_extent; nodes.carve = &root_carve; // the ROOT's extent is its own, like every node's @@ -5765,7 +5815,7 @@ inline bool WideRowLoadBuilder( WideRowBuilder & builder, const uint8_t * wire_f nodes.entries = directory; nodes.count = records + 1; nodes.arena = true; // a resolved slot holds the node's ARENA OFFSET here - nodes.worker = &builder.main; // and a map's entries are the arena's, not a node extent's (§2.8) + nodes.worker = &builder.main; // and a map's entries and a list's elements are the arena's, not a node extent's (§2.8, §2.9) { TableNodeScan scan = TableNodeScanBegin( wire, wire_bytes, out, &ids_table ); int64_t k = 0; @@ -5804,10 +5854,14 @@ inline bool WideRowLoadBuilder( WideRowBuilder & builder, const uint8_t * wire_f } TableReader r( wire, wire_bytes, out, &ids_table ); r.nested = false; // the ROOT body, the one that carries the node table - TableMapCarve root_carve; + TableExtentCarve root_carve; root_carve.worker = &builder.main; nodes.carve = &root_carve; bool ok = WideRowLoadBody( r, nodes, *root ); + // A COUNT ABOVE THE int32 CAP is this path's refusal (docs/SPEC-TABLES.md + // §2.9): the partial builder is the caller's to discard, and the report + // holds what it held when the count was met + ok = ok && !nodes.refused; allocator.free( allocator.context, directory ); return ok; } @@ -5887,8 +5941,8 @@ inline void RowEntriesEntryCookBody( uint8_t * at, const RowEntriesEntry & value template inline bool RowCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Row & value, TableByteOrder order ) { - (void) ctx; (void) region; // no reference below this node: the class was decided by a pointer elsewhere in its closure - table_cook_put( at + 0, 0, 8, order ); // entries: the entry array's delta, filled by the extent writer + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // entries: the array's delta, filled by the extent writer table_cook_put( at + 8, 0, 4, order ); // and its count table_cook_put( at + 16, (uint64_t) value.after, 4, order ); return true; @@ -5902,31 +5956,33 @@ inline void WideRowEntriesEntryCookBody( uint8_t * at, const WideRowEntriesEntry template inline bool WideRowCookBody( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const WideRow & value, TableByteOrder order ) { - (void) ctx; (void) region; // no reference below this node: the class was decided by a pointer elsewhere in its closure - table_cook_put( at + 0, 0, 8, order ); // entries: the entry array's delta, filled by the extent writer + (void) ctx; (void) region; // no reference resolves in this body: a list's and a map's slots are the extent writer's, and the class was decided elsewhere in the closure + table_cook_put( at + 0, 0, 8, order ); // entries: the array's delta, filled by the extent writer table_cook_put( at + 8, 0, 4, order ); // and its count table_cook_put( at + 16, (uint64_t) value.after, 4, order ); return true; } -template inline bool RowEntriesEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const RowEntriesEntry & value, TableByteOrder order ); -template inline bool RowCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Row & value, TableByteOrder order ); -template inline bool WideRowEntriesEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRowEntriesEntry & value, TableByteOrder order ); -template inline bool WideRowCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRow & value, TableByteOrder order ); +template inline bool RowEntriesEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const RowEntriesEntry & value, TableByteOrder order ); +template inline bool RowCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Row & value, TableByteOrder order ); +template inline bool WideRowEntriesEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRowEntriesEntry & value, TableByteOrder order ); +template inline bool WideRowCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRow & value, TableByteOrder order ); -// RowEntriesEntryCookMaps: RowEntriesEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool RowEntriesEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const RowEntriesEntry & value, TableByteOrder order ) +// RowEntriesEntryCookExtent: RowEntriesEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool RowEntriesEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const RowEntriesEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// RowCookMaps: Row's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool RowCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Row & value, TableByteOrder order ) +// RowCookExtent: Row's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool RowCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const Row & value, TableByteOrder order ) { - (void) region; // a map's entries carry their own references through their own bodies + (void) region; // a table element's and an entry's references resolve through their own bodies { // entries TableMapCursor cursor = TableMapOrder( ctx, value.entries ); if ( !cursor.ok ) { return false; } @@ -5945,19 +6001,21 @@ template inline bool RowCookMaps( const Ctx & ctx, const TableCoo return true; } -// WideRowEntriesEntryCookMaps: WideRowEntriesEntry's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool WideRowEntriesEntryCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRowEntriesEntry & value, TableByteOrder order ) +// WideRowEntriesEntryCookExtent: WideRowEntriesEntry's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool WideRowEntriesEntryCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRowEntriesEntry & value, TableByteOrder order ) { (void) ctx; (void) region; (void) extent; (void) at; (void) record; (void) value; (void) order; - return true; // no map below this record + return true; // no list or map below this record } -// WideRowCookMaps: WideRow's map arrays into the node's extent, PRE-ORDER, the entries -// in ASCENDING key order, each through its own cook body (§2.8, §7.6). -template inline bool WideRowCookMaps( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRow & value, TableByteOrder order ) +// WideRowCookExtent: WideRow's arrays into the node's extent, PRE-ORDER, a map's entries +// in ASCENDING key order and a list's elements in INDEX order, each through its +// own cook writer (§2.8, §2.9, §7.6). +template inline bool WideRowCookExtent( const Ctx & ctx, const TableCookRegion & region, uint8_t * extent, int64_t & at, uint8_t * record, const WideRow & value, TableByteOrder order ) { - (void) region; // a map's entries carry their own references through their own bodies + (void) region; // a table element's and an entry's references resolve through their own bodies { // entries TableMapCursor cursor = TableMapOrder( ctx, value.entries ); if ( !cursor.ok ) { return false; } @@ -5976,36 +6034,36 @@ template inline bool WideRowCookMaps( const Ctx & ctx, const Tabl return true; } -// RowEntriesEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// RowEntriesEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool RowEntriesEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const RowEntriesEntry & value, TableByteOrder order ) { RowEntriesEntryCookBody( at, value, order ); int64_t extent_at = 0; - return RowEntriesEntryCookMaps( ctx, region, at + 24, extent_at, at, value, order ); + return RowEntriesEntryCookExtent( ctx, region, at + 24, extent_at, at, value, order ); } -// RowCookNode: one node — the record, then the extent its maps take (§2.8). +// RowCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool RowCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const Row & value, TableByteOrder order ) { if ( !RowCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return RowCookMaps( ctx, region, at + 24, extent_at, at, value, order ); + return RowCookExtent( ctx, region, at + 24, extent_at, at, value, order ); } -// WideRowEntriesEntryCookNode: one node — the record, then the extent its maps take (§2.8). +// WideRowEntriesEntryCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool WideRowEntriesEntryCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const WideRowEntriesEntry & value, TableByteOrder order ) { WideRowEntriesEntryCookBody( at, value, order ); int64_t extent_at = 0; - return WideRowEntriesEntryCookMaps( ctx, region, at + 8, extent_at, at, value, order ); + return WideRowEntriesEntryCookExtent( ctx, region, at + 8, extent_at, at, value, order ); } -// WideRowCookNode: one node — the record, then the extent its maps take (§2.8). +// WideRowCookNode: one node, the record, then the extent its lists and maps take (§2.8, §2.9). template inline bool WideRowCookNode( const Ctx & ctx, const TableCookRegion & region, uint8_t * at, const WideRow & value, TableByteOrder order ) { if ( !WideRowCookBody( ctx, region, at, value, order ) ) { return false; } int64_t extent_at = 0; - return WideRowCookMaps( ctx, region, at + 24, extent_at, at, value, order ); + return WideRowCookExtent( ctx, region, at + 24, extent_at, at, value, order ); } // RowCookLayout: the tool's own Layout (docs/SPEC-TABLES.md §7.2) over one @@ -6015,15 +6073,15 @@ template inline bool WideRowCookNode( const Ctx & ctx, const Tabl // eight. The offsets go into the region's table when it has one, and are only // summed when it does not (a measure). A type id the numbering carries that // this root cannot name is the two walks disagreeing, and it is refused. -// A NODE'S SIZE DEPENDS ON ITS VALUE where a map rides in its extent +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent // (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context -// the numbering walked and reads the same maps that walk read. +// the numbering walked and reads the same arrays that walk read. template inline bool RowCookLayout( const Ctx & ctx, const Row & root, const TableNumbering & numbering, TableCookRegion & region ) { region.numbering = &numbering; region.count = numbering.count + 1; - const int64_t root_extent = RowMapExtent( ctx, root ); + const int64_t root_extent = RowExtent( ctx, root ); if ( root_extent < 0 ) { return false; } int64_t offset = 24 + root_extent; // the root at zero, its extent behind it int64_t align = 8; @@ -6178,15 +6236,15 @@ inline bool RowCook( const RowBuilder & builder, void * out, uint64_t capacity, // eight. The offsets go into the region's table when it has one, and are only // summed when it does not (a measure). A type id the numbering carries that // this root cannot name is the two walks disagreeing, and it is refused. -// A NODE'S SIZE DEPENDS ON ITS VALUE where a map rides in its extent +// A NODE'S SIZE DEPENDS ON ITS VALUE where a list or a map rides in its extent // (docs/SPEC-TABLES.md §2.8), so the layout takes the resolution context -// the numbering walked and reads the same maps that walk read. +// the numbering walked and reads the same arrays that walk read. template inline bool WideRowCookLayout( const Ctx & ctx, const WideRow & root, const TableNumbering & numbering, TableCookRegion & region ) { region.numbering = &numbering; region.count = numbering.count + 1; - const int64_t root_extent = WideRowMapExtent( ctx, root ); + const int64_t root_extent = WideRowExtent( ctx, root ); if ( root_extent < 0 ) { return false; } int64_t offset = 24 + root_extent; // the root at zero, its extent behind it int64_t align = 8; @@ -6399,29 +6457,29 @@ extern const TableTypeInfo WideRowEntriesEntryTableInfo; extern const TableTypeInfo WideRowTableInfo; inline const TableFieldInfo RowEntriesEntryTableFields[] = { - { "key", "key", "string", 0x3dc94a19365b10ecull, 12, false, false, NULL, NULL, true, false, 8, (uint32_t) offsetof( RowEntriesEntry, key ), (uint32_t) sizeof( RowEntriesEntry::key ), (uint32_t) offsetof( RowEntriesEntry, key_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( RowEntriesEntry, value ), (uint32_t) sizeof( RowEntriesEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "string", 0x3dc94a19365b10ecull, 12, false, false, NULL, NULL, true, false, 8, (uint32_t) offsetof( RowEntriesEntry, key ), (uint32_t) sizeof( RowEntriesEntry::key ), (uint32_t) offsetof( RowEntriesEntry, key_length ), 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( RowEntriesEntry, value ), (uint32_t) sizeof( RowEntriesEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo RowEntriesEntryTableInfo = { "RowEntriesEntry", (uint32_t) sizeof( RowEntriesEntry ), 2, RowEntriesEntryTableFields, +[]( void * p ) { RowEntriesEntryReset( *(RowEntriesEntry *) p ); }, false }; inline const TableTypeInfo * RowEntriesEntryTableType() { return &RowEntriesEntryTableInfo; } inline const TableFieldInfo RowTableFields[] = { - { "entries", "entries", "map[string(8)]Item", 0xc5b2a72c0845a253ull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Row, entries ), (uint32_t) sizeof( Row::entries ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &RowEntriesEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * { if ( key == NULL || key_length > kRowEntriesEntryKeyBound ) { return NULL; } RowEntriesEntry * placed = TableMapPlace( worker, *(TableMap *) slot, key ); if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }, "" }, - { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Row, after ), (uint32_t) sizeof( Row::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "entries", "entries", "map[string(8)]Item", 0xc5b2a72c0845a253ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( Row, entries ), (uint32_t) sizeof( RowEntriesEntry ), (uint32_t) offsetof( Row, entries.count ), 0xffffffffu, &RowEntriesEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char * key, int32_t key_length, int64_t ) -> void * { if ( key == NULL || key_length > kRowEntriesEntryKeyBound ) { return NULL; } RowEntriesEntry * placed = TableMapPlace( worker, *(TableMap *) slot, key ); if ( placed != NULL ) { TableEntrySetKey( *placed, key, key_length ); } return (void *) placed; }, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( Row, after ), (uint32_t) sizeof( Row::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo RowTableInfo = { "Row", (uint32_t) sizeof( Row ), 2, RowTableFields, +[]( void * p ) { RowReset( *(Row *) p ); }, true }; inline const TableTypeInfo * RowTableType() { return &RowTableInfo; } inline const TableFieldInfo WideRowEntriesEntryTableFields[] = { - { "key", "key", "uint32", 0x3dc94a19365b10ecull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRowEntriesEntry, key ), (uint32_t) sizeof( WideRowEntriesEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, - { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRowEntriesEntry, value ), (uint32_t) sizeof( WideRowEntriesEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "key", "key", "uint32", 0x3dc94a19365b10ecull, 8, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRowEntriesEntry, key ), (uint32_t) sizeof( WideRowEntriesEntry::key ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "value", "value", "Item", 0x7ce4fd9430e80ceaull, 13, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRowEntriesEntry, value ), (uint32_t) sizeof( WideRowEntriesEntry::value ), 0xffffffffu, 0xffffffffu, &ItemTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo WideRowEntriesEntryTableInfo = { "WideRowEntriesEntry", (uint32_t) sizeof( WideRowEntriesEntry ), 2, WideRowEntriesEntryTableFields, +[]( void * p ) { WideRowEntriesEntryReset( *(WideRowEntriesEntry *) p ); }, false }; inline const TableTypeInfo * WideRowEntriesEntryTableType() { return &WideRowEntriesEntryTableInfo; } inline const TableFieldInfo WideRowTableFields[] = { - { "entries", "entries", "map[uint32]Item", 0xc5b2a72c0845a253ull, 0, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRow, entries ), (uint32_t) sizeof( WideRow::entries ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, &WideRowEntriesEntryTableInfo, []( const void * slot ) -> int32_t { return ( (const TableMap *) slot )->count; }, []( const void * slot, int32_t index ) -> const void * { return (const void *) ( ( (const TableMap *) slot )->Entries() + index ); }, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { WideRowEntriesEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint32_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint32_t) key_value ); } return (void *) placed; }, "" }, - { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRow, after ), (uint32_t) sizeof( WideRow::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, + { "entries", "entries", "map[uint32]Item", 0xc5b2a72c0845a253ull, 13, true, false, NULL, NULL, true, false, 0, (uint32_t) offsetof( WideRow, entries ), (uint32_t) sizeof( WideRowEntriesEntry ), (uint32_t) offsetof( WideRow, entries.count ), 0xffffffffu, &WideRowEntriesEntryTableInfo, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, []( TableWorker & worker, void * slot, const char *, int32_t, int64_t key_value ) -> void * { WideRowEntriesEntry * placed = TableMapPlace( worker, *(TableMap *) slot, (uint32_t) key_value ); if ( placed != NULL ) { TableEntrySetKey( *placed, (uint32_t) key_value ); } return (void *) placed; }, "" }, + { "after", "after", "int32", 0xbf82010f6f71eae9ull, 4, false, false, NULL, NULL, false, false, 0, (uint32_t) offsetof( WideRow, after ), (uint32_t) sizeof( WideRow::after ), 0xffffffffu, 0xffffffffu, NULL, false, 0.0, 0.0, 0, NULL, -1, NULL, NULL, NULL, NULL, NULL, NULL, NULL, "" }, }; inline const TableTypeInfo WideRowTableInfo = { "WideRow", (uint32_t) sizeof( WideRow ), 2, WideRowTableFields, +[]( void * p ) { WideRowReset( *(WideRow *) p ); }, true }; inline const TableTypeInfo * WideRowTableType() { return &WideRowTableInfo; } diff --git a/testdata/golden/tables/messages/MessagesTable.cpp b/testdata/golden/tables/messages/MessagesTable.cpp index 9121df2ba..99d41c3b2 100644 --- a/testdata/golden/tables/messages/MessagesTable.cpp +++ b/testdata/golden/tables/messages/MessagesTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace messagedemo #endif // MESSAGEDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/pointers/GraphTable.cpp b/testdata/golden/tables/pointers/GraphTable.cpp index 72392fb14..2fe115c19 100644 --- a/testdata/golden/tables/pointers/GraphTable.cpp +++ b/testdata/golden/tables/pointers/GraphTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2750,6 +2767,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace graphdemo #endif // GRAPHDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/pointers/MarksTable.cpp b/testdata/golden/tables/pointers/MarksTable.cpp index 6a01b6738..4a36577ac 100644 --- a/testdata/golden/tables/pointers/MarksTable.cpp +++ b/testdata/golden/tables/pointers/MarksTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2750,6 +2767,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace graphdemo #endif // GRAPHDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/pointers/PartsTable.cpp b/testdata/golden/tables/pointers/PartsTable.cpp index cac4f7cff..28baf609f 100644 --- a/testdata/golden/tables/pointers/PartsTable.cpp +++ b/testdata/golden/tables/pointers/PartsTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2750,6 +2767,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace graphdemo #endif // GRAPHDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/scalars/ScalarsTable.cpp b/testdata/golden/tables/scalars/ScalarsTable.cpp index 10fdae356..2bce9867d 100644 --- a/testdata/golden/tables/scalars/ScalarsTable.cpp +++ b/testdata/golden/tables/scalars/ScalarsTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2269,6 +2286,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace scalardemo #endif // SCALARDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/golden/tables/stream/StreamTable.cpp b/testdata/golden/tables/stream/StreamTable.cpp index 13aa93919..9cdb19c3c 100644 --- a/testdata/golden/tables/stream/StreamTable.cpp +++ b/testdata/golden/tables/stream/StreamTable.cpp @@ -46,20 +46,28 @@ inline bool TableJsonWritePointer( TableJsonOut & out, const void * slot, const // skips the value whole, as it skips everything else it does not place. inline bool TableJsonSkippedAmpersand( TableJsonIn & in, const char * key, int32_t depth ); -// ---- the map adapters (docs/SPEC-TABLES.md §2.8, §16) ---- +// ---- the map and list adapters (docs/SPEC-TABLES.md §2.8, §2.9, §16) ---- // -// A MAP is the other construct the walk cannot walk alone: its entries live -// behind a TableMap this walk has no name for, reading one needs the -// builder's arena, and neither exists in a unit that declares no map. Same -// shape as the pointer's three — declared here, defined after the walk by -// whichever half the unit carries. - -// a map field: its descriptor carries the generated ENTRY's +// A MAP and an UNBOUNDED ARRAY are the other constructs the walk cannot walk +// alone: their arrays live behind a TableMap or a TableList this +// walk has no name for, reading one needs the builder's arena, and neither +// exists in a unit that declares neither construct. Same shape as the +// pointer's three: declared here, defined after the walk by whichever half +// the unit carries. Both are OUT-OF-LINE ARRAYS to the descriptors (§8.1): +// array_bound = 0 is the tell, and the type name says which of the two. + +// a map field: an out-of-line array whose type name spells the map inline bool TableJsonIsMap( const TableFieldInfo * f ); // the map as a plain JSON object keyed by the KEY, in ASCENDING key order inline bool TableJsonWriteMap( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); // that object back into the slot, in whatever order the text gives it inline bool TableJsonReadMap( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); +// an unbounded array: the other out-of-line array +inline bool TableJsonIsList( const TableFieldInfo * f ); +// the list as a JSON array, in INDEX order +inline bool TableJsonWriteList( TableJsonOut & out, const void * slot, const TableFieldInfo * f, int32_t depth ); +// that array back into the slot, every element the text carries +inline bool TableJsonReadList( TableJsonIn & in, void * slot, const TableFieldInfo * f, int32_t depth ); // ---- json walk: begin ---- // @@ -819,6 +827,10 @@ inline bool TableJsonWriteField( TableJsonOut & out, const void * base, const Ta { return TableJsonWriteMap( out, (const void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonWriteList( out, (const void *) storage, f, depth ); + } if ( f->kind == 17 && !f->is_array ) { return TableJsonWritePointer( out, storage, f, depth ); @@ -1826,6 +1838,11 @@ inline bool TableJsonReadField( TableJsonIn & in, void * base, const TableFieldI { return TableJsonReadMap( in, (void *) storage, f, depth ); } + if ( TableJsonIsList( f ) ) + { + return TableJsonReadList( in, (void *) storage, f, depth ); + } + if ( f->kind == 12 ) { int32_t length = 0; @@ -2750,6 +2767,25 @@ inline bool TableJsonReadMap( TableJsonIn & in, void *, const TableFieldInfo *, return false; } +// ---- this unit declares no unbounded array ---- +// +// No descriptor of this unit is an out-of-line array that is not a map, so the +// two slot adapters are unreachable and say so. + +inline bool TableJsonIsList( const TableFieldInfo * ) { return false; } + +inline bool TableJsonWriteList( TableJsonOut &, const void *, const TableFieldInfo *, int32_t ) +{ + return false; +} + +inline bool TableJsonReadList( TableJsonIn & in, void *, const TableFieldInfo *, int32_t ) +{ + in.report->malformed = true; + in.bad = true; + return false; +} + } // namespace streamdemo #endif // STREAMDEMO_SCHEMA_TABLE_JSON diff --git a/testdata/wire/tables/list_before_pointer.bin b/testdata/wire/tables/list_before_pointer.bin new file mode 100644 index 000000000..30c86c0ab Binary files /dev/null and b/testdata/wire/tables/list_before_pointer.bin differ diff --git a/testdata/wire/tables/list_empty.bin b/testdata/wire/tables/list_empty.bin new file mode 100644 index 000000000..f795145ae Binary files /dev/null and b/testdata/wire/tables/list_empty.bin differ diff --git a/testdata/wire/tables/list_erased.bin b/testdata/wire/tables/list_erased.bin new file mode 100644 index 000000000..1ab58dfd2 Binary files /dev/null and b/testdata/wire/tables/list_erased.bin differ diff --git a/testdata/wire/tables/list_migrates.bin b/testdata/wire/tables/list_migrates.bin new file mode 100644 index 000000000..7703c1ee0 Binary files /dev/null and b/testdata/wire/tables/list_migrates.bin differ diff --git a/testdata/wire/tables/list_mixed.bin b/testdata/wire/tables/list_mixed.bin new file mode 100644 index 000000000..05526ac48 Binary files /dev/null and b/testdata/wire/tables/list_mixed.bin differ diff --git a/testdata/wire/tables/list_nested.bin b/testdata/wire/tables/list_nested.bin new file mode 100644 index 000000000..c43c45b99 Binary files /dev/null and b/testdata/wire/tables/list_nested.bin differ diff --git a/testdata/wire/tables/list_nested_cook.bin b/testdata/wire/tables/list_nested_cook.bin new file mode 100644 index 000000000..4706884ae Binary files /dev/null and b/testdata/wire/tables/list_nested_cook.bin differ diff --git a/testdata/wire/tables/list_of_maps.bin b/testdata/wire/tables/list_of_maps.bin new file mode 100644 index 000000000..db030336f Binary files /dev/null and b/testdata/wire/tables/list_of_maps.bin differ diff --git a/testdata/wire/tables/list_of_maps_cook.bin b/testdata/wire/tables/list_of_maps_cook.bin new file mode 100644 index 000000000..0b2c7c03e Binary files /dev/null and b/testdata/wire/tables/list_of_maps_cook.bin differ diff --git a/testdata/wire/tables/list_scalars.bin b/testdata/wire/tables/list_scalars.bin new file mode 100644 index 000000000..57134b815 Binary files /dev/null and b/testdata/wire/tables/list_scalars.bin differ diff --git a/testdata/wire/tables/list_shared.bin b/testdata/wire/tables/list_shared.bin new file mode 100644 index 000000000..90a49a013 Binary files /dev/null and b/testdata/wire/tables/list_shared.bin differ diff --git a/testdata/wire/tables/list_tables.bin b/testdata/wire/tables/list_tables.bin new file mode 100644 index 000000000..7160fa37c Binary files /dev/null and b/testdata/wire/tables/list_tables.bin differ