From 9a2f9c7907bbcb63eccda84a391004e2b124b2e7 Mon Sep 17 00:00:00 2001 From: Neelesh Salian Date: Mon, 14 Sep 2026 13:58:18 -0700 Subject: [PATCH 1/4] Add type conformance fixtures and validation --- .github/workflows/license_check.yml | 32 +++++++ .github/workflows/validate-fixtures.yml | 48 +++++++++++ .gitignore | 7 ++ dev/.rat-excludes | 26 ++++++ dev/check-license | 78 +++++++++++++++++ dev/requirements.txt | 19 ++++ dev/schema/cases.base.schema.json | 46 ++++++++++ dev/schema/cases.types.schema.json | 21 +++++ dev/validate-fixtures.py | 102 ++++++++++++++++++++++ table-spec/types/README.md | 110 ++++++++++++++++++++++++ table-spec/types/geospatial/cases.json | 83 ++++++++++++++++++ table-spec/types/nested/cases.json | 96 +++++++++++++++++++++ table-spec/types/primitive/cases.json | 33 +++++++ table-spec/types/variant/cases.json | 5 ++ 14 files changed, 706 insertions(+) create mode 100644 .github/workflows/license_check.yml create mode 100644 .github/workflows/validate-fixtures.yml create mode 100644 dev/.rat-excludes create mode 100755 dev/check-license create mode 100644 dev/requirements.txt create mode 100644 dev/schema/cases.base.schema.json create mode 100644 dev/schema/cases.types.schema.json create mode 100755 dev/validate-fixtures.py create mode 100644 table-spec/types/README.md create mode 100644 table-spec/types/geospatial/cases.json create mode 100644 table-spec/types/nested/cases.json create mode 100644 table-spec/types/primitive/cases.json create mode 100644 table-spec/types/variant/cases.json diff --git a/.github/workflows/license_check.yml b/.github/workflows/license_check.yml new file mode 100644 index 0000000..e873f1c --- /dev/null +++ b/.github/workflows/license_check.yml @@ -0,0 +1,32 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +name: "Run License Check" +on: pull_request + +permissions: + contents: read + +jobs: + rat: + runs-on: ubuntu-22.04 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - run: | + dev/check-license diff --git a/.github/workflows/validate-fixtures.yml b/.github/workflows/validate-fixtures.yml new file mode 100644 index 0000000..5bf01ba --- /dev/null +++ b/.github/workflows/validate-fixtures.yml @@ -0,0 +1,48 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +name: "Validate Fixtures" + +on: + push: + branches: + - main + pull_request: + +concurrency: + group: ${{ github.repository }}-${{ github.head_ref || github.sha }}-${{ github.workflow }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} + +permissions: + contents: read + +jobs: + validate: + name: Validate fixture cases + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - name: Install Python + uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0 + with: + python-version: '3.12' + - name: Install validator deps + run: python3 -m pip install -r dev/requirements.txt + - name: Validate cases.json + run: python3 dev/validate-fixtures.py diff --git a/.gitignore b/.gitignore index 9d61ad5..b8ac23d 100644 --- a/.gitignore +++ b/.gitignore @@ -4,3 +4,10 @@ *~ .idea/ .vscode/ + +# Apache RAT jar downloaded by dev/check-license +/lib/ + +# Python +__pycache__/ +*.py[cod] diff --git a/dev/.rat-excludes b/dev/.rat-excludes new file mode 100644 index 0000000..48de711 --- /dev/null +++ b/dev/.rat-excludes @@ -0,0 +1,26 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +.gitignore +LICENSE +NOTICE +lib + +# Fixture data and JSON Schemas have no comment syntax, so they cannot carry an +# Apache header; the surface README states the corpus is AL2-licensed. +**/*.json +**/*.jsonl diff --git a/dev/check-license b/dev/check-license new file mode 100755 index 0000000..5fc2741 --- /dev/null +++ b/dev/check-license @@ -0,0 +1,78 @@ +#!/usr/bin/env bash + +# Licensed to the Apache Software Foundation (ASF) under one or more +# contributor license agreements. See the NOTICE file distributed with +# this work for additional information regarding copyright ownership. +# The ASF licenses this file to You under the Apache License, Version 2.0 +# (the "License"); you may not use this file except in compliance with +# the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +acquire_rat_jar () { + + URL="https://repo.maven.apache.org/maven2/org/apache/rat/apache-rat/${RAT_VERSION}/apache-rat-${RAT_VERSION}.jar" + + JAR="$rat_jar" + + # Download rat launch jar if it hasn't been downloaded yet + if [ ! -f "$JAR" ]; then + # Download + printf "Attempting to fetch rat\n" + JAR_DL="${JAR}.part" + if [ $(command -v curl) ]; then + curl -L --silent "${URL}" > "$JAR_DL" && mv "$JAR_DL" "$JAR" + elif [ $(command -v wget) ]; then + wget --quiet ${URL} -O "$JAR_DL" && mv "$JAR_DL" "$JAR" + else + printf "You do not have curl or wget installed, please install rat manually.\n" + exit -1 + fi + fi + + unzip -tq "$JAR" &> /dev/null + if [ $? -ne 0 ]; then + # We failed to download + rm "$JAR" + printf "Our attempt to download rat locally to ${JAR} failed. Please install rat manually.\n" + exit -1 + fi +} + +# Go to the project root directory +FWDIR="$(cd "`dirname "$0"`"/..; pwd)" +cd "$FWDIR" + +if test -x "$JAVA_HOME/bin/java"; then + declare java_cmd="$JAVA_HOME/bin/java" +else + declare java_cmd=java +fi + +export RAT_VERSION=0.17 +export rat_jar="$FWDIR"/lib/apache-rat-${RAT_VERSION}.jar +mkdir -p "$FWDIR"/lib + +[[ -f "$rat_jar" ]] || acquire_rat_jar || { + echo "Download failed. Obtain the rat jar manually and place it at $rat_jar" + exit 1 +} + +# --input-exclude-std GIT skips .git and anything .gitignore'd. Without it RAT +# flags the .git pointer *file* that git worktrees use, since its built-in SCM +# exclusion only covers .git directories. +$java_cmd -jar "$rat_jar" \ + --input-exclude-file "$FWDIR"/dev/.rat-excludes \ + --input-exclude-std GIT IDEA MAC \ + --input-include-std HIDDEN_DIR \ + --output-style missing-headers \ + --log-level ERROR \ + -- "$FWDIR" || exit 1 + +echo "RAT checks passed." diff --git a/dev/requirements.txt b/dev/requirements.txt new file mode 100644 index 0000000..382d9eb --- /dev/null +++ b/dev/requirements.txt @@ -0,0 +1,19 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Dependencies for the dev/ tooling (fixture validation). +jsonschema>=4.18 diff --git a/dev/schema/cases.base.schema.json b/dev/schema/cases.base.schema.json new file mode 100644 index 0000000..5d2b90b --- /dev/null +++ b/dev/schema/cases.base.schema.json @@ -0,0 +1,46 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://iceberg.apache.org/verification/cases.base.schema.json", + "title": "Conformance cases (base)", + "description": "Structure shared by every table-spec surface's cases.json.", + "type": "object", + "required": ["cases"], + "properties": { + "cases": { + "type": "array", + "items": { "$ref": "#/$defs/case" } + } + }, + "$defs": { + "case": { + "type": "object", + "required": ["id", "valid", "input"], + "properties": { + "id": { "type": "string", "minLength": 1 }, + "valid": { "type": "boolean" }, + "input": true, + "decoded": true, + "canonical": { "type": "string" }, + "clause": { "type": "string" }, + "spec_ref": { "type": "string" } + }, + "allOf": [ + { + "$comment": "a valid case must carry decoded", + "if": { "properties": { "valid": { "const": true } }, "required": ["valid"] }, + "then": { "required": ["decoded"] } + }, + { + "$comment": "an invalid case must not carry decoded", + "if": { "properties": { "valid": { "const": false } }, "required": ["valid"] }, + "then": { "not": { "required": ["decoded"] } } + }, + { + "$comment": "canonical is only allowed on a valid case", + "if": { "required": ["canonical"] }, + "then": { "properties": { "valid": { "const": true } }, "required": ["valid"] } + } + ] + } + } +} diff --git a/dev/schema/cases.types.schema.json b/dev/schema/cases.types.schema.json new file mode 100644 index 0000000..bd010dc --- /dev/null +++ b/dev/schema/cases.types.schema.json @@ -0,0 +1,21 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://iceberg.apache.org/verification/cases.types.schema.json", + "title": "Conformance cases (types surface extension)", + "description": "Extra constraints for table-spec/types: every case cites the spec. Applied on top of cases.base.schema.json by dev/validate-fixtures.py.", + "type": "object", + "required": ["cases"], + "properties": { + "cases": { + "type": "array", + "items": { + "type": "object", + "required": ["clause", "spec_ref"], + "properties": { + "clause": { "type": "string", "minLength": 1 }, + "spec_ref": { "type": "string", "minLength": 1 } + } + } + } + } +} diff --git a/dev/validate-fixtures.py b/dev/validate-fixtures.py new file mode 100755 index 0000000..0d3339e --- /dev/null +++ b/dev/validate-fixtures.py @@ -0,0 +1,102 @@ +#!/usr/bin/env python3 +# +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. +"""Validate the conformance type fixtures against the JSON Schemas in dev/schema/. + +Each cases.json is validated against the base structural schema and the types schema +(which requires clause and spec_ref); case ids must be globally unique. +""" + +import glob +import json +import os +import sys + +import jsonschema + +HERE = os.path.dirname(os.path.abspath(__file__)) +SCHEMA_DIR = os.path.join(HERE, "schema") + + +def _load_schema(name): + with open(os.path.join(SCHEMA_DIR, name), encoding="utf-8") as fh: + return json.load(fh) + + +def _rel(path): + return path.replace(os.sep, "/") + + +def _is_types_surface(path): + p = _rel(path) + return "/table-spec/types/" in p or p.startswith("table-spec/types/") + + +def main(): + root = sys.argv[1] if len(sys.argv) > 1 else "." + base_validator = jsonschema.Draft202012Validator(_load_schema("cases.base.schema.json")) + types_validator = jsonschema.Draft202012Validator(_load_schema("cases.types.schema.json")) + + files = sorted(glob.glob(f"{root}/table-spec/**/cases.json", recursive=True)) + if not files: + print("no cases.json files found", file=sys.stderr) + return 1 + + errors = [] + seen_ids = {} + total = 0 + for path in files: + try: + with open(path, encoding="utf-8") as fh: + doc = json.load(fh) + except json.JSONDecodeError as e: + errors.append(f"{path}: invalid JSON: {e}") + continue + + validators = [base_validator] + if _is_types_surface(path): + validators.append(types_validator) + for validator in validators: + for err in sorted(validator.iter_errors(doc), key=str): + where = "/".join(str(p) for p in err.absolute_path) or "(root)" + errors.append(f"{path}: {where}: {err.message}") + + cases = doc.get("cases") if isinstance(doc, dict) else None + if isinstance(cases, list): + for i, case in enumerate(cases): + if isinstance(case, dict) and isinstance(case.get("id"), str) and case["id"]: + cid = case["id"] + if cid in seen_ids: + errors.append(f"{path}[{i}]: duplicate id '{cid}' (first seen at {seen_ids[cid]})") + else: + seen_ids[cid] = f"{path}[{i}]" + total += len(cases) + print(f" {path}: {len(cases)} cases") + + if errors: + print(f"\nFAILED: {len(errors)} problem(s) in {len(files)} file(s):", file=sys.stderr) + for e in errors: + print(f" {e}", file=sys.stderr) + return 1 + + print(f"\nOK: {total} cases across {len(files)} file(s), {len(seen_ids)} unique ids.") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/table-spec/types/README.md b/table-spec/types/README.md new file mode 100644 index 0000000..c9ef994 --- /dev/null +++ b/table-spec/types/README.md @@ -0,0 +1,110 @@ + + +# Type decoding + +Parsing a type string produces the same type in every implementation. This +surface pins each type the spec defines and the parse rules that attach to it. + +## Assertion + +``` +parse(input) == decoded +``` + +`input` is a type string, or a JSON object for a nested type; `decoded` is the +language-neutral shape below. Bytes are not compared - each implementation maps +its own type object to `decoded`, so the comparison does not depend on one +language's representation. + +- `valid: true` - the parser succeeds and the decoded type equals `decoded`. A + type an implementation does not model is UNSUPPORTED, not a failure. If the case + also carries `canonical`, re-serializing the parsed type must equal it byte for + byte (the write direction). +- `valid: false` - the parser must reject `input`. A rejection passes; a + successful parse fails. + +`canonical` is present only where the spec pins one spelling. `decimal` has two +blessed forms (`decimal(9,2)` and `decimal(9, 2)`), so its cases have no +`canonical` and are compared by `decoded` alone. + +## Scope + +Each type in isolation, per the Primitive Types table and Appendix C. The full +schema document (schema-id, identifier-field-ids, field ordering) is the `schema` +surface. Whether a type is legal at a given format version is not decided here, +because a type in isolation carries no version. + +## Inputs + +- `primitive/` - every v1/v2 primitive (`boolean`, `int`, `long`, `float`, + `double`, `date`, `time`, `timestamp`, `timestamptz`, `string`, `uuid`, + `binary`, `decimal`, `fixed`) plus the v3 additions `timestamp_ns`, + `timestamptz_ns`, `unknown`. +- `variant/` - `variant` (v3). +- `nested/` - `struct`, `list`, `map`, including nesting (a `struct` field whose + type is a `list`). +- `geospatial/` - `geometry` and `geography` (v3), with explicit and default CRS. + +## Case format + +One `cases.json` per directory, a JSON object with a `cases` array: + +| field | meaning | +| --- | --- | +| `id` | unique case id | +| `valid` | `true` if the parser must accept `input`, `false` if it must reject it | +| `input` | the type string (`"decimal(9,2)"`), or a JSON object for a nested type | +| `decoded` | the decoded shape; present only when `valid` is `true` | +| `canonical` | the exact re-serialized string; present only where the spec pins one spelling | +| `clause` | the spec rule this case pins | +| `spec_ref` | anchor into `format/spec.md` | + +`decoded` is language-neutral: + +- a simple type with no parameters: `{"type": ""}`, e.g. `{"type": "int"}` +- `decimal`: `{"type": "decimal", "precision": P, "scale": S}` +- `fixed`: `{"type": "fixed", "length": L}` +- `geometry`: `{"type": "geometry", "crs": C}`; `geography` adds `"algorithm": A` +- `struct`: `{"type": "struct", "fields": [{"id", "name", "required", "type"}, ...]}` +- `list`: `{"type": "list", "element-id", "element-required", "element"}` +- `map`: `{"type": "map", "key-id", "key", "value-id", "value-required", "value"}` + +A nested type's child `type` values are the same shape, recursively. + +## Provenance + +`input` and `decoded` are derived from `format/spec.md` (the Primitive Types +table and Appendix C), cross-checked against Apache Iceberg Java. There is no +binary artifact; Java is a cross-check, not the source of the inputs. + +## Left out on purpose + +Inputs the spec neither permits nor forbids, so no answer can be spec-derived: + +- `scale > precision`, e.g. `decimal(5, 10)`. +- lower bounds, e.g. `decimal(0, 0)` / `fixed[0]`. +- internal whitespace around every parameter, e.g. `decimal( 9 , 2 )`. The one + spaced case we do ship, `decimal(9, 2)`, is a *recommended* (SHOULD) accept, + not a hard requirement: the spec says readers *should*, not *must*, accept + optional whitespace, so an implementation that rejects it is still conformant. +- keyword case, e.g. `DECIMAL(9,2)`. +- geospatial CRS *quoting*, e.g. `geometry('OGC:CRS84')`. The spec's canonical + form is unquoted (`geometry(OGC:CRS84)`), which the shipped cases use; whether + the quoted form is also accepted is unpinned. diff --git a/table-spec/types/geospatial/cases.json b/table-spec/types/geospatial/cases.json new file mode 100644 index 0000000..726a839 --- /dev/null +++ b/table-spec/types/geospatial/cases.json @@ -0,0 +1,83 @@ +{ + "cases": [ + { + "id": "geometry-crs84", + "valid": true, + "input": "geometry(OGC:CRS84)", + "decoded": {"type": "geometry", "crs": "OGC:CRS84"}, + "clause": "geometry(C) with explicit CRS; the canonical serialized form is unquoted \"geometry()\"", + "spec_ref": "format/spec.md#appendix-c-json-serialization" + }, + { + "id": "geometry-srid", + "valid": true, + "input": "geometry(srid:4326)", + "decoded": {"type": "geometry", "crs": "srid:4326"}, + "clause": "geometry(C) example from Appendix C is the unquoted \"geometry(srid:4326)\"", + "spec_ref": "format/spec.md#appendix-c-json-serialization" + }, + { + "id": "geometry-default-crs", + "valid": true, + "input": "geometry", + "decoded": {"type": "geometry", "crs": "OGC:CRS84"}, + "clause": "geometry(C): if C is not specified, C is OGC:CRS84", + "spec_ref": "format/spec.md#primitive-types" + }, + { + "id": "geography-crs84-spherical", + "valid": true, + "input": "geography(OGC:CRS84, spherical)", + "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"}, + "clause": "geography(C, A); the canonical serialized form is unquoted \"geography(, )\"", + "spec_ref": "format/spec.md#appendix-c-json-serialization" + }, + { + "id": "geography-default", + "valid": true, + "input": "geography", + "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"}, + "clause": "geography(C, A): if not specified, C is OGC:CRS84 and A is spherical", + "spec_ref": "format/spec.md#primitive-types" + }, + { + "id": "geography-vincenty", + "valid": true, + "input": "geography(OGC:CRS84, vincenty)", + "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "vincenty"}, + "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney", + "spec_ref": "format/spec.md#primitive-types" + }, + { + "id": "geography-thomas", + "valid": true, + "input": "geography(OGC:CRS84, thomas)", + "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "thomas"}, + "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney", + "spec_ref": "format/spec.md#primitive-types" + }, + { + "id": "geography-andoyer", + "valid": true, + "input": "geography(OGC:CRS84, andoyer)", + "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "andoyer"}, + "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney", + "spec_ref": "format/spec.md#primitive-types" + }, + { + "id": "geography-karney", + "valid": true, + "input": "geography(OGC:CRS84, karney)", + "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "karney"}, + "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney", + "spec_ref": "format/spec.md#primitive-types" + }, + { + "id": "geography-unknown-algorithm", + "valid": false, + "input": "geography(OGC:CRS84, bogus)", + "clause": "geography edge-interpolation algorithm A must be one of the closed set (spherical, vincenty, thomas, andoyer, karney); an out-of-set value is rejected", + "spec_ref": "format/spec.md#primitive-types" + } + ] +} diff --git a/table-spec/types/nested/cases.json b/table-spec/types/nested/cases.json new file mode 100644 index 0000000..cb1d3b5 --- /dev/null +++ b/table-spec/types/nested/cases.json @@ -0,0 +1,96 @@ +{ + "cases": [ + { + "id": "struct-single-field", + "valid": true, + "input": {"type": "struct", "fields": [{"id": 1, "name": "a", "required": true, "type": "int"}]}, + "decoded": {"type": "struct", "fields": [{"id": 1, "name": "a", "required": true, "type": {"type": "int"}}]}, + "clause": "struct is a tuple of typed fields; each field has an integer id, a name, a required flag, and a type", + "spec_ref": "format/spec.md#nested-types" + }, + { + "id": "struct-empty", + "valid": true, + "input": {"type": "struct", "fields": []}, + "decoded": {"type": "struct", "fields": []}, + "clause": "a struct's fields array may be empty (Appendix C struct serialization)", + "spec_ref": "format/spec.md#appendix-c-json-serialization" + }, + { + "id": "struct-optional-field", + "valid": true, + "input": {"type": "struct", "fields": [{"id": 1, "name": "a", "required": false, "type": "string"}]}, + "decoded": {"type": "struct", "fields": [{"id": 1, "name": "a", "required": false, "type": {"type": "string"}}]}, + "clause": "each struct field can be optional or required (required=false permits null values)", + "spec_ref": "format/spec.md#nested-types" + }, + { + "id": "list-required-element", + "valid": true, + "input": {"type": "list", "element-id": 2, "element-required": true, "element": "string"}, + "decoded": {"type": "list", "element-id": 2, "element-required": true, "element": {"type": "string"}}, + "clause": "a list has an element type; the element field has an integer id and a required flag", + "spec_ref": "format/spec.md#appendix-c-json-serialization" + }, + { + "id": "list-optional-element", + "valid": true, + "input": {"type": "list", "element-id": 2, "element-required": false, "element": "long"}, + "decoded": {"type": "list", "element-id": 2, "element-required": false, "element": {"type": "long"}}, + "clause": "list elements can be optional or required", + "spec_ref": "format/spec.md#nested-types" + }, + { + "id": "map-string-double", + "valid": true, + "input": {"type": "map", "key-id": 3, "key": "string", "value-id": 4, "value-required": false, "value": "double"}, + "decoded": {"type": "map", "key-id": 3, "key": {"type": "string"}, "value-id": 4, "value-required": false, "value": {"type": "double"}}, + "clause": "a map has a key type and a value type; keys are required, values may be optional or required", + "spec_ref": "format/spec.md#appendix-c-json-serialization" + }, + { + "id": "struct-nested-list", + "valid": true, + "input": {"type": "struct", "fields": [{"id": 1, "name": "tags", "required": true, "type": {"type": "list", "element-id": 2, "element-required": true, "element": "string"}}]}, + "decoded": {"type": "struct", "fields": [{"id": 1, "name": "tags", "required": true, "type": {"type": "list", "element-id": 2, "element-required": true, "element": {"type": "string"}}}]}, + "clause": "fields may be any type, including nested types (a struct field whose type is a list)", + "spec_ref": "format/spec.md#nested-types" + }, + { + "id": "struct-field-missing-id", + "valid": false, + "input": {"type": "struct", "fields": [{"name": "a", "required": true, "type": "int"}]}, + "clause": "each field in a struct has an integer id; a field without an id is invalid", + "spec_ref": "format/spec.md#nested-types" + }, + { + "id": "map-string-double-required", + "valid": true, + "input": {"type": "map", "key-id": 3, "key": "string", "value-id": 4, "value-required": true, "value": "double"}, + "decoded": {"type": "map", "key-id": 3, "key": {"type": "string"}, "value-id": 4, "value-required": true, "value": {"type": "double"}}, + "clause": "map values may be optional or required (value-required=true)", + "spec_ref": "format/spec.md#nested-types" + }, + { + "id": "list-missing-element-id", + "valid": false, + "input": {"type": "list", "element-required": true, "element": "string"}, + "clause": "a list's element field has an integer element-id; a list without element-id is invalid", + "spec_ref": "format/spec.md#appendix-c-json-serialization" + }, + { + "id": "map-missing-key-id", + "valid": false, + "input": {"type": "map", "key": "string", "value-id": 4, "value-required": false, "value": "double"}, + "clause": "a map's key field has an integer key-id; a map without key-id is invalid", + "spec_ref": "format/spec.md#appendix-c-json-serialization" + }, + { + "id": "map-missing-value-id", + "valid": false, + "input": {"type": "map", "key-id": 3, "key": "string", "value-required": false, "value": "double"}, + "clause": "a map's value field has an integer value-id; a map without value-id is invalid", + "spec_ref": "format/spec.md#appendix-c-json-serialization" + } + ] +} diff --git a/table-spec/types/primitive/cases.json b/table-spec/types/primitive/cases.json new file mode 100644 index 0000000..b41a30b --- /dev/null +++ b/table-spec/types/primitive/cases.json @@ -0,0 +1,33 @@ +{ + "cases": [ + { "id": "boolean", "valid": true, "input": "boolean", "decoded": { "type": "boolean" }, "canonical": "boolean", "clause": "Primitive Types: boolean; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "int", "valid": true, "input": "int", "decoded": { "type": "int" }, "canonical": "int", "clause": "Primitive Types: int; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "long", "valid": true, "input": "long", "decoded": { "type": "long" }, "canonical": "long", "clause": "Primitive Types: long; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "float", "valid": true, "input": "float", "decoded": { "type": "float" }, "canonical": "float", "clause": "Primitive Types: float; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "double", "valid": true, "input": "double", "decoded": { "type": "double" }, "canonical": "double", "clause": "Primitive Types: double; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "date", "valid": true, "input": "date", "decoded": { "type": "date" }, "canonical": "date", "clause": "Primitive Types: date; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "time", "valid": true, "input": "time", "decoded": { "type": "time" }, "canonical": "time", "clause": "Primitive Types: time; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "timestamp", "valid": true, "input": "timestamp", "decoded": { "type": "timestamp" }, "canonical": "timestamp", "clause": "Primitive Types: timestamp; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "timestamptz", "valid": true, "input": "timestamptz", "decoded": { "type": "timestamptz" }, "canonical": "timestamptz", "clause": "Primitive Types: timestamptz; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "timestamp_ns", "valid": true, "input": "timestamp_ns", "decoded": { "type": "timestamp_ns" }, "canonical": "timestamp_ns", "clause": "Primitive Types: timestamp_ns added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#primitive-types" }, + { "id": "timestamptz_ns", "valid": true, "input": "timestamptz_ns", "decoded": { "type": "timestamptz_ns" }, "canonical": "timestamptz_ns", "clause": "Primitive Types: timestamptz_ns added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#primitive-types" }, + { "id": "string", "valid": true, "input": "string", "decoded": { "type": "string" }, "canonical": "string", "clause": "Primitive Types: string; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "uuid", "valid": true, "input": "uuid", "decoded": { "type": "uuid" }, "canonical": "uuid", "clause": "Primitive Types: uuid; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "binary", "valid": true, "input": "binary", "decoded": { "type": "binary" }, "canonical": "binary", "clause": "Primitive Types: binary; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "unknown", "valid": true, "input": "unknown", "decoded": { "type": "unknown" }, "canonical": "unknown", "clause": "Primitive Types: unknown added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#primitive-types" }, + { "id": "fixed-1", "valid": true, "input": "fixed[1]", "decoded": { "type": "fixed", "length": 1 }, "canonical": "fixed[1]", "clause": "Appendix C: fixed canonical string is fixed[]", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "fixed-16", "valid": true, "input": "fixed[16]", "decoded": { "type": "fixed", "length": 16 }, "canonical": "fixed[16]", "clause": "Appendix C: fixed canonical string is fixed[]", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "decimal-9-2", "valid": true, "input": "decimal(9,2)", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "clause": "Appendix C: both decimal(9,2) and decimal(9, 2) are canonical, so no byte-exact form is pinned", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "decimal-9-2-spaced", "valid": true, "input": "decimal(9, 2)", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "clause": "Appendix C: the spaced decimal(9, 2) form parses to the same decimal", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "decimal-38-0", "valid": true, "input": "decimal(38,0)", "decoded": { "type": "decimal", "precision": 38, "scale": 0 }, "clause": "Primitive Types: decimal precision must be 38 or less (38 is the maximum)", "spec_ref": "format/spec.md#primitive-types" }, + { "id": "decimal-precision-over-38", "valid": false, "input": "decimal(39,0)", "clause": "Primitive Types: decimal precision must be 38 or less", "spec_ref": "format/spec.md#primitive-types" }, + { "id": "decimal-missing-scale", "valid": false, "input": "decimal(9)", "clause": "Appendix C: decimal is written decimal(P,S); scale is required", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "decimal-empty-params", "valid": false, "input": "decimal()", "clause": "Appendix C: decimal requires precision and scale", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "decimal-non-numeric", "valid": false, "input": "decimal(a,b)", "clause": "Appendix C: decimal precision and scale are integers", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "fixed-empty-length", "valid": false, "input": "fixed[]", "clause": "Appendix C: fixed is written fixed[]; length is required", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "fixed-unterminated", "valid": false, "input": "fixed[16", "clause": "Appendix C: fixed[] must be closed with a bracket", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "fixed-non-numeric", "valid": false, "input": "fixed[abc]", "clause": "Appendix C: fixed length is an integer", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "empty-type", "valid": false, "input": "", "clause": "Primitive Types: the empty string is not a type name", "spec_ref": "format/spec.md#primitive-types" }, + { "id": "unknown-type-name", "valid": false, "input": "notatype", "clause": "Primitive Types: only the listed type names are valid", "spec_ref": "format/spec.md#primitive-types" } + ] +} diff --git a/table-spec/types/variant/cases.json b/table-spec/types/variant/cases.json new file mode 100644 index 0000000..ace69ac --- /dev/null +++ b/table-spec/types/variant/cases.json @@ -0,0 +1,5 @@ +{ + "cases": [ + { "id": "variant", "valid": true, "input": "variant", "decoded": { "type": "variant" }, "canonical": "variant", "clause": "Semi-structured Types: variant added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" } + ] +} From 63064f321537abb50358c71f5a00f006e8a34e64 Mon Sep 17 00:00:00 2001 From: Neelesh Salian Date: Thu, 17 Sep 2026 10:52:55 -0700 Subject: [PATCH 2/4] rebase main and pr comment fixes --- .github/workflows/license_check.yml | 13 +++++++++++-- .github/workflows/validate-fixtures.yml | 10 ++++++++++ dev/check-license | 10 +++++----- dev/requirements.txt | 2 +- dev/schema/cases.base.schema.json | 5 ++++- dev/validate-fixtures.py | 11 ++++++++--- table-spec/types/README.md | 16 +++++++++++----- table-spec/types/geospatial/cases.json | 17 +++++++++++++---- table-spec/types/primitive/cases.json | 3 ++- table-spec/types/variant/cases.json | 3 ++- 10 files changed, 67 insertions(+), 23 deletions(-) diff --git a/.github/workflows/license_check.yml b/.github/workflows/license_check.yml index e873f1c..ee54cb4 100644 --- a/.github/workflows/license_check.yml +++ b/.github/workflows/license_check.yml @@ -16,14 +16,23 @@ # under the License. name: "Run License Check" -on: pull_request + +on: + push: + branches: + - main + pull_request: + +concurrency: + group: ${{ github.repository }}-${{ github.head_ref || github.sha }}-${{ github.workflow }} + cancel-in-progress: ${{ github.event_name == 'pull_request' }} permissions: contents: read jobs: rat: - runs-on: ubuntu-22.04 + runs-on: ubuntu-24.04 steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: diff --git a/.github/workflows/validate-fixtures.yml b/.github/workflows/validate-fixtures.yml index 5bf01ba..2e37580 100644 --- a/.github/workflows/validate-fixtures.yml +++ b/.github/workflows/validate-fixtures.yml @@ -46,3 +46,13 @@ jobs: run: python3 -m pip install -r dev/requirements.txt - name: Validate cases.json run: python3 dev/validate-fixtures.py + - name: Validator rejects a malformed fixture (self-test) + run: | + tmp="$(mktemp -d)" + mkdir -p "$tmp/table-spec/types/x" + printf '%s' '{"cases":[{"id":"bad","valid":true,"input":"int","clause":"c","spec_ref":"s"}]}' > "$tmp/table-spec/types/x/cases.json" + if python3 dev/validate-fixtures.py "$tmp"; then + echo "self-test FAILED: validator accepted a malformed fixture (valid:true with no decoded)" + exit 1 + fi + echo "self-test ok: validator rejected the malformed fixture" diff --git a/dev/check-license b/dev/check-license index 5fc2741..3c951b9 100755 --- a/dev/check-license +++ b/dev/check-license @@ -32,7 +32,7 @@ acquire_rat_jar () { wget --quiet ${URL} -O "$JAR_DL" && mv "$JAR_DL" "$JAR" else printf "You do not have curl or wget installed, please install rat manually.\n" - exit -1 + exit 1 fi fi @@ -41,7 +41,7 @@ acquire_rat_jar () { # We failed to download rm "$JAR" printf "Our attempt to download rat locally to ${JAR} failed. Please install rat manually.\n" - exit -1 + exit 1 fi } @@ -56,8 +56,8 @@ else fi export RAT_VERSION=0.17 -export rat_jar="$FWDIR"/lib/apache-rat-${RAT_VERSION}.jar -mkdir -p "$FWDIR"/lib +export rat_jar="$FWDIR/lib/apache-rat-${RAT_VERSION}.jar" +mkdir -p "$FWDIR/lib" [[ -f "$rat_jar" ]] || acquire_rat_jar || { echo "Download failed. Obtain the rat jar manually and place it at $rat_jar" @@ -68,7 +68,7 @@ mkdir -p "$FWDIR"/lib # flags the .git pointer *file* that git worktrees use, since its built-in SCM # exclusion only covers .git directories. $java_cmd -jar "$rat_jar" \ - --input-exclude-file "$FWDIR"/dev/.rat-excludes \ + --input-exclude-file "$FWDIR/dev/.rat-excludes" \ --input-exclude-std GIT IDEA MAC \ --input-include-std HIDDEN_DIR \ --output-style missing-headers \ diff --git a/dev/requirements.txt b/dev/requirements.txt index 382d9eb..29c27ff 100644 --- a/dev/requirements.txt +++ b/dev/requirements.txt @@ -16,4 +16,4 @@ # under the License. # Dependencies for the dev/ tooling (fixture validation). -jsonschema>=4.18 +jsonschema>=4.18,<5 diff --git a/dev/schema/cases.base.schema.json b/dev/schema/cases.base.schema.json index 5d2b90b..60fb70b 100644 --- a/dev/schema/cases.base.schema.json +++ b/dev/schema/cases.base.schema.json @@ -5,6 +5,7 @@ "description": "Structure shared by every table-spec surface's cases.json.", "type": "object", "required": ["cases"], + "additionalProperties": false, "properties": { "cases": { "type": "array", @@ -15,6 +16,7 @@ "case": { "type": "object", "required": ["id", "valid", "input"], + "additionalProperties": false, "properties": { "id": { "type": "string", "minLength": 1 }, "valid": { "type": "boolean" }, @@ -22,7 +24,8 @@ "decoded": true, "canonical": { "type": "string" }, "clause": { "type": "string" }, - "spec_ref": { "type": "string" } + "spec_ref": { "type": "string" }, + "normative_level": { "enum": ["must", "should"], "description": "must (default) or should. A should case is advisory (the spec only recommends the behavior); a runner must report a failed should distinctly from a pass, so the advisory tier stays visible rather than reading as inert." } }, "allOf": [ { diff --git a/dev/validate-fixtures.py b/dev/validate-fixtures.py index 0d3339e..b1548c2 100755 --- a/dev/validate-fixtures.py +++ b/dev/validate-fixtures.py @@ -78,13 +78,18 @@ def main(): cases = doc.get("cases") if isinstance(doc, dict) else None if isinstance(cases, list): + # ids are unique per surface (the top-level dir under table-spec), so a + # future surface may reuse names like int/string/timestamp. + parts = _rel(os.path.relpath(path, root)).split("/") + surface = parts[parts.index("table-spec") + 1] if "table-spec" in parts and parts.index("table-spec") + 1 < len(parts) else parts[0] for i, case in enumerate(cases): if isinstance(case, dict) and isinstance(case.get("id"), str) and case["id"]: cid = case["id"] - if cid in seen_ids: - errors.append(f"{path}[{i}]: duplicate id '{cid}' (first seen at {seen_ids[cid]})") + key = (surface, cid) + if key in seen_ids: + errors.append(f"{path}[{i}]: duplicate id '{cid}' in surface '{surface}' (first seen at {seen_ids[key]})") else: - seen_ids[cid] = f"{path}[{i}]" + seen_ids[key] = f"{path}[{i}]" total += len(cases) print(f" {path}: {len(cases)} cases") diff --git a/table-spec/types/README.md b/table-spec/types/README.md index c9ef994..9593d3a 100644 --- a/table-spec/types/README.md +++ b/table-spec/types/README.md @@ -68,13 +68,14 @@ One `cases.json` per directory, a JSON object with a `cases` array: | field | meaning | | --- | --- | -| `id` | unique case id | +| `id` | case id, unique within the surface | | `valid` | `true` if the parser must accept `input`, `false` if it must reject it | | `input` | the type string (`"decimal(9,2)"`), or a JSON object for a nested type | | `decoded` | the decoded shape; present only when `valid` is `true` | | `canonical` | the exact re-serialized string; present only where the spec pins one spelling | | `clause` | the spec rule this case pins | | `spec_ref` | anchor into `format/spec.md` | +| `normative_level` | optional; `must` (default) or `should` for an advisory case a conformant reader may reject | `decoded` is language-neutral: @@ -88,6 +89,15 @@ One `cases.json` per directory, a JSON object with a `cases` array: A nested type's child `type` values are the same shape, recursively. +Case ids are unique within a surface (the top-level directory under `table-spec/`), +so a future surface may reuse a name like `int` or `string`. + +Cases are normative MUST by default. A case marked `normative_level: "should"` is +advisory: the spec only recommends the behavior, so a reader that diverges is still +conformant. Example: `decimal( 9 , 2 )`, since readers *should* accept optional +whitespace around parameters and separators (Appendix C). A runner must report a +failed `should` case distinctly from a pass, so the advisory tier stays visible. + ## Provenance `input` and `decoded` are derived from `format/spec.md` (the Primitive Types @@ -100,10 +110,6 @@ Inputs the spec neither permits nor forbids, so no answer can be spec-derived: - `scale > precision`, e.g. `decimal(5, 10)`. - lower bounds, e.g. `decimal(0, 0)` / `fixed[0]`. -- internal whitespace around every parameter, e.g. `decimal( 9 , 2 )`. The one - spaced case we do ship, `decimal(9, 2)`, is a *recommended* (SHOULD) accept, - not a hard requirement: the spec says readers *should*, not *must*, accept - optional whitespace, so an implementation that rejects it is still conformant. - keyword case, e.g. `DECIMAL(9,2)`. - geospatial CRS *quoting*, e.g. `geometry('OGC:CRS84')`. The spec's canonical form is unquoted (`geometry(OGC:CRS84)`), which the shipped cases use; whether diff --git a/table-spec/types/geospatial/cases.json b/table-spec/types/geospatial/cases.json index 726a839..bae03ea 100644 --- a/table-spec/types/geospatial/cases.json +++ b/table-spec/types/geospatial/cases.json @@ -5,6 +5,7 @@ "valid": true, "input": "geometry(OGC:CRS84)", "decoded": {"type": "geometry", "crs": "OGC:CRS84"}, + "canonical": "geometry(OGC:CRS84)", "clause": "geometry(C) with explicit CRS; the canonical serialized form is unquoted \"geometry()\"", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, @@ -13,6 +14,7 @@ "valid": true, "input": "geometry(srid:4326)", "decoded": {"type": "geometry", "crs": "srid:4326"}, + "canonical": "geometry(srid:4326)", "clause": "geometry(C) example from Appendix C is the unquoted \"geometry(srid:4326)\"", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, @@ -21,14 +23,16 @@ "valid": true, "input": "geometry", "decoded": {"type": "geometry", "crs": "OGC:CRS84"}, - "clause": "geometry(C): if C is not specified, C is OGC:CRS84", - "spec_ref": "format/spec.md#primitive-types" + "canonical": "geometry(OGC:CRS84)", + "clause": "geometry(C): if C is not specified, C is OGC:CRS84; the canonical serialized form is fully parameterized (Appendix C)", + "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "geography-crs84-spherical", "valid": true, "input": "geography(OGC:CRS84, spherical)", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"}, + "canonical": "geography(OGC:CRS84, spherical)", "clause": "geography(C, A); the canonical serialized form is unquoted \"geography(, )\"", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, @@ -37,14 +41,16 @@ "valid": true, "input": "geography", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"}, - "clause": "geography(C, A): if not specified, C is OGC:CRS84 and A is spherical", - "spec_ref": "format/spec.md#primitive-types" + "canonical": "geography(OGC:CRS84, spherical)", + "clause": "geography(C, A): if not specified, C is OGC:CRS84 and A is spherical; the canonical serialized form is fully parameterized (Appendix C)", + "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "geography-vincenty", "valid": true, "input": "geography(OGC:CRS84, vincenty)", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "vincenty"}, + "canonical": "geography(OGC:CRS84, vincenty)", "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney", "spec_ref": "format/spec.md#primitive-types" }, @@ -53,6 +59,7 @@ "valid": true, "input": "geography(OGC:CRS84, thomas)", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "thomas"}, + "canonical": "geography(OGC:CRS84, thomas)", "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney", "spec_ref": "format/spec.md#primitive-types" }, @@ -61,6 +68,7 @@ "valid": true, "input": "geography(OGC:CRS84, andoyer)", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "andoyer"}, + "canonical": "geography(OGC:CRS84, andoyer)", "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney", "spec_ref": "format/spec.md#primitive-types" }, @@ -69,6 +77,7 @@ "valid": true, "input": "geography(OGC:CRS84, karney)", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "karney"}, + "canonical": "geography(OGC:CRS84, karney)", "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney", "spec_ref": "format/spec.md#primitive-types" }, diff --git a/table-spec/types/primitive/cases.json b/table-spec/types/primitive/cases.json index b41a30b..653b454 100644 --- a/table-spec/types/primitive/cases.json +++ b/table-spec/types/primitive/cases.json @@ -18,7 +18,8 @@ { "id": "fixed-1", "valid": true, "input": "fixed[1]", "decoded": { "type": "fixed", "length": 1 }, "canonical": "fixed[1]", "clause": "Appendix C: fixed canonical string is fixed[]", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "fixed-16", "valid": true, "input": "fixed[16]", "decoded": { "type": "fixed", "length": 16 }, "canonical": "fixed[16]", "clause": "Appendix C: fixed canonical string is fixed[]", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "decimal-9-2", "valid": true, "input": "decimal(9,2)", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "clause": "Appendix C: both decimal(9,2) and decimal(9, 2) are canonical, so no byte-exact form is pinned", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, - { "id": "decimal-9-2-spaced", "valid": true, "input": "decimal(9, 2)", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "clause": "Appendix C: the spaced decimal(9, 2) form parses to the same decimal", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "decimal-9-2-spaced", "valid": true, "input": "decimal(9, 2)", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "clause": "Appendix C lists both decimal(9,2) and decimal(9, 2) as canonical examples, so decimal(9, 2) must be accepted", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "decimal-9-2-spaced-params", "valid": true, "normative_level": "should", "input": "decimal( 9 , 2 )", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "clause": "Appendix C: readers should accept optional whitespace around parameters and separators; a reader that rejects decimal( 9 , 2 ) is still conformant", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "decimal-38-0", "valid": true, "input": "decimal(38,0)", "decoded": { "type": "decimal", "precision": 38, "scale": 0 }, "clause": "Primitive Types: decimal precision must be 38 or less (38 is the maximum)", "spec_ref": "format/spec.md#primitive-types" }, { "id": "decimal-precision-over-38", "valid": false, "input": "decimal(39,0)", "clause": "Primitive Types: decimal precision must be 38 or less", "spec_ref": "format/spec.md#primitive-types" }, { "id": "decimal-missing-scale", "valid": false, "input": "decimal(9)", "clause": "Appendix C: decimal is written decimal(P,S); scale is required", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, diff --git a/table-spec/types/variant/cases.json b/table-spec/types/variant/cases.json index ace69ac..4d92865 100644 --- a/table-spec/types/variant/cases.json +++ b/table-spec/types/variant/cases.json @@ -1,5 +1,6 @@ { "cases": [ - { "id": "variant", "valid": true, "input": "variant", "decoded": { "type": "variant" }, "canonical": "variant", "clause": "Semi-structured Types: variant added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" } + { "id": "variant", "valid": true, "input": "variant", "decoded": { "type": "variant" }, "canonical": "variant", "clause": "Semi-structured Types: variant added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "variant-with-params", "valid": false, "input": "variant(x)", "clause": "Appendix C: variant is a bare type name and takes no parameters", "spec_ref": "format/spec.md#appendix-c-json-serialization" } ] } From 9e3e3f3a678aae6ec1c69a62fa267dba281580a0 Mon Sep 17 00:00:00 2001 From: Neelesh Salian Date: Fri, 25 Sep 2026 18:20:29 -0700 Subject: [PATCH 3/4] spaced decimal canonical, skip state, should-tier geo defaults, decoded-forbid schema, self-test, check-license quoting --- .github/workflows/validate-fixtures.yml | 26 +++++++++++++-------- dev/.rat-excludes | 2 +- dev/check-license | 4 ++-- dev/schema/cases.base.schema.json | 2 +- dev/validate-fixtures.py | 2 +- table-spec/types/README.md | 13 ++++++----- table-spec/types/geospatial/cases.json | 30 ++++++++++++++++++------- table-spec/types/primitive/cases.json | 10 ++++----- 8 files changed, 57 insertions(+), 32 deletions(-) diff --git a/.github/workflows/validate-fixtures.yml b/.github/workflows/validate-fixtures.yml index 2e37580..d438a1f 100644 --- a/.github/workflows/validate-fixtures.yml +++ b/.github/workflows/validate-fixtures.yml @@ -46,13 +46,21 @@ jobs: run: python3 -m pip install -r dev/requirements.txt - name: Validate cases.json run: python3 dev/validate-fixtures.py - - name: Validator rejects a malformed fixture (self-test) + - name: Validator rejects malformed fixtures (self-test) run: | - tmp="$(mktemp -d)" - mkdir -p "$tmp/table-spec/types/x" - printf '%s' '{"cases":[{"id":"bad","valid":true,"input":"int","clause":"c","spec_ref":"s"}]}' > "$tmp/table-spec/types/x/cases.json" - if python3 dev/validate-fixtures.py "$tmp"; then - echo "self-test FAILED: validator accepted a malformed fixture (valid:true with no decoded)" - exit 1 - fi - echo "self-test ok: validator rejected the malformed fixture" + set -e + assert_reject() { # + d="$(mktemp -d)" + mkdir -p "$d/table-spec/types/x" + printf '%s' "$2" > "$d/table-spec/types/x/cases.json" + test -f "$d/table-spec/types/x/cases.json" + if python3 dev/validate-fixtures.py "$d" >/dev/null 2>&1; then + echo "self-test FAILED: validator accepted a malformed fixture ($1)"; exit 1 + fi + echo "self-test ok: rejected ($1)" + rm -rf "$d" + } + assert_reject "valid:true with no decoded" '{"cases":[{"id":"a","valid":true,"input":"int","clause":"c","spec_ref":"s"}]}' + assert_reject "valid:false carrying decoded" '{"cases":[{"id":"a","valid":false,"input":"bad","decoded":{"type":"int"},"clause":"c","spec_ref":"s"}]}' + assert_reject "unknown property" '{"cases":[{"id":"a","valid":true,"input":"int","decoded":{"type":"int"},"clause":"c","spec_ref":"s","spec-ref":"typo"}]}' + assert_reject "duplicate id in a surface" '{"cases":[{"id":"a","valid":true,"input":"int","decoded":{"type":"int"},"clause":"c","spec_ref":"s"},{"id":"a","valid":true,"input":"long","decoded":{"type":"long"},"clause":"c","spec_ref":"s"}]}' diff --git a/dev/.rat-excludes b/dev/.rat-excludes index 48de711..bc614e9 100644 --- a/dev/.rat-excludes +++ b/dev/.rat-excludes @@ -21,6 +21,6 @@ NOTICE lib # Fixture data and JSON Schemas have no comment syntax, so they cannot carry an -# Apache header; the surface README states the corpus is AL2-licensed. +# Apache header; they are covered by the repository LICENSE and NOTICE at the root. **/*.json **/*.jsonl diff --git a/dev/check-license b/dev/check-license index 3c951b9..c7440d5 100755 --- a/dev/check-license +++ b/dev/check-license @@ -29,7 +29,7 @@ acquire_rat_jar () { if [ $(command -v curl) ]; then curl -L --silent "${URL}" > "$JAR_DL" && mv "$JAR_DL" "$JAR" elif [ $(command -v wget) ]; then - wget --quiet ${URL} -O "$JAR_DL" && mv "$JAR_DL" "$JAR" + wget --quiet "${URL}" -O "$JAR_DL" && mv "$JAR_DL" "$JAR" else printf "You do not have curl or wget installed, please install rat manually.\n" exit 1 @@ -67,7 +67,7 @@ mkdir -p "$FWDIR/lib" # --input-exclude-std GIT skips .git and anything .gitignore'd. Without it RAT # flags the .git pointer *file* that git worktrees use, since its built-in SCM # exclusion only covers .git directories. -$java_cmd -jar "$rat_jar" \ +"$java_cmd" -jar "$rat_jar" \ --input-exclude-file "$FWDIR/dev/.rat-excludes" \ --input-exclude-std GIT IDEA MAC \ --input-include-std HIDDEN_DIR \ diff --git a/dev/schema/cases.base.schema.json b/dev/schema/cases.base.schema.json index 60fb70b..d184879 100644 --- a/dev/schema/cases.base.schema.json +++ b/dev/schema/cases.base.schema.json @@ -36,7 +36,7 @@ { "$comment": "an invalid case must not carry decoded", "if": { "properties": { "valid": { "const": false } }, "required": ["valid"] }, - "then": { "not": { "required": ["decoded"] } } + "then": { "properties": { "decoded": false } } }, { "$comment": "canonical is only allowed on a valid case", diff --git a/dev/validate-fixtures.py b/dev/validate-fixtures.py index b1548c2..97ebb40 100755 --- a/dev/validate-fixtures.py +++ b/dev/validate-fixtures.py @@ -19,7 +19,7 @@ """Validate the conformance type fixtures against the JSON Schemas in dev/schema/. Each cases.json is validated against the base structural schema and the types schema -(which requires clause and spec_ref); case ids must be globally unique. +(which requires clause and spec_ref); case ids must be unique within a surface. """ import glob diff --git a/table-spec/types/README.md b/table-spec/types/README.md index 9593d3a..7b4cf6e 100644 --- a/table-spec/types/README.md +++ b/table-spec/types/README.md @@ -40,9 +40,9 @@ language's representation. - `valid: false` - the parser must reject `input`. A rejection passes; a successful parse fails. -`canonical` is present only where the spec pins one spelling. `decimal` has two -blessed forms (`decimal(9,2)` and `decimal(9, 2)`), so its cases have no -`canonical` and are compared by `decoded` alone. +`canonical` is present only where the spec pins one spelling. Appendix C's canonical +for `decimal` is the spaced `decimal(

, )`; the no-space `decimal(9,2)` is an +accepted read form that re-serializes to the canonical `decimal(9, 2)`. ## Scope @@ -95,8 +95,11 @@ so a future surface may reuse a name like `int` or `string`. Cases are normative MUST by default. A case marked `normative_level: "should"` is advisory: the spec only recommends the behavior, so a reader that diverges is still conformant. Example: `decimal( 9 , 2 )`, since readers *should* accept optional -whitespace around parameters and separators (Appendix C). A runner must report a -failed `should` case distinctly from a pass, so the advisory tier stays visible. +whitespace around parameters and separators (Appendix C). A runner reports one of +`pass`, `fail` (a MUST violation, the only state that makes a run nonzero), +`advisory_fail` (an unmet SHOULD, reported but never blocking), or `skip` (the case was +not run: the surface is not subscribed, or the type is absent in the implementation), so +the advisory tier stays visible rather than reading as inert. ## Provenance diff --git a/table-spec/types/geospatial/cases.json b/table-spec/types/geospatial/cases.json index bae03ea..b95520d 100644 --- a/table-spec/types/geospatial/cases.json +++ b/table-spec/types/geospatial/cases.json @@ -3,10 +3,11 @@ { "id": "geometry-crs84", "valid": true, + "normative_level": "should", "input": "geometry(OGC:CRS84)", "decoded": {"type": "geometry", "crs": "OGC:CRS84"}, "canonical": "geometry(OGC:CRS84)", - "clause": "geometry(C) with explicit CRS; the canonical serialized form is unquoted \"geometry()\"", + "clause": "geometry(C) with explicit CRS; Appendix C's canonical is fully parameterized, but when C is the default OGC:CRS84 whether a writer may elide it is unsettled, so the write direction is advisory", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { @@ -21,28 +22,41 @@ { "id": "geometry-default-crs", "valid": true, + "normative_level": "should", "input": "geometry", "decoded": {"type": "geometry", "crs": "OGC:CRS84"}, "canonical": "geometry(OGC:CRS84)", - "clause": "geometry(C): if C is not specified, C is OGC:CRS84; the canonical serialized form is fully parameterized (Appendix C)", + "clause": "geometry(C): if C is not specified, C is OGC:CRS84; Appendix C's canonical is fully parameterized, but whether a writer may elide the default is unsettled, so the write direction is advisory", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "geography-crs84-spherical", "valid": true, + "normative_level": "should", "input": "geography(OGC:CRS84, spherical)", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"}, "canonical": "geography(OGC:CRS84, spherical)", - "clause": "geography(C, A); the canonical serialized form is unquoted \"geography(, )\"", + "clause": "geography(C, A) with explicit CRS and algorithm; Appendix C's canonical is fully parameterized, but when C/A are the defaults OGC:CRS84/spherical whether a writer may elide them is unsettled, so the write direction is advisory", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "geography-default", "valid": true, + "normative_level": "should", "input": "geography", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"}, "canonical": "geography(OGC:CRS84, spherical)", - "clause": "geography(C, A): if not specified, C is OGC:CRS84 and A is spherical; the canonical serialized form is fully parameterized (Appendix C)", + "clause": "geography(C, A): if not specified, C is OGC:CRS84 and A is spherical; Appendix C's canonical is fully parameterized, but whether a writer may elide the default is unsettled, so the write direction is advisory", + "spec_ref": "format/spec.md#appendix-c-json-serialization" + }, + { + "id": "geography-crs-only", + "valid": true, + "normative_level": "should", + "input": "geography(OGC:CRS84)", + "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"}, + "canonical": "geography(OGC:CRS84, spherical)", + "clause": "geography(C, A): if A is unspecified but C is given, A defaults to spherical (Java reference behavior; Appendix C defines geography(, ) as canonical)", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { @@ -51,7 +65,7 @@ "input": "geography(OGC:CRS84, vincenty)", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "vincenty"}, "canonical": "geography(OGC:CRS84, vincenty)", - "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney", + "clause": "geography edge-interpolation algorithm vincenty (a v3 allowed algorithm)", "spec_ref": "format/spec.md#primitive-types" }, { @@ -60,7 +74,7 @@ "input": "geography(OGC:CRS84, thomas)", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "thomas"}, "canonical": "geography(OGC:CRS84, thomas)", - "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney", + "clause": "geography edge-interpolation algorithm thomas (a v3 allowed algorithm)", "spec_ref": "format/spec.md#primitive-types" }, { @@ -69,7 +83,7 @@ "input": "geography(OGC:CRS84, andoyer)", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "andoyer"}, "canonical": "geography(OGC:CRS84, andoyer)", - "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney", + "clause": "geography edge-interpolation algorithm andoyer (a v3 allowed algorithm)", "spec_ref": "format/spec.md#primitive-types" }, { @@ -78,7 +92,7 @@ "input": "geography(OGC:CRS84, karney)", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "karney"}, "canonical": "geography(OGC:CRS84, karney)", - "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney", + "clause": "geography edge-interpolation algorithm karney (a v3 allowed algorithm)", "spec_ref": "format/spec.md#primitive-types" }, { diff --git a/table-spec/types/primitive/cases.json b/table-spec/types/primitive/cases.json index 653b454..0b80383 100644 --- a/table-spec/types/primitive/cases.json +++ b/table-spec/types/primitive/cases.json @@ -9,16 +9,16 @@ { "id": "time", "valid": true, "input": "time", "decoded": { "type": "time" }, "canonical": "time", "clause": "Primitive Types: time; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "timestamp", "valid": true, "input": "timestamp", "decoded": { "type": "timestamp" }, "canonical": "timestamp", "clause": "Primitive Types: timestamp; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "timestamptz", "valid": true, "input": "timestamptz", "decoded": { "type": "timestamptz" }, "canonical": "timestamptz", "clause": "Primitive Types: timestamptz; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, - { "id": "timestamp_ns", "valid": true, "input": "timestamp_ns", "decoded": { "type": "timestamp_ns" }, "canonical": "timestamp_ns", "clause": "Primitive Types: timestamp_ns added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#primitive-types" }, - { "id": "timestamptz_ns", "valid": true, "input": "timestamptz_ns", "decoded": { "type": "timestamptz_ns" }, "canonical": "timestamptz_ns", "clause": "Primitive Types: timestamptz_ns added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#primitive-types" }, + { "id": "timestamp_ns", "valid": true, "input": "timestamp_ns", "decoded": { "type": "timestamp_ns" }, "canonical": "timestamp_ns", "clause": "Primitive Types: timestamp_ns added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "timestamptz_ns", "valid": true, "input": "timestamptz_ns", "decoded": { "type": "timestamptz_ns" }, "canonical": "timestamptz_ns", "clause": "Primitive Types: timestamptz_ns added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "string", "valid": true, "input": "string", "decoded": { "type": "string" }, "canonical": "string", "clause": "Primitive Types: string; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "uuid", "valid": true, "input": "uuid", "decoded": { "type": "uuid" }, "canonical": "uuid", "clause": "Primitive Types: uuid; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "binary", "valid": true, "input": "binary", "decoded": { "type": "binary" }, "canonical": "binary", "clause": "Primitive Types: binary; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, - { "id": "unknown", "valid": true, "input": "unknown", "decoded": { "type": "unknown" }, "canonical": "unknown", "clause": "Primitive Types: unknown added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#primitive-types" }, + { "id": "unknown", "valid": true, "input": "unknown", "decoded": { "type": "unknown" }, "canonical": "unknown", "clause": "Primitive Types: unknown added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "fixed-1", "valid": true, "input": "fixed[1]", "decoded": { "type": "fixed", "length": 1 }, "canonical": "fixed[1]", "clause": "Appendix C: fixed canonical string is fixed[]", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "fixed-16", "valid": true, "input": "fixed[16]", "decoded": { "type": "fixed", "length": 16 }, "canonical": "fixed[16]", "clause": "Appendix C: fixed canonical string is fixed[]", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, - { "id": "decimal-9-2", "valid": true, "input": "decimal(9,2)", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "clause": "Appendix C: both decimal(9,2) and decimal(9, 2) are canonical, so no byte-exact form is pinned", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, - { "id": "decimal-9-2-spaced", "valid": true, "input": "decimal(9, 2)", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "clause": "Appendix C lists both decimal(9,2) and decimal(9, 2) as canonical examples, so decimal(9, 2) must be accepted", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "decimal-9-2", "valid": true, "input": "decimal(9,2)", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "canonical": "decimal(9, 2)", "clause": "Appendix C: the canonical serialized form is the spaced decimal(

, ); the no-space input re-serializes to it", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, + { "id": "decimal-9-2-spaced", "valid": true, "input": "decimal(9, 2)", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "canonical": "decimal(9, 2)", "clause": "Appendix C canonical form is the spaced decimal(9, 2); input is already canonical and re-serializes unchanged", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "decimal-9-2-spaced-params", "valid": true, "normative_level": "should", "input": "decimal( 9 , 2 )", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "clause": "Appendix C: readers should accept optional whitespace around parameters and separators; a reader that rejects decimal( 9 , 2 ) is still conformant", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "decimal-38-0", "valid": true, "input": "decimal(38,0)", "decoded": { "type": "decimal", "precision": 38, "scale": 0 }, "clause": "Primitive Types: decimal precision must be 38 or less (38 is the maximum)", "spec_ref": "format/spec.md#primitive-types" }, { "id": "decimal-precision-over-38", "valid": false, "input": "decimal(39,0)", "clause": "Primitive Types: decimal precision must be 38 or less", "spec_ref": "format/spec.md#primitive-types" }, From 90ebb2d4555aec70aca4909b5ce35cdc952452bc Mon Sep 17 00:00:00 2001 From: Neelesh Salian Date: Tue, 29 Sep 2026 12:50:25 -0700 Subject: [PATCH 4/4] Address round 4 review: per-type decoded schema, geo read/write split, check-license return --- .github/workflows/validate-fixtures.yml | 4 ++ dev/check-license | 4 +- dev/schema/cases.types.schema.json | 60 ++++++++++++++++++++++++- table-spec/types/geospatial/cases.json | 28 +++++++++--- 4 files changed, 85 insertions(+), 11 deletions(-) diff --git a/.github/workflows/validate-fixtures.yml b/.github/workflows/validate-fixtures.yml index d438a1f..cc78359 100644 --- a/.github/workflows/validate-fixtures.yml +++ b/.github/workflows/validate-fixtures.yml @@ -64,3 +64,7 @@ jobs: assert_reject "valid:false carrying decoded" '{"cases":[{"id":"a","valid":false,"input":"bad","decoded":{"type":"int"},"clause":"c","spec_ref":"s"}]}' assert_reject "unknown property" '{"cases":[{"id":"a","valid":true,"input":"int","decoded":{"type":"int"},"clause":"c","spec_ref":"s","spec-ref":"typo"}]}' assert_reject "duplicate id in a surface" '{"cases":[{"id":"a","valid":true,"input":"int","decoded":{"type":"int"},"clause":"c","spec_ref":"s"},{"id":"a","valid":true,"input":"long","decoded":{"type":"long"},"clause":"c","spec_ref":"s"}]}' + assert_reject "decoded decimal typo (precison)" '{"cases":[{"id":"a","valid":true,"input":"decimal(9,2)","decoded":{"type":"decimal","precison":9,"scale":2},"clause":"c","spec_ref":"s"}]}' + assert_reject "decoded fixed carrying len not length" '{"cases":[{"id":"a","valid":true,"input":"fixed[16]","decoded":{"type":"fixed","len":16},"clause":"c","spec_ref":"s"}]}' + assert_reject "decoded unknown type name" '{"cases":[{"id":"a","valid":true,"input":"x","decoded":{"type":"notatype"},"clause":"c","spec_ref":"s"}]}' + assert_reject "decoded simple type with extra key" '{"cases":[{"id":"a","valid":true,"input":"int","decoded":{"type":"int","precision":9},"clause":"c","spec_ref":"s"}]}' diff --git a/dev/check-license b/dev/check-license index c7440d5..be3aead 100755 --- a/dev/check-license +++ b/dev/check-license @@ -32,7 +32,7 @@ acquire_rat_jar () { wget --quiet "${URL}" -O "$JAR_DL" && mv "$JAR_DL" "$JAR" else printf "You do not have curl or wget installed, please install rat manually.\n" - exit 1 + return 1 fi fi @@ -41,7 +41,7 @@ acquire_rat_jar () { # We failed to download rm "$JAR" printf "Our attempt to download rat locally to ${JAR} failed. Please install rat manually.\n" - exit 1 + return 1 fi } diff --git a/dev/schema/cases.types.schema.json b/dev/schema/cases.types.schema.json index bd010dc..03c179a 100644 --- a/dev/schema/cases.types.schema.json +++ b/dev/schema/cases.types.schema.json @@ -2,7 +2,7 @@ "$schema": "https://json-schema.org/draft/2020-12/schema", "$id": "https://iceberg.apache.org/verification/cases.types.schema.json", "title": "Conformance cases (types surface extension)", - "description": "Extra constraints for table-spec/types: every case cites the spec. Applied on top of cases.base.schema.json by dev/validate-fixtures.py.", + "description": "Extra constraints for table-spec/types: every case cites the spec, and a case's decoded matches the per-type shape from Appendix C. Applied on top of cases.base.schema.json by dev/validate-fixtures.py.", "type": "object", "required": ["cases"], "properties": { @@ -13,9 +13,65 @@ "required": ["clause", "spec_ref"], "properties": { "clause": { "type": "string", "minLength": 1 }, - "spec_ref": { "type": "string", "minLength": 1 } + "spec_ref": { "type": "string", "minLength": 1 }, + "decoded": { "$ref": "#/$defs/decodedType" } } } } + }, + "$defs": { + "decodedType": { + "type": "object", + "required": ["type"], + "properties": { + "type": { "enum": ["boolean", "int", "long", "float", "double", "date", "time", "timestamp", "timestamptz", "timestamp_ns", "timestamptz_ns", "string", "uuid", "binary", "unknown", "variant", "decimal", "fixed", "geometry", "geography", "struct", "list", "map"] } + }, + "allOf": [ + { + "$comment": "simple types carry only type", + "if": { "required": ["type"], "properties": { "type": { "enum": ["boolean", "int", "long", "float", "double", "date", "time", "timestamp", "timestamptz", "timestamp_ns", "timestamptz_ns", "string", "uuid", "binary", "unknown", "variant"] } } }, + "then": { "properties": { "type": {} }, "additionalProperties": false } + }, + { + "if": { "required": ["type"], "properties": { "type": { "const": "decimal" } } }, + "then": { "required": ["precision", "scale"], "properties": { "type": {}, "precision": { "type": "integer" }, "scale": { "type": "integer" } }, "additionalProperties": false } + }, + { + "if": { "required": ["type"], "properties": { "type": { "const": "fixed" } } }, + "then": { "required": ["length"], "properties": { "type": {}, "length": { "type": "integer" } }, "additionalProperties": false } + }, + { + "if": { "required": ["type"], "properties": { "type": { "const": "geometry" } } }, + "then": { "required": ["crs"], "properties": { "type": {}, "crs": { "type": "string" } }, "additionalProperties": false } + }, + { + "if": { "required": ["type"], "properties": { "type": { "const": "geography" } } }, + "then": { "required": ["crs", "algorithm"], "properties": { "type": {}, "crs": { "type": "string" }, "algorithm": { "type": "string" } }, "additionalProperties": false } + }, + { + "if": { "required": ["type"], "properties": { "type": { "const": "struct" } } }, + "then": { "required": ["fields"], "properties": { "type": {}, "fields": { "type": "array", "items": { "$ref": "#/$defs/field" } } }, "additionalProperties": false } + }, + { + "if": { "required": ["type"], "properties": { "type": { "const": "list" } } }, + "then": { "required": ["element-id", "element-required", "element"], "properties": { "type": {}, "element-id": { "type": "integer" }, "element-required": { "type": "boolean" }, "element": { "$ref": "#/$defs/decodedType" } }, "additionalProperties": false } + }, + { + "if": { "required": ["type"], "properties": { "type": { "const": "map" } } }, + "then": { "required": ["key-id", "key", "value-id", "value-required", "value"], "properties": { "type": {}, "key-id": { "type": "integer" }, "key": { "$ref": "#/$defs/decodedType" }, "value-id": { "type": "integer" }, "value-required": { "type": "boolean" }, "value": { "$ref": "#/$defs/decodedType" } }, "additionalProperties": false } + } + ] + }, + "field": { + "type": "object", + "required": ["id", "name", "required", "type"], + "properties": { + "id": { "type": "integer" }, + "name": { "type": "string" }, + "required": { "type": "boolean" }, + "type": { "$ref": "#/$defs/decodedType" } + }, + "additionalProperties": false + } } } diff --git a/table-spec/types/geospatial/cases.json b/table-spec/types/geospatial/cases.json index b95520d..20a0f54 100644 --- a/table-spec/types/geospatial/cases.json +++ b/table-spec/types/geospatial/cases.json @@ -3,11 +3,10 @@ { "id": "geometry-crs84", "valid": true, - "normative_level": "should", "input": "geometry(OGC:CRS84)", "decoded": {"type": "geometry", "crs": "OGC:CRS84"}, "canonical": "geometry(OGC:CRS84)", - "clause": "geometry(C) with explicit CRS; Appendix C's canonical is fully parameterized, but when C is the default OGC:CRS84 whether a writer may elide it is unsettled, so the write direction is advisory", + "clause": "geometry(C) with explicit CRS; the canonical serialized form is the unquoted \"geometry()\"", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { @@ -22,31 +21,46 @@ { "id": "geometry-default-crs", "valid": true, + "input": "geometry", + "decoded": {"type": "geometry", "crs": "OGC:CRS84"}, + "clause": "geometry(C): if C is not specified, C is OGC:CRS84 (spec.md:284-285); the read/parse direction is a MUST", + "spec_ref": "format/spec.md#primitive-types" + }, + { + "id": "geometry-default-crs-write", + "valid": true, "normative_level": "should", "input": "geometry", "decoded": {"type": "geometry", "crs": "OGC:CRS84"}, "canonical": "geometry(OGC:CRS84)", - "clause": "geometry(C): if C is not specified, C is OGC:CRS84; Appendix C's canonical is fully parameterized, but whether a writer may elide the default is unsettled, so the write direction is advisory", + "clause": "geometry(C) re-serialize: Appendix C's canonical is fully parameterized, but whether a writer may elide the default is unsettled, so the write direction is advisory", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "geography-crs84-spherical", "valid": true, - "normative_level": "should", "input": "geography(OGC:CRS84, spherical)", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"}, "canonical": "geography(OGC:CRS84, spherical)", - "clause": "geography(C, A) with explicit CRS and algorithm; Appendix C's canonical is fully parameterized, but when C/A are the defaults OGC:CRS84/spherical whether a writer may elide them is unsettled, so the write direction is advisory", + "clause": "geography(C, A) with explicit CRS and algorithm; the canonical serialized form is the unquoted \"geography(, )\"", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { "id": "geography-default", "valid": true, + "input": "geography", + "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"}, + "clause": "geography(C, A): if not specified, C is OGC:CRS84 and A is spherical (spec.md:284-285); the read/parse direction is a MUST", + "spec_ref": "format/spec.md#primitive-types" + }, + { + "id": "geography-default-write", + "valid": true, "normative_level": "should", "input": "geography", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"}, "canonical": "geography(OGC:CRS84, spherical)", - "clause": "geography(C, A): if not specified, C is OGC:CRS84 and A is spherical; Appendix C's canonical is fully parameterized, but whether a writer may elide the default is unsettled, so the write direction is advisory", + "clause": "geography(C, A) re-serialize: Appendix C's canonical is fully parameterized, but whether a writer may elide the defaults is unsettled, so the write direction is advisory", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, { @@ -56,7 +70,7 @@ "input": "geography(OGC:CRS84)", "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"}, "canonical": "geography(OGC:CRS84, spherical)", - "clause": "geography(C, A): if A is unspecified but C is given, A defaults to spherical (Java reference behavior; Appendix C defines geography(, ) as canonical)", + "clause": "geography(C): the 1-arg form (CRS given, algorithm omitted) is not in Appendix C grammar; cross-checked against Java reference behavior, which defaults A to spherical and re-serializes to the Appendix C canonical geography(, )", "spec_ref": "format/spec.md#appendix-c-json-serialization" }, {