-SNAPSHOT to test dev HEAD via the ASF snapshot.
+def icebergVersion = (project.findProperty('icebergVersion') ?: '1.11.0').toString()
+
+repositories {
+ mavenCentral()
+ // Only consulted for SNAPSHOTs (the nightly lane); releases resolve from Central.
+ if (icebergVersion.endsWith('SNAPSHOT')) {
+ maven { url = uri('https://repository.apache.org/content/repositories/snapshots') }
+ }
+}
+
+java {
+ toolchain {
+ languageVersion = JavaLanguageVersion.of(17)
+ }
+}
+
+dependencies {
+ // The implementation under test. iceberg-api carries the primitive/geo type
+ // parser (org.apache.iceberg.types.Types.fromTypeName); iceberg-core carries
+ // the nested-type JSON reader (org.apache.iceberg.SchemaParser.fromJson).
+ // dev/run-local.sh overrides these with --include-build onto a clone of apache/main.
+ implementation "org.apache.iceberg:iceberg-api:${icebergVersion}"
+ implementation "org.apache.iceberg:iceberg-core:${icebergVersion}"
+ // Only used to read the cases.json fixtures.
+ implementation 'com.fasterxml.jackson.core:jackson-databind:2.18.2'
+}
+
+application {
+ mainClass = 'org.apache.iceberg.conformance.VerifyTypes'
+}
+
+// Google Java Format, matching Apache Iceberg. Run `./gradlew spotlessApply`.
+spotless {
+ java {
+ googleJavaFormat("1.22.0")
+ removeUnusedImports()
+ trimTrailingWhitespace()
+ endWithNewline()
+ }
+}
+
+// `./gradlew run` walks up to the repo root (the directory containing
+// table-spec/) on its own, so no args are needed.
diff --git a/runners/java/gradle/wrapper/gradle-wrapper.jar b/runners/java/gradle/wrapper/gradle-wrapper.jar
new file mode 100644
index 0000000..1b33c55
Binary files /dev/null and b/runners/java/gradle/wrapper/gradle-wrapper.jar differ
diff --git a/runners/java/gradle/wrapper/gradle-wrapper.properties b/runners/java/gradle/wrapper/gradle-wrapper.properties
new file mode 100644
index 0000000..a4cf193
--- /dev/null
+++ b/runners/java/gradle/wrapper/gradle-wrapper.properties
@@ -0,0 +1,8 @@
+distributionBase=GRADLE_USER_HOME
+distributionPath=wrapper/dists
+distributionSha256Sum=6f74b601422d6d6fc4e1f9a1ab6522f642c2fdcbc15ae33ebd30ba3d7198e854
+distributionUrl=https\://services.gradle.org/distributions/gradle-8.14.5-bin.zip
+networkTimeout=10000
+validateDistributionUrl=true
+zipStoreBase=GRADLE_USER_HOME
+zipStorePath=wrapper/dists
diff --git a/runners/java/gradlew b/runners/java/gradlew
new file mode 100755
index 0000000..30403b3
--- /dev/null
+++ b/runners/java/gradlew
@@ -0,0 +1,253 @@
+#!/bin/sh
+
+#
+# Copyright © 2015-2021 the original authors.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# https://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+
+##############################################################################
+#
+# Gradle start up script for POSIX generated by Gradle.
+#
+# Important for running:
+#
+# (1) You need a POSIX-compliant shell to run this script. If your /bin/sh is
+# noncompliant, but you have some other compliant shell such as ksh or
+# bash, then to run this script, type that shell name before the whole
+# command line, like:
+#
+# ksh Gradle
+#
+# Busybox and similar reduced shells will NOT work, because this script
+# requires all of these POSIX shell features:
+# * functions;
+# * expansions «$var», «${var}», «${var:-default}», «${var+SET}»,
+# «${var#prefix}», «${var%suffix}», and «$( cmd )»;
+# * compound commands having a testable exit status, especially «case»;
+# * various built-in commands including «command», «set», and «ulimit».
+#
+# Important for patching:
+#
+# (2) This script targets any POSIX shell, so it avoids extensions provided
+# by Bash, Ksh, etc; in particular arrays are avoided.
+#
+# The "traditional" practice of packing multiple parameters into a
+# space-separated string is a well documented source of bugs and security
+# problems, so this is (mostly) avoided, by progressively accumulating
+# options in "$@", and eventually passing that to Java.
+#
+# Where the inherited environment variables (DEFAULT_JVM_OPTS, JAVA_OPTS,
+# and GRADLE_OPTS) rely on word-splitting, this is performed explicitly;
+# see the in-line comments for details.
+#
+# There are tweaks for specific operating systems such as AIX, CygWin,
+# Darwin, MinGW, and NonStop.
+#
+# (3) This script is generated from the Groovy template
+# https://github.com/gradle/gradle/blob/HEAD/platforms/jvm/plugins-application/src/main/resources/org/gradle/api/internal/plugins/unixStartScript.txt
+# within the Gradle project.
+#
+# You can find Gradle at https://github.com/gradle/gradle/.
+#
+##############################################################################
+
+# Attempt to set APP_HOME
+
+# Resolve links: $0 may be a link
+app_path=$0
+
+# Need this for daisy-chained symlinks.
+while
+ APP_HOME=${app_path%"${app_path##*/}"} # leaves a trailing /; empty if no leading path
+ [ -h "$app_path" ]
+do
+ ls=$( ls -ld "$app_path" )
+ link=${ls#*' -> '}
+ case $link in #(
+ /*) app_path=$link ;; #(
+ *) app_path=$APP_HOME$link ;;
+ esac
+done
+
+# This is normally unused
+# shellcheck disable=SC2034
+APP_BASE_NAME=${0##*/}
+# Discard cd standard output in case $CDPATH is set (https://github.com/gradle/gradle/issues/25036)
+APP_HOME=$( cd -P "${APP_HOME:-./}" > /dev/null && printf '%s\n' "$PWD" ) || exit
+
+if [ ! -e $APP_HOME/gradle/wrapper/gradle-wrapper.jar ]; then
+ curl -o $APP_HOME/gradle/wrapper/gradle-wrapper.jar https://raw.githubusercontent.com/gradle/gradle/v8.14.5/gradle/wrapper/gradle-wrapper.jar
+fi
+
+# Use the maximum available, or set MAX_FD != -1 to use that value.
+MAX_FD=maximum
+
+warn () {
+ echo "$*"
+} >&2
+
+die () {
+ echo
+ echo "$*"
+ echo
+ exit 1
+} >&2
+
+# OS specific support (must be 'true' or 'false').
+cygwin=false
+msys=false
+darwin=false
+nonstop=false
+case "$( uname )" in #(
+ CYGWIN* ) cygwin=true ;; #(
+ Darwin* ) darwin=true ;; #(
+ MSYS* | MINGW* ) msys=true ;; #(
+ NONSTOP* ) nonstop=true ;;
+esac
+
+CLASSPATH=$APP_HOME/gradle/wrapper/gradle-wrapper.jar
+
+
+# Determine the Java command to use to start the JVM.
+if [ -n "$JAVA_HOME" ] ; then
+ if [ -x "$JAVA_HOME/jre/sh/java" ] ; then
+ # IBM's JDK on AIX uses strange locations for the executables
+ JAVACMD=$JAVA_HOME/jre/sh/java
+ else
+ JAVACMD=$JAVA_HOME/bin/java
+ fi
+ if [ ! -x "$JAVACMD" ] ; then
+ die "ERROR: JAVA_HOME is set to an invalid directory: $JAVA_HOME
+
+Please set the JAVA_HOME variable in your environment to match the
+location of your Java installation."
+ fi
+else
+ JAVACMD=java
+ if ! command -v java >/dev/null 2>&1
+ then
+ die "ERROR: JAVA_HOME is not set and no 'java' command could be found in your PATH.
+
+Please set the JAVA_HOME variable in your environment to match the
+location of your Java installation."
+ fi
+fi
+
+# Increase the maximum file descriptors if we can.
+if ! "$cygwin" && ! "$darwin" && ! "$nonstop" ; then
+ case $MAX_FD in #(
+ max*)
+ # In POSIX sh, ulimit -H is undefined. That's why the result is checked to see if it worked.
+ # shellcheck disable=SC2039,SC3045
+ MAX_FD=$( ulimit -H -n ) ||
+ warn "Could not query maximum file descriptor limit"
+ esac
+ case $MAX_FD in #(
+ '' | soft) :;; #(
+ *)
+ # In POSIX sh, ulimit -n is undefined. That's why the result is checked to see if it worked.
+ # shellcheck disable=SC2039,SC3045
+ ulimit -n "$MAX_FD" ||
+ warn "Could not set maximum file descriptor limit to $MAX_FD"
+ esac
+fi
+
+# Collect all arguments for the java command, stacking in reverse order:
+# * args from the command line
+# * the main class name
+# * -classpath
+# * -D...appname settings
+# * --module-path (only if needed)
+# * DEFAULT_JVM_OPTS, JAVA_OPTS, and GRADLE_OPTS environment variables.
+
+# For Cygwin or MSYS, switch paths to Windows format before running java
+if "$cygwin" || "$msys" ; then
+ APP_HOME=$( cygpath --path --mixed "$APP_HOME" )
+ CLASSPATH=$( cygpath --path --mixed "$CLASSPATH" )
+
+ JAVACMD=$( cygpath --unix "$JAVACMD" )
+
+ # Now convert the arguments - kludge to limit ourselves to /bin/sh
+ for arg do
+ if
+ case $arg in #(
+ -*) false ;; # don't mess with options #(
+ /?*) t=${arg#/} t=/${t%%/*} # looks like a POSIX filepath
+ [ -e "$t" ] ;; #(
+ *) false ;;
+ esac
+ then
+ arg=$( cygpath --path --ignore --mixed "$arg" )
+ fi
+ # Roll the args list around exactly as many times as the number of
+ # args, so each arg winds up back in the position where it started, but
+ # possibly modified.
+ #
+ # NB: a `for` loop captures its iteration list before it begins, so
+ # changing the positional parameters here affects neither the number of
+ # iterations, nor the values presented in `arg`.
+ shift # remove old arg
+ set -- "$@" "$arg" # push replacement arg
+ done
+fi
+
+
+# Add default JVM options here. You can also use JAVA_OPTS and GRADLE_OPTS to pass JVM options to this script.
+DEFAULT_JVM_OPTS='"-Xmx64m" "-Xms64m"'
+
+# Collect all arguments for the java command:
+# * DEFAULT_JVM_OPTS, JAVA_OPTS, and optsEnvironmentVar are not allowed to contain shell fragments,
+# and any embedded shellness will be escaped.
+# * For example: A user cannot expect ${Hostname} to be expanded, as it is an environment variable and will be
+# treated as '${Hostname}' itself on the command line.
+
+set -- \
+ "-Dorg.gradle.appname=$APP_BASE_NAME" \
+ -classpath "$CLASSPATH" \
+ org.gradle.wrapper.GradleWrapperMain \
+ "$@"
+
+# Stop when "xargs" is not available.
+if ! command -v xargs >/dev/null 2>&1
+then
+ die "xargs is not available"
+fi
+
+# Use "xargs" to parse quoted args.
+#
+# With -n1 it outputs one arg per line, with the quotes and backslashes removed.
+#
+# In Bash we could simply go:
+#
+# readarray ARGS < <( xargs -n1 <<<"$var" ) &&
+# set -- "${ARGS[@]}" "$@"
+#
+# but POSIX shell has neither arrays nor command substitution, so instead we
+# post-process each arg (as a line of input to sed) to backslash-escape any
+# character that might be a shell metacharacter, then use eval to reverse
+# that process (while maintaining the separation between arguments), and wrap
+# the whole thing up as a single "set" statement.
+#
+# This will of course break if any of these variables contains a newline or
+# an unmatched quote.
+#
+
+eval "set -- $(
+ printf '%s\n' "$DEFAULT_JVM_OPTS $JAVA_OPTS $GRADLE_OPTS" |
+ xargs -n1 |
+ sed ' s~[^-[:alnum:]+,./:=@_]~\\&~g; ' |
+ tr '\n' ' '
+ )" '"$@"'
+
+exec "$JAVACMD" "$@"
diff --git a/runners/java/gradlew.bat b/runners/java/gradlew.bat
new file mode 100644
index 0000000..07742d4
--- /dev/null
+++ b/runners/java/gradlew.bat
@@ -0,0 +1,95 @@
+@rem
+@rem Copyright 2015 the original author or authors.
+@rem
+@rem Licensed under the Apache License, Version 2.0 (the "License");
+@rem you may not use this file except in compliance with the License.
+@rem You may obtain a copy of the License at
+@rem
+@rem https://www.apache.org/licenses/LICENSE-2.0
+@rem
+@rem Unless required by applicable law or agreed to in writing, software
+@rem distributed under the License is distributed on an "AS IS" BASIS,
+@rem WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+@rem See the License for the specific language governing permissions and
+@rem limitations under the License.
+@rem
+@rem SPDX-License-Identifier: Apache-2.0
+@rem
+
+@if "%DEBUG%"=="" @echo off
+@rem ##########################################################################
+@rem
+@rem Gradle startup script for Windows
+@rem
+@rem ##########################################################################
+
+@rem Set local scope for the variables with windows NT shell
+if "%OS%"=="Windows_NT" setlocal
+
+set DIRNAME=%~dp0
+if "%DIRNAME%"=="" set DIRNAME=.\
+
+@rem This is normally unused
+set APP_BASE_NAME=%~n0
+set APP_HOME=%DIRNAME%
+
+@rem Resolve any "." and ".." in APP_HOME to make it shorter.
+for %%i in ("%APP_HOME%") do set APP_HOME=%%~fi
+
+@rem Add default JVM options here. You can also use JAVA_OPTS and GRADLE_OPTS to pass JVM options to this script.
+set DEFAULT_JVM_OPTS="-Xmx64m" "-Xms64m"
+
+@rem Find java.exe
+if defined JAVA_HOME goto findJavaFromJavaHome
+
+set JAVA_EXE=java.exe
+%JAVA_EXE% -version >NUL 2>&1
+if %ERRORLEVEL% equ 0 goto execute
+
+echo. 1>&2
+echo ERROR: JAVA_HOME is not set and no 'java' command could be found in your PATH. 1>&2
+echo. 1>&2
+echo Please set the JAVA_HOME variable in your environment to match the 1>&2
+echo location of your Java installation. 1>&2
+
+goto fail
+
+:findJavaFromJavaHome
+set JAVA_HOME=%JAVA_HOME:"=%
+set JAVA_EXE=%JAVA_HOME%/bin/java.exe
+
+if exist "%JAVA_EXE%" goto execute
+
+echo. 1>&2
+echo ERROR: JAVA_HOME is set to an invalid directory: %JAVA_HOME% 1>&2
+echo. 1>&2
+echo Please set the JAVA_HOME variable in your environment to match the 1>&2
+echo location of your Java installation. 1>&2
+
+goto fail
+
+:execute
+@rem Setup the command line
+
+set CLASSPATH=%APP_HOME%\gradle\wrapper\gradle-wrapper.jar
+
+
+@rem Execute Gradle
+"%JAVA_EXE%" %DEFAULT_JVM_OPTS% %JAVA_OPTS% %GRADLE_OPTS% "-Dorg.gradle.appname=%APP_BASE_NAME%" -classpath "%CLASSPATH%" org.gradle.wrapper.GradleWrapperMain %*
+
+:end
+@rem End local scope for the variables with windows NT shell
+if %ERRORLEVEL% equ 0 goto mainEnd
+
+:fail
+rem Set variable GRADLE_EXIT_CONSOLE if you need the _script_ return code instead of
+rem the _cmd.exe /c_ return code!
+set EXIT_CODE=%ERRORLEVEL%
+if %EXIT_CODE% equ 0 set EXIT_CODE=1
+if not ""=="%GRADLE_EXIT_CONSOLE%" exit %EXIT_CODE%
+exit /b %EXIT_CODE%
+
+:mainEnd
+if "%OS%"=="Windows_NT" endlocal
+
+:omega
diff --git a/runners/java/settings.gradle b/runners/java/settings.gradle
new file mode 100644
index 0000000..8a899a2
--- /dev/null
+++ b/runners/java/settings.gradle
@@ -0,0 +1,20 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+rootProject.name = 'conformance-java'
diff --git a/runners/java/src/main/java/org/apache/iceberg/conformance/VerifyTypes.java b/runners/java/src/main/java/org/apache/iceberg/conformance/VerifyTypes.java
new file mode 100644
index 0000000..7a834f6
--- /dev/null
+++ b/runners/java/src/main/java/org/apache/iceberg/conformance/VerifyTypes.java
@@ -0,0 +1,364 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.iceberg.conformance;
+
+import com.fasterxml.jackson.databind.JsonNode;
+import com.fasterxml.jackson.databind.ObjectMapper;
+import java.io.IOException;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.nio.file.Paths;
+import java.util.ArrayList;
+import java.util.Comparator;
+import java.util.List;
+import java.util.stream.Collectors;
+import java.util.stream.Stream;
+import org.apache.iceberg.SchemaParser;
+import org.apache.iceberg.types.Type;
+import org.apache.iceberg.types.Types;
+
+/**
+ * Reference runner that checks Apache Iceberg (Java) against the type-surface conformance fixtures.
+ * It reads every {@code table-spec/**}/cases.json and parses each {@code input} with the published
+ * reader: a string input via {@link Types#fromTypeName(String)}, an object input (a nested type) via
+ * {@link org.apache.iceberg.SchemaParser#fromJson(String)}. It then applies the assertion contract:
+ *
+ *
+ * valid=false -> the parser must throw (PASS); returning a type is a FAIL
+ * valid=true -> parse ok and decoded shape == decoded (PASS); a mismatch is a FAIL
+ * valid=true, not modeled -> the parser throws because Java does not model the type (UNSUPPORTED)
+ *
+ *
+ * Java currently models every table-spec type, so the UNSUPPORTED path is dormant; the
+ * classification matches the Go and Rust runners regardless. The process exits 0 when no case
+ * fails, 1 on any FAIL, and 2 on a setup error such as a missing tree or an unreadable fixture.
+ */
+public final class VerifyTypes {
+
+ // Directories under the repo root that hold cases.
+ private static final String[] SURFACE_ROOTS = {"table-spec"};
+
+ private VerifyTypes() {}
+
+ public static void main(String[] args) {
+ Path start = Paths.get(System.getProperty("user.dir"));
+ List surfaces = new ArrayList<>();
+ for (int i = 0; i < args.length; i++) {
+ String a = args[i];
+ if (a.equals("--surface") && i + 1 < args.length) {
+ surfaces.add(args[++i]);
+ } else if (a.startsWith("--surface=")) {
+ surfaces.add(a.substring("--surface=".length()));
+ } else {
+ start = Paths.get(a);
+ }
+ }
+ String env = System.getenv("CONFORMANCE_SURFACES");
+ if (env != null) {
+ for (String s : env.split("[,\\s]+")) {
+ if (!s.isEmpty()) {
+ surfaces.add(s);
+ }
+ }
+ }
+ Path root = findRepoRoot(start.toAbsolutePath());
+ if (root == null) {
+ System.err.println("Could not locate repo root (no table-spec/) above " + start);
+ System.exit(2);
+ }
+
+ List cases;
+ try {
+ cases = loadCases(root, surfaces);
+ } catch (IOException e) {
+ // A fixture-read failure is a setup error (exit 2), not a conformance FAIL.
+ System.err.println("Setup error reading fixtures: " + e.getMessage());
+ System.exit(2);
+ return;
+ }
+ if (cases.isEmpty()) {
+ System.err.println("No cases found under " + root);
+ System.exit(2);
+ }
+
+ int pass = 0;
+ int fail = 0;
+ int unsupported = 0;
+ List failIds = new ArrayList<>();
+ List unsupportedIds = new ArrayList<>();
+
+ for (JsonNode node : cases) {
+ String id = node.get("id").asText();
+ JsonNode input = node.get("input");
+ boolean valid = node.get("valid").asBoolean();
+
+ Type parsed = null;
+ RuntimeException parseError = null;
+ try {
+ parsed = parseType(input);
+ } catch (RuntimeException e) {
+ parseError = e;
+ }
+
+ if (!valid) {
+ if (parseError != null) {
+ System.out.printf("%-32s PASS (rejected: %s)%n", id, parseError.getMessage());
+ pass++;
+ } else {
+ System.out.printf("%-32s FAIL (expected reject, parsed as \"%s\")%n", id, parsed);
+ fail++;
+ failIds.add(id);
+ }
+ continue;
+ }
+
+ // valid=true: a parser throw means Java does not model this type, not a failure.
+ if (parseError != null) {
+ System.out.printf("%-32s UNSUPPORTED (not modeled: %s)%n", id, parseError.getMessage());
+ unsupported++;
+ unsupportedIds.add(id);
+ continue;
+ }
+
+ JsonNode decoded = node.get("decoded");
+ String mismatch = decodedMismatch(parsed, decoded);
+ if (mismatch != null) {
+ System.out.printf(
+ "%-32s FAIL (decoded mismatch: expected %s, actual %s [%s])%n",
+ id, decoded, describe(parsed), mismatch);
+ fail++;
+ failIds.add(id);
+ continue;
+ }
+
+ // Optional canonical check: when a case carries a "canonical" field, the re-serialized type
+ // string must match it. Cases without the field remain decode-only.
+ if (node.hasNonNull("canonical")) {
+ String canonical = node.get("canonical").asText();
+ String actual = parsed.toString();
+ if (!actual.equals(canonical)) {
+ System.out.printf(
+ "%-32s FAIL (Canonical mismatch: expected \"%s\", actual \"%s\")%n",
+ id, canonical, actual);
+ fail++;
+ failIds.add(id);
+ continue;
+ }
+ }
+
+ System.out.printf("%-32s PASS%n", id);
+ pass++;
+ }
+
+ System.out.printf(
+ "%nTOTALS: %d cases | PASS=%d FAIL=%d UNSUPPORTED=%d SKIP=0%n",
+ cases.size(), pass, fail, unsupported);
+ if (!failIds.isEmpty()) {
+ System.out.println("FAIL ids: " + failIds);
+ }
+ if (!unsupportedIds.isEmpty()) {
+ System.out.println("UNSUPPORTED ids: " + unsupportedIds);
+ }
+ if (fail > 0) {
+ System.exit(1);
+ }
+ }
+
+ /**
+ * Parses a type from its Appendix-C JSON: a bare string for primitive and geospatial types (via
+ * {@link Types#fromTypeName(String)}), or an object for a nested type, wrapped as the single field
+ * of a schema and parsed via {@link SchemaParser#fromJson(String)}. The wrapper field uses the
+ * highest non-reserved field id (2147483447) so it never collides with the ids inside a nested
+ * type under test, which Iceberg would otherwise reject as duplicate schema ids.
+ */
+ private static Type parseType(JsonNode input) {
+ if (input.isTextual()) {
+ return Types.fromTypeName(input.asText());
+ }
+ String schemaJson =
+ "{\"type\":\"struct\",\"schema-id\":0,\"fields\":[{\"id\":2147483447,\"name\":\"f\","
+ + "\"required\":true,\"type\":"
+ + input
+ + "}]}";
+ return SchemaParser.fromJson(schemaJson).columns().get(0).type();
+ }
+
+ /** Walks up from {@code start} until it finds a directory containing table-spec/. */
+ private static Path findRepoRoot(Path start) {
+ for (Path dir = start; dir != null; dir = dir.getParent()) {
+ if (Files.isDirectory(dir.resolve("table-spec"))) {
+ return dir;
+ }
+ }
+ return null;
+ }
+
+ /** True if rel is under one of the requested surfaces (all if empty). */
+ private static boolean surfaceMatch(String rel, List surfaces) {
+ if (surfaces.isEmpty()) {
+ return true;
+ }
+ for (String s : surfaces) {
+ String prefix = "table-spec/" + s.replaceAll("/+$", "");
+ if (rel.equals(prefix) || rel.startsWith(prefix + "/")) {
+ return true;
+ }
+ }
+ return false;
+ }
+
+ /** Reads every cases.json under the surface roots, sorted by case id. */
+ private static List loadCases(Path root, List surfaces) throws IOException {
+ ObjectMapper mapper = new ObjectMapper();
+ List cases = new ArrayList<>();
+ for (String surface : SURFACE_ROOTS) {
+ Path base = root.resolve(surface);
+ if (!Files.isDirectory(base)) {
+ continue;
+ }
+ List files;
+ try (Stream walk = Files.walk(base)) {
+ files =
+ walk.filter(Files::isRegularFile)
+ .filter(p -> p.getFileName().toString().equals("cases.json"))
+ .collect(Collectors.toList());
+ }
+ for (Path file : files) {
+ String rel = root.relativize(file).toString().replace(java.io.File.separatorChar, '/');
+ if (!surfaceMatch(rel, surfaces)) {
+ continue;
+ }
+ JsonNode document = mapper.readTree(file.toFile());
+ JsonNode array = document.get("cases");
+ if (array == null || !array.isArray()) {
+ throw new IOException("Missing \"cases\" array in " + file);
+ }
+ for (JsonNode node : array) {
+ cases.add(node);
+ }
+ }
+ }
+ cases.sort(Comparator.comparing(node -> node.get("id").asText()));
+ return cases;
+ }
+
+ /**
+ * Compares a parsed type against the language-neutral {@code decoded} shape. Returns null on
+ * match, or a short reason on mismatch.
+ */
+ private static String decodedMismatch(Type type, JsonNode decoded) {
+ String kind = decoded.get("type").asText();
+ switch (kind) {
+ case "decimal":
+ if (!(type instanceof Types.DecimalType)) {
+ return "Not a decimal";
+ }
+ Types.DecimalType d = (Types.DecimalType) type;
+ return d.precision() == decoded.get("precision").asInt()
+ && d.scale() == decoded.get("scale").asInt()
+ ? null
+ : "Precision/scale";
+ case "fixed":
+ if (!(type instanceof Types.FixedType)) {
+ return "Not a fixed";
+ }
+ return ((Types.FixedType) type).length() == decoded.get("length").asInt() ? null : "Length";
+ case "geometry":
+ if (!(type instanceof Types.GeometryType)) {
+ return "Not a geometry";
+ }
+ return ((Types.GeometryType) type).crs().equals(decoded.get("crs").asText()) ? null : "Crs";
+ case "geography":
+ if (!(type instanceof Types.GeographyType)) {
+ return "Not a geography";
+ }
+ Types.GeographyType g = (Types.GeographyType) type;
+ return g.crs().equals(decoded.get("crs").asText())
+ && g.algorithm().toString().equals(decoded.get("algorithm").asText())
+ ? null
+ : "Crs/algorithm";
+ case "struct":
+ if (!(type instanceof Types.StructType)) {
+ return "Not a struct";
+ }
+ List fields = ((Types.StructType) type).fields();
+ JsonNode fieldsNode = decoded.get("fields");
+ if (fields.size() != fieldsNode.size()) {
+ return "Field count";
+ }
+ for (int i = 0; i < fields.size(); i++) {
+ Types.NestedField f = fields.get(i);
+ JsonNode fn = fieldsNode.get(i);
+ if (f.fieldId() != fn.get("id").asInt()) {
+ return "Field id";
+ }
+ if (!f.name().equals(fn.get("name").asText())) {
+ return "Field name";
+ }
+ if (f.isRequired() != fn.get("required").asBoolean()) {
+ return "Field required";
+ }
+ String sub = decodedMismatch(f.type(), fn.get("type"));
+ if (sub != null) {
+ return "field[" + i + "]: " + sub;
+ }
+ }
+ return null;
+ case "list":
+ if (!(type instanceof Types.ListType)) {
+ return "Not a list";
+ }
+ Types.ListType lt = (Types.ListType) type;
+ if (lt.elementId() != decoded.get("element-id").asInt()) {
+ return "element-id";
+ }
+ if (lt.isElementRequired() != decoded.get("element-required").asBoolean()) {
+ return "element-required";
+ }
+ return decodedMismatch(lt.elementType(), decoded.get("element"));
+ case "map":
+ if (!(type instanceof Types.MapType)) {
+ return "Not a map";
+ }
+ Types.MapType mt = (Types.MapType) type;
+ if (mt.keyId() != decoded.get("key-id").asInt()) {
+ return "key-id";
+ }
+ if (mt.valueId() != decoded.get("value-id").asInt()) {
+ return "value-id";
+ }
+ if (mt.isValueRequired() != decoded.get("value-required").asBoolean()) {
+ return "value-required";
+ }
+ String keyMismatch = decodedMismatch(mt.keyType(), decoded.get("key"));
+ if (keyMismatch != null) {
+ return "key: " + keyMismatch;
+ }
+ return decodedMismatch(mt.valueType(), decoded.get("value"));
+ default:
+ // Primitive: the language-neutral name equals the canonical type string.
+ return type.toString().equals(kind) ? null : "Primitive name";
+ }
+ }
+
+ private static String describe(Type type) {
+ return type.getClass().getSimpleName() + "(\"" + type + "\")";
+ }
+}
diff --git a/runners/python/.gitignore b/runners/python/.gitignore
new file mode 100644
index 0000000..d0ceff3
--- /dev/null
+++ b/runners/python/.gitignore
@@ -0,0 +1,3 @@
+# A local virtualenv and Python bytecode cache created when running the runner.
+/.venv/
+__pycache__/
diff --git a/runners/python/README.md b/runners/python/README.md
new file mode 100644
index 0000000..71f9efe
--- /dev/null
+++ b/runners/python/README.md
@@ -0,0 +1,56 @@
+
+
+# Python conformance runner
+
+Reference runner that checks [pyiceberg](https://github.com/apache/iceberg-python)
+against the type-surface fixtures. It reads every `table-spec/**/cases.json`,
+parses each `input` with pyiceberg's own type parser
+(`pyiceberg.types.IcebergType.model_validate`), and applies the assertion
+contract in `runners/README.md` and `table-spec/types/README.md` (`valid`/reject;
+decoded-shape compare, no bytes, plus byte-exact `canonical` when the case has it). It exits non-zero if any case FAILs.
+
+See `runners/README.md` for the shared runner contract; this README only covers
+how to run the Python one.
+
+## Running it
+
+The runner depends on an installed `pyiceberg`. Install a released build and run
+it:
+
+```sh
+# from runners/python, using uv
+uv venv --python 3.12
+source .venv/bin/activate
+uv pip install pyiceberg
+
+# or with plain pip
+python -m venv .venv && source .venv/bin/activate
+pip install pyiceberg
+
+python runner.py
+```
+
+`python runner.py` discovers the repository root (the directory containing
+`table-spec/`) by walking up, so it works from any working directory. Output is
+one line per case plus totals; the exit code is non-zero on any FAIL.
+
+A type pyiceberg does not model is reported UNSUPPORTED, not FAIL.
+Today `variant` is UNSUPPORTED: pyiceberg has no variant type, so its parser
+rejects the `variant` keyword outright rather than a specific malformed form.
diff --git a/runners/python/runner.py b/runners/python/runner.py
new file mode 100755
index 0000000..9b5fd1f
--- /dev/null
+++ b/runners/python/runner.py
@@ -0,0 +1,256 @@
+#!/usr/bin/env python3
+# Licensed to the Apache Software Foundation (ASF) under one
+# or more contributor license agreements. See the NOTICE file
+# distributed with this work for additional information
+# regarding copyright ownership. The ASF licenses this file
+# to you under the Apache License, Version 2.0 (the
+# "License"); you may not use this file except in compliance
+# with the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an
+# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+# KIND, either express or implied. See the License for the
+# specific language governing permissions and limitations
+# under the License.
+
+"""Reference runner that checks pyiceberg against the type-surface fixtures.
+
+It reads every table-spec/**/cases.json (each a JSON object {"cases": [...]}),
+parses each `input` with pyiceberg's own type parser
+(pyiceberg.types.IcebergType.model_validate), and applies the case contract:
+
+ valid=false -> the parser must raise (PASS); if it parses, FAIL
+ valid=true -> parse ok and the decoded shape == `decoded`;
+ if the parser raises for a type pyiceberg does
+ not model at all, the case is UNSUPPORTED, not a
+ failure
+
+The process exits non-zero if any case FAILs.
+"""
+
+import json
+import os
+import sys
+
+from pyiceberg.types import (
+ DecimalType,
+ FixedType,
+ IcebergType,
+ ListType,
+ MapType,
+ PrimitiveType,
+ StructType,
+)
+
+# Geometry/Geography were added after some pyiceberg releases; tolerate their
+# absence so the runner still loads. Those cases then parse-fail and are handled
+# by the usual UNSUPPORTED/FAIL logic.
+try:
+ from pyiceberg.types import GeographyType, GeometryType
+except ImportError:
+ GeometryType = GeographyType = None
+
+# Directory under the repo root that holds the type-surface cases.
+SURFACE_ROOT = "table-spec"
+
+# Keywords pyiceberg's handle_primitive_type validator recognizes. Mirrors the
+# branches in pyiceberg/types.py: if a type keyword is NOT here, pyiceberg lacks
+# the type entirely (UNSUPPORTED) rather than rejecting a specific form (FAIL).
+EXACT_KEYWORDS = {
+ "boolean", "string", "int", "long", "float", "double",
+ "timestamp", "timestamptz", "timestamp_ns", "timestamptz_ns",
+ "date", "time", "uuid", "binary", "unknown",
+}
+PREFIX_KEYWORDS = ("fixed", "decimal", "geometry", "geography")
+
+
+def find_repo_root(start):
+ """Walk up from start until a directory containing table-spec/ is found."""
+ d = os.path.abspath(start)
+ while True:
+ if os.path.isdir(os.path.join(d, SURFACE_ROOT)):
+ return d
+ parent = os.path.dirname(d)
+ if parent == d:
+ raise RuntimeError(
+ f"could not locate repo root (no {SURFACE_ROOT}/) above {start}"
+ )
+ d = parent
+
+
+def surface_match(rel, surfaces):
+ """True if rel is under one of the requested surfaces (all if empty)."""
+ if not surfaces:
+ return True
+ r = rel.replace(os.sep, "/")
+ for s in surfaces:
+ prefix = f"{SURFACE_ROOT}/{s}".rstrip("/")
+ if r == prefix or r.startswith(prefix + "/"):
+ return True
+ return False
+
+
+def load_cases(root, surfaces=None):
+ """Read every cases.json under the surface root, filtered to surfaces (all if none), sorted by id."""
+ cases = []
+ base = os.path.join(root, SURFACE_ROOT)
+ for dirpath, _dirnames, filenames in os.walk(base):
+ if "cases.json" not in filenames:
+ continue
+ path = os.path.join(dirpath, "cases.json")
+ if not surface_match(os.path.relpath(path, root), surfaces):
+ continue
+ with open(path) as f:
+ doc = json.load(f)
+ for c in doc["cases"]:
+ c["_source"] = os.path.relpath(path, root)
+ cases.append(c)
+ cases.sort(key=lambda c: c["id"])
+ return cases
+
+
+def is_recognized(inp):
+ """True if pyiceberg models this type at all (else UNSUPPORTED)."""
+ if isinstance(inp, dict):
+ return inp.get("type") in ("struct", "list", "map")
+ if inp in EXACT_KEYWORDS:
+ return True
+ return any(inp.startswith(p) for p in PREFIX_KEYWORDS)
+
+
+def decoded_shape(t):
+ """Map a pyiceberg type object to the fixture's language-neutral shape.
+
+ Every value is read from the parsed object `t`; only the tag/key names are
+ literal (the neutral vocabulary shared with the fixtures).
+ """
+ if isinstance(t, DecimalType):
+ return {"type": "decimal", "precision": t.precision, "scale": t.scale}
+ if isinstance(t, FixedType):
+ return {"type": "fixed", "length": len(t)}
+ if GeographyType is not None and isinstance(t, GeographyType):
+ return {"type": "geography", "crs": t.crs, "algorithm": t.algorithm}
+ if GeometryType is not None and isinstance(t, GeometryType):
+ return {"type": "geometry", "crs": t.crs}
+ if isinstance(t, StructType):
+ return {"type": "struct", "fields": [
+ {"id": f.field_id, "name": f.name, "required": f.required,
+ "type": decoded_shape(f.field_type)} for f in t.fields]}
+ if isinstance(t, ListType):
+ return {"type": "list", "element-id": t.element_id,
+ "element-required": t.element_required,
+ "element": decoded_shape(t.element_type)}
+ if isinstance(t, MapType):
+ return {"type": "map", "key-id": t.key_id, "key": decoded_shape(t.key_type),
+ "value-id": t.value_id, "value-required": t.value_required,
+ "value": decoded_shape(t.value_type)}
+ if isinstance(t, PrimitiveType):
+ return {"type": json.loads(t.model_dump_json())}
+ raise TypeError(f"unmapped type object: {t!r}")
+
+
+def main():
+ start = os.getcwd()
+ surfaces = []
+ argv = sys.argv[1:]
+ i = 0
+ while i < len(argv):
+ a = argv[i]
+ if a == "--surface" and i + 1 < len(argv):
+ surfaces.append(argv[i + 1])
+ i += 2
+ continue
+ if a.startswith("--surface="):
+ surfaces.append(a[len("--surface="):])
+ else:
+ start = a
+ i += 1
+ env = os.environ.get("CONFORMANCE_SURFACES", "")
+ surfaces += [s for s in env.replace(",", " ").split() if s]
+ try:
+ root = find_repo_root(start)
+ except RuntimeError as e:
+ print(e, file=sys.stderr)
+ sys.exit(2)
+
+ cases = load_cases(root, surfaces)
+ if not cases:
+ print(f"no cases found under {root}", file=sys.stderr)
+ sys.exit(2)
+
+ results = [] # (id, status, detail)
+ for c in cases:
+ cid = c["id"]
+ inp = c["input"]
+ valid = c["valid"]
+
+ parsed_obj = None
+ err = None
+ try:
+ parsed_obj = IcebergType.model_validate(inp)
+ except Exception as e: # noqa: BLE001 - see valid-branch handling below
+ err = e
+
+ if not valid:
+ if err is not None:
+ results.append((cid, "PASS", f"rejected: {type(err).__name__}"))
+ else:
+ results.append((cid, "FAIL",
+ f"expected reject, parsed as {decoded_shape(parsed_obj)!r}"))
+ continue
+
+ if err is not None:
+ if not is_recognized(inp):
+ results.append((cid, "UNSUPPORTED",
+ f"pyiceberg lacks type ({type(err).__name__})"))
+ else:
+ results.append((cid, "FAIL",
+ f"expected accept, parse raised {type(err).__name__}: {err}"))
+ continue
+
+ got = decoded_shape(parsed_obj)
+ want = c["decoded"]
+ if got != want:
+ results.append((cid, "FAIL", f"decoded mismatch: expected {want}, actual {got}"))
+ continue
+
+ if "canonical" in c:
+ want_canon = c["canonical"]
+ got_canon = json.loads(parsed_obj.model_dump_json())
+ if got_canon != want_canon:
+ results.append((cid, "FAIL",
+ f"canonical mismatch: expected {want_canon!r}, actual {got_canon!r}"))
+ continue
+
+ results.append((cid, "PASS", ""))
+
+ width = max(len(cid) for cid, _, _ in results)
+ for cid, status, detail in results:
+ line = f"{cid.ljust(width)} {status}"
+ if detail and status != "PASS":
+ line += f" ({detail})"
+ print(line)
+
+ counts = {"PASS": 0, "FAIL": 0, "UNSUPPORTED": 0, "SKIP": 0}
+ for _, status, _ in results:
+ counts[status] = counts.get(status, 0) + 1
+
+ print(f"\nTOTALS: {len(results)} cases | "
+ f"PASS={counts['PASS']} FAIL={counts['FAIL']} "
+ f"UNSUPPORTED={counts['UNSUPPORTED']} SKIP=0")
+
+ fail_ids = [cid for cid, s, _ in results if s == "FAIL"]
+ unsupported_ids = [cid for cid, s, _ in results if s == "UNSUPPORTED"]
+ if fail_ids:
+ print(f"FAIL ids: {fail_ids}")
+ if unsupported_ids:
+ print(f"UNSUPPORTED ids: {unsupported_ids}")
+
+ sys.exit(1 if fail_ids else 0)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/runners/rust/.gitignore b/runners/rust/.gitignore
new file mode 100644
index 0000000..0e280cc
--- /dev/null
+++ b/runners/rust/.gitignore
@@ -0,0 +1,7 @@
+# The implementation under test is checked out (CI) or symlinked (local) here.
+# No trailing slash, so a local symlink is ignored as well as a checked-out dir.
+/iceberg-rust
+/target/
+# Resolved against the ephemeral ./iceberg-rust above, so it is not committed;
+# CI regenerates it against the freshly checked-out implementation.
+/Cargo.lock
diff --git a/runners/rust/Cargo.toml b/runners/rust/Cargo.toml
new file mode 100644
index 0000000..c18ddc1
--- /dev/null
+++ b/runners/rust/Cargo.toml
@@ -0,0 +1,33 @@
+# Licensed to the Apache Software Foundation (ASF) under one
+# or more contributor license agreements. See the NOTICE file
+# distributed with this work for additional information
+# regarding copyright ownership. The ASF licenses this file
+# to you under the Apache License, Version 2.0 (the
+# "License"); you may not use this file except in compliance
+# with the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an
+# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+# KIND, either express or implied. See the License for the
+# specific language governing permissions and limitations
+# under the License.
+
+[package]
+name = "conformance-rust"
+version = "0.0.0"
+edition = "2021"
+publish = false
+
+# Depends on the published iceberg crate (latest release). default-features =
+# false keeps the type parser without pulling storage backends. dev/run-local.sh
+# overrides this with a [patch.crates-io] onto a clone of apache/main.
+[dependencies]
+iceberg = { version = "0.10.1", default-features = false }
+serde_json = "1"
+
+[[bin]]
+name = "conformance-rust"
+path = "src/main.rs"
diff --git a/runners/rust/README.md b/runners/rust/README.md
new file mode 100644
index 0000000..65111e9
--- /dev/null
+++ b/runners/rust/README.md
@@ -0,0 +1,60 @@
+
+
+# Rust conformance runner
+
+Reference runner that checks [iceberg-rust](https://github.com/apache/iceberg-rust)
+against the type-surface fixtures. It reads every `table-spec/**/cases.json`,
+deserializes each `input` into iceberg-rust's own `Type` with `serde_json`, and
+applies the assertion contract in `table-spec/types/README.md` (`valid`/reject;
+decoded-shape compare, no bytes, plus byte-exact `canonical` when the case has it).
+It exits non-zero if any case FAILs. See [`../README.md`](../README.md) for the
+shared contract that every language runner follows.
+
+## Running it
+
+The runner depends on the `iceberg` crate pinned in `Cargo.toml`; `cargo run`
+resolves it from crates.io. CI tests `apache/main` via a git dependency:
+`cargo add iceberg --git https://github.com/apache/iceberg-rust --branch main --no-default-features`.
+
+```sh
+# from runners/rust (latest release)
+cargo run
+```
+
+`cargo run` discovers the repository root (the directory containing `table-spec/`)
+by walking up, so it works from any working directory. Output is one line per
+case plus totals; the exit code is non-zero on any FAIL.
+
+## Expected outcomes
+
+A type iceberg-rust's `Type` enum does not model is reported UNSUPPORTED, not
+FAIL. The shipped fixtures exercise this for `unknown`, `geometry`, and
+`geography` (iceberg-rust has `struct` / `list` / `map` / `variant` and the v3
+`timestamp_ns` / `timestamptz_ns`, but not `unknown` or the geospatial types).
+
+Two cases FAIL today, and these are real iceberg-rust divergences from the
+spec-derived expectation, not runner bugs:
+
+- `decimal-precision-over-38` - iceberg-rust accepts `decimal(39, 0)`; the spec
+ caps precision at 38.
+- `fixed-unterminated` - iceberg-rust accepts the unterminated `fixed[16`.
+
+Both should stay red until iceberg-rust tightens its parser (or the community
+decides otherwise).
diff --git a/runners/rust/rust-toolchain.toml b/runners/rust/rust-toolchain.toml
new file mode 100644
index 0000000..1518fcd
--- /dev/null
+++ b/runners/rust/rust-toolchain.toml
@@ -0,0 +1,6 @@
+# Pin the runner-dir toolchain to current stable. iceberg-rust main tracks a
+# recent MSRV (it required rustc 1.95 as of iceberg 0.10.0), so the runner must
+# build with a stable new enough for whatever main currently requires. rustup
+# auto-installs this channel on first cargo invocation in this directory.
+[toolchain]
+channel = "stable"
diff --git a/runners/rust/src/main.rs b/runners/rust/src/main.rs
new file mode 100644
index 0000000..abf6368
--- /dev/null
+++ b/runners/rust/src/main.rs
@@ -0,0 +1,329 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements. See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership. The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License. You may obtain a copy of the License at
+//
+// http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied. See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! Reference runner that checks iceberg-rust against the type-surface conformance
+//! fixtures. It reads every table-spec/**/cases.json (a JSON object with a top
+//! level `cases` array), deserializes each `input` type string into iceberg-rust's
+//! own `Type` via serde_json, and applies the surface's assertion contract:
+//!
+//! valid=false -> deserialization must return an error (PASS); Ok is FAIL
+//! valid=true -> deserialization must succeed and the decoded shape
+//! must equal `decoded`
+//!
+//! UNSUPPORTED is decided post-parse: if a valid case fails to deserialize AND the
+//! expected `decoded` type is one iceberg-rust's `Type` enum cannot represent
+//! (unknown / geometry / geography), the case is reported UNSUPPORTED, not FAIL.
+//! The process exits 0 (no FAIL), 1 (any FAIL), or 2 (setup / fixture-load error).
+
+use std::path::{Path, PathBuf};
+
+use iceberg::spec::{PrimitiveType, Type};
+use serde_json::{json, Value};
+
+// The directory under the repo root that holds the fixture cases.
+const SURFACE_ROOT: &str = "table-spec";
+
+// Type names iceberg-rust's `Type` enum does not model. When a valid case fails
+// to parse and its expected `decoded.type` is one of these, it is reported
+// UNSUPPORTED (a skip-list candidate), not FAIL. iceberg-rust models struct /
+// list / map / variant, but no unknown / geometry / geography.
+fn is_unsupported_type(name: &str) -> bool {
+ matches!(name, "unknown" | "geometry" | "geography")
+}
+
+// find_repo_root walks up from `start` until it finds a directory containing
+// table-spec/, so the runner works from any working directory.
+fn find_repo_root(start: &Path) -> Option {
+ let mut dir = start.to_path_buf();
+ loop {
+ if dir.join(SURFACE_ROOT).is_dir() {
+ return Some(dir);
+ }
+ if !dir.pop() {
+ return None;
+ }
+ }
+}
+
+// collect_case_files recursively finds every cases.json under `dir`.
+fn collect_case_files(dir: &Path, out: &mut Vec) -> std::io::Result<()> {
+ for entry in std::fs::read_dir(dir)? {
+ let path = entry?.path();
+ if path.is_dir() {
+ collect_case_files(&path, out)?;
+ } else if path.file_name().and_then(|n| n.to_str()) == Some("cases.json") {
+ out.push(path);
+ }
+ }
+ Ok(())
+}
+
+// surface_match reports whether rel is under one of the requested surfaces (all if empty).
+fn surface_match(rel: &str, surfaces: &[String]) -> bool {
+ if surfaces.is_empty() {
+ return true;
+ }
+ let r = rel.replace('\\', "/");
+ surfaces.iter().any(|s| {
+ let prefix = format!("{SURFACE_ROOT}/{}", s.trim_end_matches('/'));
+ r == prefix || r.starts_with(&format!("{prefix}/"))
+ })
+}
+
+// Map a decoded iceberg-rust Type to the surface's language-neutral `decoded` shape.
+fn decoded_to_shape(ty: &Type) -> Value {
+ match ty {
+ Type::Primitive(p) => match p {
+ PrimitiveType::Boolean => json!({"type": "boolean"}),
+ PrimitiveType::Int => json!({"type": "int"}),
+ PrimitiveType::Long => json!({"type": "long"}),
+ PrimitiveType::Float => json!({"type": "float"}),
+ PrimitiveType::Double => json!({"type": "double"}),
+ PrimitiveType::Decimal { precision, scale } => {
+ json!({"type": "decimal", "precision": precision, "scale": scale})
+ }
+ PrimitiveType::Date => json!({"type": "date"}),
+ PrimitiveType::Time => json!({"type": "time"}),
+ PrimitiveType::Timestamp => json!({"type": "timestamp"}),
+ PrimitiveType::Timestamptz => json!({"type": "timestamptz"}),
+ PrimitiveType::TimestampNs => json!({"type": "timestamp_ns"}),
+ PrimitiveType::TimestamptzNs => json!({"type": "timestamptz_ns"}),
+ PrimitiveType::String => json!({"type": "string"}),
+ PrimitiveType::Uuid => json!({"type": "uuid"}),
+ PrimitiveType::Fixed(l) => json!({"type": "fixed", "length": l}),
+ PrimitiveType::Binary => json!({"type": "binary"}),
+ },
+ Type::Struct(s) => {
+ let fields: Vec = s
+ .fields()
+ .iter()
+ .map(|f| {
+ json!({
+ "id": f.id,
+ "name": f.name,
+ "required": f.required,
+ "type": decoded_to_shape(f.field_type.as_ref()),
+ })
+ })
+ .collect();
+ json!({"type": "struct", "fields": fields})
+ }
+ Type::List(l) => json!({
+ "type": "list",
+ "element-id": l.element_field.id,
+ "element-required": l.element_field.required,
+ "element": decoded_to_shape(l.element_field.field_type.as_ref()),
+ }),
+ Type::Map(m) => json!({
+ "type": "map",
+ "key-id": m.key_field.id,
+ "key": decoded_to_shape(m.key_field.field_type.as_ref()),
+ "value-id": m.value_field.id,
+ "value-required": m.value_field.required,
+ "value": decoded_to_shape(m.value_field.field_type.as_ref()),
+ }),
+ Type::Variant(_) => json!({"type": "variant"}),
+ }
+}
+
+// serialized_wire_form re-serializes a parsed Type back to its language-neutral
+// wire string via serde (e.g. "fixed[16]"). iceberg-rust serializes a Type to a
+// bare JSON string for primitives, so we take serde_json::to_value and read it as
+// a str; Display is NOT used (it emits the wrong "fixed(16)" form).
+fn serialized_wire_form(ty: &Type) -> Option {
+ serde_json::to_value(ty)
+ .ok()?
+ .as_str()
+ .map(|s| s.to_string())
+}
+
+// load_cases reads every cases.json under the surface root, flattening the
+// top level `cases` array of each file and tagging cases with their source
+// path for diagnostics. Returns Err on any fixture-load / parse-of-file error.
+fn load_cases(root: &Path, surfaces: &[String]) -> Result, String> {
+ let mut files = Vec::new();
+ collect_case_files(&root.join(SURFACE_ROOT), &mut files)
+ .map_err(|e| format!("walk {SURFACE_ROOT}: {e}"))?;
+ files.sort();
+
+ let mut cases: Vec<(String, Value)> = Vec::new();
+ for file in &files {
+ let rel = file
+ .strip_prefix(root)
+ .unwrap_or(file)
+ .display()
+ .to_string();
+ if !surface_match(&rel, surfaces) {
+ continue;
+ }
+ let raw = std::fs::read_to_string(file).map_err(|e| format!("{rel}: read: {e}"))?;
+ let doc: Value = serde_json::from_str(&raw).map_err(|e| format!("{rel}: parse: {e}"))?;
+ let arr = doc
+ .get("cases")
+ .and_then(|c| c.as_array())
+ .ok_or_else(|| format!("{rel}: missing top-level \"cases\" array"))?;
+ for case in arr {
+ cases.push((rel.clone(), case.clone()));
+ }
+ }
+ Ok(cases)
+}
+
+fn run() -> Result {
+ let cwd = std::env::current_dir().map_err(|e| format!("getcwd: {e}"))?;
+ let mut start = cwd;
+ let mut surfaces: Vec = Vec::new();
+ let args: Vec = std::env::args().skip(1).collect();
+ let mut i = 0;
+ while i < args.len() {
+ let a = &args[i];
+ if a == "--surface" {
+ if i + 1 < args.len() {
+ surfaces.push(args[i + 1].clone());
+ i += 1;
+ }
+ } else if let Some(v) = a.strip_prefix("--surface=") {
+ surfaces.push(v.to_string());
+ } else {
+ start = PathBuf::from(a);
+ }
+ i += 1;
+ }
+ if let Ok(env) = std::env::var("CONFORMANCE_SURFACES") {
+ surfaces.extend(
+ env.split(|c| c == ',' || c == ' ')
+ .filter(|s| !s.is_empty())
+ .map(String::from),
+ );
+ }
+ let root = find_repo_root(&start).ok_or_else(|| {
+ format!("could not locate repo root (no {SURFACE_ROOT}/) above {start:?}")
+ })?;
+
+ let mut cases = load_cases(&root, &surfaces)?;
+ cases.sort_by(|a, b| a.1["id"].as_str().cmp(&b.1["id"].as_str()));
+
+ if cases.is_empty() {
+ return Err(format!("no cases found under {}", root.display()));
+ }
+
+ let mut pass = 0usize;
+ let mut fail = 0usize;
+ let mut unsupported = 0usize;
+ let mut fail_ids: Vec = Vec::new();
+ let mut unsupported_ids: Vec = Vec::new();
+
+ for (src, case) in &cases {
+ let id = case["id"]
+ .as_str()
+ .ok_or_else(|| format!("{src}: case missing string \"id\""))?
+ .to_string();
+ let valid = case["valid"]
+ .as_bool()
+ .ok_or_else(|| format!("{src}: case {id}: missing bool \"valid\""))?;
+ let input = case["input"].clone();
+
+ let parsed: Result = serde_json::from_value::(input);
+
+ if !valid {
+ match parsed {
+ Err(e) => {
+ println!("{id:<32} PASS (rejected: {e})");
+ pass += 1;
+ }
+ Ok(ty) => {
+ println!("{id:<32} FAIL (expected reject, parsed as {ty:?})");
+ fail += 1;
+ fail_ids.push(id);
+ }
+ }
+ continue;
+ }
+
+ // valid == true: parse must succeed and the decoded shape must match.
+ let expected = &case["decoded"];
+ match parsed {
+ Ok(ty) => {
+ let actual = decoded_to_shape(&ty);
+ if &actual == expected {
+ // Optional canonical (serialize) check: when the case carries a
+ // "canonical" wire string, re-serializing the parsed Type must
+ // reproduce it. Absent the field, decode-only PASS as before.
+ if let Some(expected_canon) = case["canonical"].as_str() {
+ let actual_canon = serialized_wire_form(&ty);
+ if actual_canon.as_deref() == Some(expected_canon) {
+ println!("{id:<32} PASS");
+ pass += 1;
+ } else {
+ println!(
+ "{id:<32} FAIL (canonical mismatch: expected {expected_canon:?}, actual {actual_canon:?})"
+ );
+ fail += 1;
+ fail_ids.push(id);
+ }
+ } else {
+ println!("{id:<32} PASS");
+ pass += 1;
+ }
+ } else {
+ println!(
+ "{id:<32} FAIL (decoded mismatch: expected {expected}, actual {actual})"
+ );
+ fail += 1;
+ fail_ids.push(id);
+ }
+ }
+ Err(e) => {
+ // A parse failure on a valid case is UNSUPPORTED only when the
+ // expected type is one iceberg-rust's Type enum cannot model.
+ let name = expected["type"].as_str().unwrap_or("");
+ if is_unsupported_type(name) {
+ println!("{id:<32} UNSUPPORTED (no iceberg-rust Type for {name:?})");
+ unsupported += 1;
+ unsupported_ids.push(id);
+ } else {
+ println!("{id:<32} FAIL (expected accept, parse error: {e})");
+ fail += 1;
+ fail_ids.push(id);
+ }
+ }
+ }
+ }
+
+ println!(
+ "\nTOTALS: {} cases | PASS={pass} FAIL={fail} UNSUPPORTED={unsupported} SKIP=0",
+ cases.len()
+ );
+ if !fail_ids.is_empty() {
+ println!("FAIL ids: {fail_ids:?}");
+ }
+ if !unsupported_ids.is_empty() {
+ println!("UNSUPPORTED ids: {unsupported_ids:?}");
+ }
+
+ Ok(if fail > 0 { 1 } else { 0 })
+}
+
+fn main() {
+ match run() {
+ Ok(code) => std::process::exit(code),
+ Err(e) => {
+ eprintln!("{e}");
+ std::process::exit(2);
+ }
+ }
+}
diff --git a/table-spec/manifest.json b/table-spec/manifest.json
new file mode 100644
index 0000000..3b67fab
--- /dev/null
+++ b/table-spec/manifest.json
@@ -0,0 +1,10 @@
+{
+ "surfaces": [
+ {
+ "name": "types",
+ "path": "table-spec/types",
+ "readme": "table-spec/types/README.md",
+ "subdirs": ["primitive", "variant", "nested", "geospatial"]
+ }
+ ]
+}
diff --git a/table-spec/types/README.md b/table-spec/types/README.md
new file mode 100644
index 0000000..7936ca3
--- /dev/null
+++ b/table-spec/types/README.md
@@ -0,0 +1,110 @@
+
+
+# Type decoding
+
+Parsing a type string produces the same type in every implementation. This
+surface pins each type the spec defines and the parse rules that attach to it.
+
+## Assertion
+
+```
+parse(input) == decoded
+```
+
+`input` is a type string, or a JSON object for a nested type; `decoded` is the
+language-neutral shape below. Bytes are not compared - each implementation maps
+its own type object to `decoded`, so the comparison does not depend on one
+language's representation.
+
+- `valid: true` - the parser succeeds and the decoded type equals `decoded`. A
+ type an implementation does not model is UNSUPPORTED, not a failure. If the case
+ also carries `canonical`, re-serializing the parsed type must equal it byte for
+ byte (the write direction).
+- `valid: false` - the parser must reject `input`. A rejection passes; a
+ successful parse fails.
+
+`canonical` is present only where the spec pins one spelling. `decimal` has two
+blessed forms (`decimal(9,2)` and `decimal(9, 2)`), so its cases have no
+`canonical` and are compared by `decoded` alone.
+
+## Scope
+
+Each type in isolation, per the Primitive Types table and Appendix C. The full
+schema document (schema-id, identifier-field-ids, field ordering) is the `schema`
+surface. Whether a type is legal at a given format version is not decided here,
+because a type in isolation carries no version.
+
+## Inputs
+
+- `primitive/` - every v1/v2 primitive (`boolean`, `int`, `long`, `float`,
+ `double`, `date`, `time`, `timestamp`, `timestamptz`, `string`, `uuid`,
+ `binary`, `decimal`, `fixed`) plus the v3 additions `timestamp_ns`,
+ `timestamptz_ns`, `unknown`.
+- `variant/` - `variant` (v3).
+- `nested/` - `struct`, `list`, `map`, including nesting (a `struct` field whose
+ type is a `list`).
+- `geospatial/` - `geometry` and `geography` (v3), with explicit and default CRS.
+
+## Case format
+
+One `cases.json` per directory, a JSON object with a `cases` array:
+
+| field | meaning |
+| --- | --- |
+| `id` | unique case id |
+| `valid` | `true` if the parser must accept `input`, `false` if it must reject it |
+| `input` | the type string (`"decimal(9,2)"`), or a JSON object for a nested type |
+| `decoded` | the decoded shape; present only when `valid` is `true` |
+| `canonical` | the exact re-serialized string; present only where the spec pins one spelling |
+| `clause` | the spec rule this case pins |
+| `spec_ref` | anchor into `format/spec.md` |
+
+`decoded` is language-neutral:
+
+- primitive: `{"type": ""}`, e.g. `{"type": "int"}`
+- `decimal`: `{"type": "decimal", "precision": P, "scale": S}`
+- `fixed`: `{"type": "fixed", "length": L}`
+- `geometry`: `{"type": "geometry", "crs": C}`; `geography` adds `"algorithm": A`
+- `struct`: `{"type": "struct", "fields": [{"id", "name", "required", "type"}, ...]}`
+- `list`: `{"type": "list", "element-id", "element-required", "element"}`
+- `map`: `{"type": "map", "key-id", "key", "value-id", "value-required", "value"}`
+
+A nested type's child `type` values are the same shape, recursively.
+
+## Provenance
+
+`input` and `decoded` are derived from `format/spec.md` (the Primitive Types
+table and Appendix C), cross-checked against Apache Iceberg Java. There is no
+binary artifact; Java is a cross-check, not the source of the inputs.
+
+## Left out on purpose
+
+Inputs the spec neither permits nor forbids, so no answer can be spec-derived:
+
+- `scale > precision`, e.g. `decimal(5, 10)`.
+- lower bounds, e.g. `decimal(0, 0)` / `fixed[0]`.
+- internal whitespace around every parameter, e.g. `decimal( 9 , 2 )`. The one
+ spaced case we do ship, `decimal(9, 2)`, is a *recommended* (SHOULD) accept,
+ not a hard requirement: the spec says readers *should*, not *must*, accept
+ optional whitespace, so an implementation that rejects it is still conformant.
+- keyword case, e.g. `DECIMAL(9,2)`.
+- geospatial CRS *quoting*, e.g. `geometry('OGC:CRS84')`. The spec's canonical
+ form is unquoted (`geometry(OGC:CRS84)`), which the shipped cases use; whether
+ the quoted form is also accepted is unpinned.
diff --git a/table-spec/types/geospatial/cases.json b/table-spec/types/geospatial/cases.json
new file mode 100644
index 0000000..5d59090
--- /dev/null
+++ b/table-spec/types/geospatial/cases.json
@@ -0,0 +1,52 @@
+{
+ "cases": [
+ {
+ "id": "geometry-crs84",
+ "valid": true,
+ "input": "geometry(OGC:CRS84)",
+ "decoded": {"type": "geometry", "crs": "OGC:CRS84"},
+ "clause": "geometry(C) with explicit CRS; the canonical serialized form is unquoted \"geometry()\"",
+ "spec_ref": "format/spec.md#appendix-c-json-serialization"
+ },
+ {
+ "id": "geometry-srid",
+ "valid": true,
+ "input": "geometry(srid:4326)",
+ "decoded": {"type": "geometry", "crs": "srid:4326"},
+ "clause": "geometry(C) example from Appendix C is the unquoted \"geometry(srid:4326)\"",
+ "spec_ref": "format/spec.md#appendix-c-json-serialization"
+ },
+ {
+ "id": "geometry-default-crs",
+ "valid": true,
+ "input": "geometry",
+ "decoded": {"type": "geometry", "crs": "OGC:CRS84"},
+ "clause": "geometry(C): if C is not specified, C is OGC:CRS84",
+ "spec_ref": "format/spec.md#primitive-types"
+ },
+ {
+ "id": "geography-crs84-spherical",
+ "valid": true,
+ "input": "geography(OGC:CRS84, spherical)",
+ "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"},
+ "clause": "geography(C, A); the canonical serialized form is unquoted \"geography(, )\"",
+ "spec_ref": "format/spec.md#appendix-c-json-serialization"
+ },
+ {
+ "id": "geography-default",
+ "valid": true,
+ "input": "geography",
+ "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "spherical"},
+ "clause": "geography(C, A): if not specified, C is OGC:CRS84 and A is spherical",
+ "spec_ref": "format/spec.md#primitive-types"
+ },
+ {
+ "id": "geography-vincenty",
+ "valid": true,
+ "input": "geography(OGC:CRS84, vincenty)",
+ "decoded": {"type": "geography", "crs": "OGC:CRS84", "algorithm": "vincenty"},
+ "clause": "geography edge-interpolation algorithm A is one of spherical, vincenty, thomas, andoyer, karney",
+ "spec_ref": "format/spec.md#appendix-g-geospatial-notes"
+ }
+ ]
+}
diff --git a/table-spec/types/nested/cases.json b/table-spec/types/nested/cases.json
new file mode 100644
index 0000000..4645fc5
--- /dev/null
+++ b/table-spec/types/nested/cases.json
@@ -0,0 +1,67 @@
+{
+ "cases": [
+ {
+ "id": "struct-single-field",
+ "valid": true,
+ "input": {"type": "struct", "fields": [{"id": 1, "name": "a", "required": true, "type": "int"}]},
+ "decoded": {"type": "struct", "fields": [{"id": 1, "name": "a", "required": true, "type": {"type": "int"}}]},
+ "clause": "struct is a tuple of typed fields; each field has an integer id, a name, a required flag, and a type",
+ "spec_ref": "format/spec.md#nested-types"
+ },
+ {
+ "id": "struct-empty",
+ "valid": true,
+ "input": {"type": "struct", "fields": []},
+ "decoded": {"type": "struct", "fields": []},
+ "clause": "a struct's fields array may be empty (Appendix C struct serialization)",
+ "spec_ref": "format/spec.md#appendix-c-json-serialization"
+ },
+ {
+ "id": "struct-optional-field",
+ "valid": true,
+ "input": {"type": "struct", "fields": [{"id": 1, "name": "a", "required": false, "type": "string"}]},
+ "decoded": {"type": "struct", "fields": [{"id": 1, "name": "a", "required": false, "type": {"type": "string"}}]},
+ "clause": "each struct field can be optional or required (required=false permits null values)",
+ "spec_ref": "format/spec.md#nested-types"
+ },
+ {
+ "id": "list-required-element",
+ "valid": true,
+ "input": {"type": "list", "element-id": 2, "element-required": true, "element": "string"},
+ "decoded": {"type": "list", "element-id": 2, "element-required": true, "element": {"type": "string"}},
+ "clause": "a list has an element type; the element field has an integer id and a required flag",
+ "spec_ref": "format/spec.md#appendix-c-json-serialization"
+ },
+ {
+ "id": "list-optional-element",
+ "valid": true,
+ "input": {"type": "list", "element-id": 2, "element-required": false, "element": "long"},
+ "decoded": {"type": "list", "element-id": 2, "element-required": false, "element": {"type": "long"}},
+ "clause": "list elements can be optional or required",
+ "spec_ref": "format/spec.md#nested-types"
+ },
+ {
+ "id": "map-string-double",
+ "valid": true,
+ "input": {"type": "map", "key-id": 3, "key": "string", "value-id": 4, "value-required": false, "value": "double"},
+ "decoded": {"type": "map", "key-id": 3, "key": {"type": "string"}, "value-id": 4, "value-required": false, "value": {"type": "double"}},
+ "clause": "a map has a key type and a value type; keys are required, values may be optional or required",
+ "spec_ref": "format/spec.md#appendix-c-json-serialization"
+ },
+ {
+ "id": "struct-nested-list",
+ "valid": true,
+ "input": {"type": "struct", "fields": [{"id": 1, "name": "tags", "required": true, "type": {"type": "list", "element-id": 2, "element-required": true, "element": "string"}}]},
+ "decoded": {"type": "struct", "fields": [{"id": 1, "name": "tags", "required": true, "type": {"type": "list", "element-id": 2, "element-required": true, "element": {"type": "string"}}}]},
+ "clause": "fields may be any type, including nested types (a struct field whose type is a list)",
+ "spec_ref": "format/spec.md#nested-types"
+ },
+ {
+ "id": "struct-field-missing-id",
+ "valid": false,
+ "input": {"type": "struct", "fields": [{"name": "a", "required": true, "type": "int"}]},
+ "clause": "each field in a struct has an integer id; a field without an id is invalid",
+ "spec_ref": "format/spec.md#nested-types"
+ }
+ ]
+}
diff --git a/table-spec/types/primitive/cases.json b/table-spec/types/primitive/cases.json
new file mode 100644
index 0000000..b41a30b
--- /dev/null
+++ b/table-spec/types/primitive/cases.json
@@ -0,0 +1,33 @@
+{
+ "cases": [
+ { "id": "boolean", "valid": true, "input": "boolean", "decoded": { "type": "boolean" }, "canonical": "boolean", "clause": "Primitive Types: boolean; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "int", "valid": true, "input": "int", "decoded": { "type": "int" }, "canonical": "int", "clause": "Primitive Types: int; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "long", "valid": true, "input": "long", "decoded": { "type": "long" }, "canonical": "long", "clause": "Primitive Types: long; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "float", "valid": true, "input": "float", "decoded": { "type": "float" }, "canonical": "float", "clause": "Primitive Types: float; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "double", "valid": true, "input": "double", "decoded": { "type": "double" }, "canonical": "double", "clause": "Primitive Types: double; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "date", "valid": true, "input": "date", "decoded": { "type": "date" }, "canonical": "date", "clause": "Primitive Types: date; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "time", "valid": true, "input": "time", "decoded": { "type": "time" }, "canonical": "time", "clause": "Primitive Types: time; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "timestamp", "valid": true, "input": "timestamp", "decoded": { "type": "timestamp" }, "canonical": "timestamp", "clause": "Primitive Types: timestamp; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "timestamptz", "valid": true, "input": "timestamptz", "decoded": { "type": "timestamptz" }, "canonical": "timestamptz", "clause": "Primitive Types: timestamptz; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "timestamp_ns", "valid": true, "input": "timestamp_ns", "decoded": { "type": "timestamp_ns" }, "canonical": "timestamp_ns", "clause": "Primitive Types: timestamp_ns added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#primitive-types" },
+ { "id": "timestamptz_ns", "valid": true, "input": "timestamptz_ns", "decoded": { "type": "timestamptz_ns" }, "canonical": "timestamptz_ns", "clause": "Primitive Types: timestamptz_ns added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#primitive-types" },
+ { "id": "string", "valid": true, "input": "string", "decoded": { "type": "string" }, "canonical": "string", "clause": "Primitive Types: string; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "uuid", "valid": true, "input": "uuid", "decoded": { "type": "uuid" }, "canonical": "uuid", "clause": "Primitive Types: uuid; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "binary", "valid": true, "input": "binary", "decoded": { "type": "binary" }, "canonical": "binary", "clause": "Primitive Types: binary; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "unknown", "valid": true, "input": "unknown", "decoded": { "type": "unknown" }, "canonical": "unknown", "clause": "Primitive Types: unknown added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#primitive-types" },
+ { "id": "fixed-1", "valid": true, "input": "fixed[1]", "decoded": { "type": "fixed", "length": 1 }, "canonical": "fixed[1]", "clause": "Appendix C: fixed canonical string is fixed[]", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "fixed-16", "valid": true, "input": "fixed[16]", "decoded": { "type": "fixed", "length": 16 }, "canonical": "fixed[16]", "clause": "Appendix C: fixed canonical string is fixed[]", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "decimal-9-2", "valid": true, "input": "decimal(9,2)", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "clause": "Appendix C: both decimal(9,2) and decimal(9, 2) are canonical, so no byte-exact form is pinned", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "decimal-9-2-spaced", "valid": true, "input": "decimal(9, 2)", "decoded": { "type": "decimal", "precision": 9, "scale": 2 }, "clause": "Appendix C: the spaced decimal(9, 2) form parses to the same decimal", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "decimal-38-0", "valid": true, "input": "decimal(38,0)", "decoded": { "type": "decimal", "precision": 38, "scale": 0 }, "clause": "Primitive Types: decimal precision must be 38 or less (38 is the maximum)", "spec_ref": "format/spec.md#primitive-types" },
+ { "id": "decimal-precision-over-38", "valid": false, "input": "decimal(39,0)", "clause": "Primitive Types: decimal precision must be 38 or less", "spec_ref": "format/spec.md#primitive-types" },
+ { "id": "decimal-missing-scale", "valid": false, "input": "decimal(9)", "clause": "Appendix C: decimal is written decimal(P,S); scale is required", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "decimal-empty-params", "valid": false, "input": "decimal()", "clause": "Appendix C: decimal requires precision and scale", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "decimal-non-numeric", "valid": false, "input": "decimal(a,b)", "clause": "Appendix C: decimal precision and scale are integers", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "fixed-empty-length", "valid": false, "input": "fixed[]", "clause": "Appendix C: fixed is written fixed[]; length is required", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "fixed-unterminated", "valid": false, "input": "fixed[16", "clause": "Appendix C: fixed[] must be closed with a bracket", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "fixed-non-numeric", "valid": false, "input": "fixed[abc]", "clause": "Appendix C: fixed length is an integer", "spec_ref": "format/spec.md#appendix-c-json-serialization" },
+ { "id": "empty-type", "valid": false, "input": "", "clause": "Primitive Types: the empty string is not a type name", "spec_ref": "format/spec.md#primitive-types" },
+ { "id": "unknown-type-name", "valid": false, "input": "notatype", "clause": "Primitive Types: only the listed type names are valid", "spec_ref": "format/spec.md#primitive-types" }
+ ]
+}
diff --git a/table-spec/types/variant/cases.json b/table-spec/types/variant/cases.json
new file mode 100644
index 0000000..ace69ac
--- /dev/null
+++ b/table-spec/types/variant/cases.json
@@ -0,0 +1,5 @@
+{
+ "cases": [
+ { "id": "variant", "valid": true, "input": "variant", "decoded": { "type": "variant" }, "canonical": "variant", "clause": "Semi-structured Types: variant added in v3; Appendix C canonical string", "spec_ref": "format/spec.md#appendix-c-json-serialization" }
+ ]
+}