From dc2c7b94bb5da1cc6807cab57062133a821ed539 Mon Sep 17 00:00:00 2001
From: Divjot Arora
Date: Wed, 5 Aug 2026 18:33:40 +0000
Subject: [PATCH] Inline parquet.thrift
---
dev/check-parquet-thrift-release.sh | 84 +
dev/parquet-thrift-lib.sh | 63 +
dev/prepare-release.sh | 3 +
dev/update-parquet-thrift.sh | 165 ++
parquet-format-structures/README.md | 95 ++
parquet-format-structures/pom.xml | 32 +-
.../src/main/thrift/parquet-format.version | 4 +
.../src/main/thrift/parquet.thrift | 1443 +++++++++++++++++
pom.xml | 2 +-
9 files changed, 1859 insertions(+), 32 deletions(-)
create mode 100755 dev/check-parquet-thrift-release.sh
create mode 100755 dev/parquet-thrift-lib.sh
create mode 100755 dev/update-parquet-thrift.sh
create mode 100644 parquet-format-structures/README.md
create mode 100644 parquet-format-structures/src/main/thrift/parquet-format.version
create mode 100644 parquet-format-structures/src/main/thrift/parquet.thrift
diff --git a/dev/check-parquet-thrift-release.sh b/dev/check-parquet-thrift-release.sh
new file mode 100755
index 0000000000..2a99cc15dc
--- /dev/null
+++ b/dev/check-parquet-thrift-release.sh
@@ -0,0 +1,84 @@
+#!/usr/bin/env bash
+#
+# Licensed to the Apache Software Foundation (ASF) under one
+# or more contributor license agreements. See the NOTICE file
+# distributed with this work for additional information
+# regarding copyright ownership. The ASF licenses this file
+# to you under the Apache License, Version 2.0 (the
+# "License"); you may not use this file except in compliance
+# with the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an
+# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+# KIND, either express or implied. See the License for the
+# specific language governing permissions and limitations
+# under the License.
+
+# Verifies that the parquet-format.version sidecar file is tracking a released
+# parquet-format version and that the inlined parquet.thrift byte-matches the
+# upstream file at that version.
+
+set -euo pipefail
+
+SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
+REPO_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd)"
+THRIFT_FILE="${REPO_ROOT}/parquet-format-structures/src/main/thrift/parquet.thrift"
+SIDECAR_FILE="${REPO_ROOT}/parquet-format-structures/src/main/thrift/parquet-format.version"
+
+# shellcheck source=parquet-thrift-lib.sh
+source "${SCRIPT_DIR}/parquet-thrift-lib.sh"
+
+if [[ ! -f "$SIDECAR_FILE" ]]; then
+ echo "ERROR: sidecar not found: ${SIDECAR_FILE}" >&2
+ exit 1
+fi
+
+if [[ ! -f "$THRIFT_FILE" ]]; then
+ echo "ERROR: inlined thrift not found: ${THRIFT_FILE}" >&2
+ exit 1
+fi
+
+version="$(grep '^parquet-format.version=' "$SIDECAR_FILE" | cut -d= -f2 || true)"
+commit="$(grep '^parquet-format.commit=' "$SIDECAR_FILE" | cut -d= -f2 || true)"
+
+# Require a valid X.Y.Z release version
+if [[ -z "$version" ]] || ! [[ "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
+ echo "ERROR: inlined parquet.thrift is not from a released parquet-format version." >&2
+ echo " parquet-format.version = '${version}'" >&2
+ echo "" >&2
+ echo "A parquet-java release must use a released parquet-format IDL." >&2
+ echo "Run: dev/update-parquet-thrift.sh --version " >&2
+ exit 1
+fi
+
+echo "Checking inlined parquet.thrift against parquet-format ${version} (commit ${commit}) ..."
+
+# Fetch the canonical IDL from the release tag. Never writes to tracked files.
+# Fails closed: a network/git failure or an empty file all exit non-zero.
+tmp="$(mktemp)"
+trap 'rm -f "$tmp"' EXIT
+
+if ! fetch_thrift "apache-parquet-format-${version}" "$tmp"; then
+ echo "ERROR: failed to fetch parquet.thrift for tag apache-parquet-format-${version}" >&2
+ echo "Cannot verify inlined IDL — failing closed." >&2
+ exit 1
+fi
+
+if [[ ! -s "$tmp" ]]; then
+ echo "ERROR: fetched parquet.thrift is empty for tag apache-parquet-format-${version}" >&2
+ echo "Cannot verify inlined IDL — failing closed." >&2
+ exit 1
+fi
+
+if diff -u "$tmp" "$THRIFT_FILE"; then
+ echo "OK: inlined parquet.thrift matches parquet-format ${version}."
+ exit 0
+else
+ echo "" >&2
+ echo "ERROR: inlined parquet.thrift differs from parquet-format ${version}." >&2
+ echo "Run: dev/update-parquet-thrift.sh --version ${version}" >&2
+ exit 1
+fi
diff --git a/dev/parquet-thrift-lib.sh b/dev/parquet-thrift-lib.sh
new file mode 100755
index 0000000000..d225ff4fe4
--- /dev/null
+++ b/dev/parquet-thrift-lib.sh
@@ -0,0 +1,63 @@
+#!/usr/bin/env bash
+#
+# Licensed to the Apache Software Foundation (ASF) under one
+# or more contributor license agreements. See the NOTICE file
+# distributed with this work for additional information
+# regarding copyright ownership. The ASF licenses this file
+# to you under the Apache License, Version 2.0 (the
+# "License"); you may not use this file except in compliance
+# with the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an
+# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+# KIND, either express or implied. See the License for the
+# specific language governing permissions and limitations
+# under the License.
+#
+# Shared helpers for update-parquet-thrift.sh and check-parquet-thrift-release.sh.
+# Source this file; do not execute it. No top-level side effects, no `exit` calls.
+
+PARQUET_FORMAT_REPO="https://github.com/apache/parquet-format"
+PARQUET_FORMAT_RAW="https://raw.githubusercontent.com/apache/parquet-format"
+THRIFT_PATH_IN_FORMAT="src/main/thrift/parquet.thrift"
+
+# Resolve a parquet-format release version to its full 40-char commit sha using
+# git ls-remote. Prefers the dereferenced ref (^{}) so it works for annotated
+# and lightweight tags. Echoes the sha and returns 0 on success; returns 1 if
+# the tag does not exist; returns git's exit code (and lets git's stderr through)
+# on git/network failure.
+resolve_version_to_commit() {
+ local ver="$1" sha ls_out ls_rc
+ ls_out="$(git ls-remote --tags "$PARQUET_FORMAT_REPO" \
+ "refs/tags/apache-parquet-format-${ver}" \
+ "refs/tags/apache-parquet-format-${ver}^{}")" || { ls_rc=$?; return $ls_rc; }
+ sha="$(printf '%s\n' "$ls_out" \
+ | awk '{ if ($2 ~ /\^\{\}$/) deref=$1; else plain=$1 }
+ END { print (deref != "" ? deref : plain) }')"
+ [[ -n "$sha" ]] || return 1
+ printf '%s\n' "$sha"
+}
+
+# Fetch the parquet.thrift IDL from the parquet-format repo at the given commit
+# sha or tag, writing output to $2. Reads from raw.githubusercontent.com, which
+# serves any commit sha or tag and is not subject to the GitHub REST API rate
+# limit. Returns 0 on success, 1 on any failure.
+# The caller is responsible for providing a temp path for $2 so tracked source
+# files are never overwritten on failure.
+fetch_thrift() {
+ local ref="$1" dest="$2" url attempt
+ url="${PARQUET_FORMAT_RAW}/${ref}/${THRIFT_PATH_IN_FORMAT}"
+ for attempt in 1 2 3; do
+ if curl -fsSL -o "$dest" "$url" && [[ -s "$dest" ]]; then
+ return 0
+ fi
+ if [[ "$attempt" -eq 3 ]]; then
+ echo "ERROR: failed to fetch ${url} after ${attempt} attempts" >&2
+ return 1
+ fi
+ sleep $((attempt * 2))
+ done
+}
diff --git a/dev/prepare-release.sh b/dev/prepare-release.sh
index b53f7707b0..7b8afe4017 100755
--- a/dev/prepare-release.sh
+++ b/dev/prepare-release.sh
@@ -38,6 +38,9 @@ new_development_version="$release_version-SNAPSHOT"
tag="apache-parquet-$release_version-rc$2"
+# Ensure the inlined parquet.thrift matches an official parquet-format release.
+"$(dirname "$0")/check-parquet-thrift-release.sh"
+
./mvnw release:clean
./mvnw release:prepare -DskipTests -Darguments=-DskipTests -Dtag="$tag" "-DreleaseVersion=$release_version" -DdevelopmentVersion="$new_development_version"
diff --git a/dev/update-parquet-thrift.sh b/dev/update-parquet-thrift.sh
new file mode 100755
index 0000000000..0cd9d19a02
--- /dev/null
+++ b/dev/update-parquet-thrift.sh
@@ -0,0 +1,165 @@
+#!/usr/bin/env bash
+#
+# Licensed to the Apache Software Foundation (ASF) under one
+# or more contributor license agreements. See the NOTICE file
+# distributed with this work for additional information
+# regarding copyright ownership. The ASF licenses this file
+# to you under the Apache License, Version 2.0 (the
+# "License"); you may not use this file except in compliance
+# with the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an
+# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+# KIND, either express or implied. See the License for the
+# specific language governing permissions and limitations
+# under the License.
+#
+# Updates the inlined parquet.thrift and sidecar to a specific parquet-format
+# commit or release.
+#
+# Usage:
+# update-parquet-thrift.sh -- POC: sets version=UNRELEASED
+# update-parquet-thrift.sh --version -- release sync: derives commit from tag
+
+set -euo pipefail
+
+SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
+REPO_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd)"
+THRIFT_FILE="${REPO_ROOT}/parquet-format-structures/src/main/thrift/parquet.thrift"
+SIDECAR_FILE="${REPO_ROOT}/parquet-format-structures/src/main/thrift/parquet-format.version"
+
+# shellcheck source=parquet-thrift-lib.sh
+source "${SCRIPT_DIR}/parquet-thrift-lib.sh"
+
+usage() {
+ cat < POC sync — fetch parquet.thrift at that commit; version=UNRELEASED
+ $0 --version Release sync — resolve tag apache-parquet-format-
+
+Only one of a commit hash or --version may be given, not both.
+Commit mode requires a full 40-character sha (short shas cannot be fetched from a remote).
+EOF
+ exit 1
+}
+
+if [[ $# -eq 0 ]]; then
+ usage
+fi
+
+mode=""
+commit_arg=""
+version_arg=""
+
+if [[ "$1" == "--version" ]]; then
+ [[ $# -ne 2 ]] && usage
+ mode="version"
+ version_arg="$2"
+elif [[ "$1" == --* ]]; then
+ usage
+else
+ [[ $# -ne 1 ]] && usage
+ mode="commit"
+ commit_arg="$1"
+fi
+
+# Validate arguments
+if [[ "$mode" == "version" ]]; then
+ if ! [[ "$version_arg" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
+ echo "ERROR: version must be X.Y.Z (e.g. 2.13.0), got: $version_arg" >&2
+ exit 1
+ fi
+else
+ # Full 40-char sha required — short shas cannot be fetched from a remote (git exits 128).
+ if ! [[ "$commit_arg" =~ ^[0-9a-f]{40}$ ]]; then
+ echo "ERROR: commit hash must be exactly 40 hex chars, got: $commit_arg" >&2
+ echo " Obtain the full sha with: git ls-remote ${PARQUET_FORMAT_REPO} HEAD" >&2
+ exit 1
+ fi
+fi
+
+# Resolve the commit sha to use
+resolved_sha=""
+resolved_version=""
+
+if [[ "$mode" == "version" ]]; then
+ echo "Resolving tag apache-parquet-format-${version_arg} ..."
+ resolved_sha="$(resolve_version_to_commit "$version_arg")" || {
+ rc=$?
+ if [[ $rc -eq 1 ]]; then
+ echo "ERROR: tag apache-parquet-format-${version_arg} not found in ${PARQUET_FORMAT_REPO}" >&2
+ echo " Check that the version exists: git ls-remote --tags ${PARQUET_FORMAT_REPO}" >&2
+ else
+ echo "ERROR: git/network failure resolving tag apache-parquet-format-${version_arg}" >&2
+ fi
+ exit 1
+ }
+ resolved_version="$version_arg"
+ echo " -> commit: $resolved_sha"
+else
+ resolved_sha="$commit_arg"
+ resolved_version="UNRELEASED"
+fi
+
+# Read the old commit from the sidecar (for the summary)
+old_commit="(none)"
+if [[ -f "$SIDECAR_FILE" ]]; then
+ old_commit="$(grep '^parquet-format.commit=' "$SIDECAR_FILE" | cut -d= -f2 || true)"
+fi
+
+# Fetch parquet.thrift into a temp file. Only overwrite the tracked file after
+# the fetch and validation succeed, so a failure never deletes tracked source.
+tmp_thrift="$(mktemp)"
+trap 'rm -f "$tmp_thrift"' EXIT
+
+echo "Fetching parquet.thrift at ${resolved_sha} ..."
+if ! fetch_thrift "$resolved_sha" "$tmp_thrift"; then
+ echo "ERROR: could not fetch parquet.thrift at ${resolved_sha}" >&2
+ echo " The tracked files were not modified." >&2
+ exit 1
+fi
+
+# Validate: non-empty and contains the expected namespace declaration.
+if [[ ! -s "$tmp_thrift" ]]; then
+ echo "ERROR: fetched parquet.thrift is empty" >&2
+ echo " The tracked files were not modified." >&2
+ exit 1
+fi
+if ! grep -q 'namespace java org.apache.parquet.format' "$tmp_thrift"; then
+ echo "ERROR: fetched file does not look like parquet.thrift (missing namespace declaration)" >&2
+ echo " The tracked files were not modified." >&2
+ exit 1
+fi
+
+# Commit the validated thrift file, then write the sidecar. Order matters:
+# if the mv fails, the sidecar is not yet updated, keeping the two in sync.
+mv "$tmp_thrift" "$THRIFT_FILE"
+chmod 644 "$THRIFT_FILE"
+echo " -> written to ${THRIFT_FILE}"
+
+source_url="${PARQUET_FORMAT_REPO}/blob/${resolved_sha}/src/main/thrift/parquet.thrift"
+cat > "$SIDECAR_FILE" < sidecar updated: ${SIDECAR_FILE}"
+
+# Summary
+echo ""
+echo "Update complete:"
+echo " old commit: ${old_commit}"
+echo " new commit: ${resolved_sha}"
+if [[ "$resolved_version" == "UNRELEASED" ]]; then
+ echo " version: UNRELEASED (POC sync — run --version for a release sync)"
+else
+ echo " version: ${resolved_version} (release sync)"
+fi
+echo ""
+echo "Next steps: rebuild parquet-format-structures and review the diff:"
+echo " ./mvnw -pl parquet-format-structures -am install -DskipTests"
+echo " git diff parquet-format-structures/src/main/thrift/parquet.thrift"
diff --git a/parquet-format-structures/README.md b/parquet-format-structures/README.md
new file mode 100644
index 0000000000..2fba0e0a39
--- /dev/null
+++ b/parquet-format-structures/README.md
@@ -0,0 +1,95 @@
+
+
+# parquet-format-structures
+
+This module compiles the Parquet Thrift IDL (`parquet.thrift`) into Java classes
+(`org.apache.parquet.format.*`) using the thrift compiler.
+
+## Inlined IDL
+
+`parquet.thrift` is kept as a verbatim in-tree copy at
+`src/main/thrift/parquet.thrift`. The build reads it directly from there; no
+download happens at build time.
+
+The accompanying sidecar file `src/main/thrift/parquet-format.version` records
+provenance:
+
+```
+parquet-format.commit=
+parquet-format.version=
+parquet-format.source=
+```
+
+- `parquet-format.commit` — the parquet-format commit the inlined IDL was taken
+ from. Always a full 40-character sha.
+- `parquet-format.version` — the released parquet-format version whose tag points
+ at that commit, e.g. `2.13.0`. Set to `UNRELEASED` for POC/work-in-progress
+ syncs that do not correspond to a released tag.
+- `parquet-format.source` — convenience URL for browsing the upstream file at
+ that exact commit.
+
+## Upgrading the IDL
+
+Use `dev/update-parquet-thrift.sh`. It resolves release tags with `git ls-remote`
+and downloads the IDL from `raw.githubusercontent.com`, so it does not use the
+GitHub REST API and is not subject to its rate limit. A failed update leaves the
+inlined files untouched — the script writes to a temporary file, validates it,
+and only moves it into place on success.
+
+**POC / work-in-progress sync** (sets `version=UNRELEASED`):
+
+```sh
+dev/update-parquet-thrift.sh
+```
+
+The commit sha must be exactly 40 hex characters — short shas cannot be fetched
+from the remote.
+
+**Release sync** (sets `version=`):
+
+```sh
+dev/update-parquet-thrift.sh --version
+```
+
+This resolves the `apache-parquet-format-` tag to its commit sha
+automatically. You do not need to look up the sha yourself.
+
+After running either form, rebuild and review:
+
+```sh
+./mvnw -pl parquet-format-structures -am install -DskipTests
+git diff parquet-format-structures/src/main/thrift/parquet.thrift
+```
+
+## Release requirement
+
+Before cutting a parquet-java release, the inlined IDL must match an official
+parquet-format release. `dev/prepare-release.sh` enforces this automatically by
+calling `dev/check-parquet-thrift-release.sh` before running `mvnw release:prepare`.
+
+To prepare for a release, sync to the appropriate parquet-format version:
+
+```sh
+dev/update-parquet-thrift.sh --version
+```
+
+The check fetches the canonical IDL via `git` and diffs it against the inlined
+file. Any difference — including a `version=UNRELEASED` sidecar — fails the
+check and aborts the release.
diff --git a/parquet-format-structures/pom.xml b/parquet-format-structures/pom.xml
index 3b31afca90..9a8bbaff22 100644
--- a/parquet-format-structures/pom.xml
+++ b/parquet-format-structures/pom.xml
@@ -34,44 +34,14 @@
https://parquet.apache.org/
Parquet-mr related java classes to use the parquet-format thrift structures.
-
- ${project.build.directory}/parquet-format-thrift
-
-
-
-
- org.apache.maven.plugins
- maven-dependency-plugin
-
-
- unpack
- initialize
-
- unpack
-
-
-
-
- org.apache.parquet
- parquet-format
- ${parquet.format.version}
- jar
-
-
- parquet.thrift
- ${parquet.thrift.path}
-
-
-
-
org.apache.thrift
thrift-maven-plugin
- ${parquet.thrift.path}
+ ${project.basedir}/src/main/thrift
${format.thrift.executable}
diff --git a/parquet-format-structures/src/main/thrift/parquet-format.version b/parquet-format-structures/src/main/thrift/parquet-format.version
new file mode 100644
index 0000000000..d4043b076f
--- /dev/null
+++ b/parquet-format-structures/src/main/thrift/parquet-format.version
@@ -0,0 +1,4 @@
+# Provenance of the inlined parquet.thrift. Maintained by dev/update-parquet-thrift.sh.
+parquet-format.commit=c47e2a66e88943fc46fde1b028a9432f14fdf5c0
+parquet-format.version=2.13.0
+parquet-format.source=https://github.com/apache/parquet-format/blob/c47e2a66e88943fc46fde1b028a9432f14fdf5c0/src/main/thrift/parquet.thrift
diff --git a/parquet-format-structures/src/main/thrift/parquet.thrift b/parquet-format-structures/src/main/thrift/parquet.thrift
new file mode 100644
index 0000000000..fe259d61bc
--- /dev/null
+++ b/parquet-format-structures/src/main/thrift/parquet.thrift
@@ -0,0 +1,1443 @@
+/**
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+/**
+ * File format description for the parquet file format
+ */
+namespace cpp parquet
+namespace java org.apache.parquet.format
+
+/**
+ * Types supported by Parquet. These types are intended to be used in combination
+ * with the encodings to control the on disk storage format.
+ * For example INT16 is not included as a type since a good encoding of INT32
+ * would handle this.
+ */
+enum Type {
+ BOOLEAN = 0;
+ INT32 = 1;
+ INT64 = 2;
+ INT96 = 3; // deprecated, new Parquet writers should not write data in INT96
+ FLOAT = 4;
+ DOUBLE = 5;
+ BYTE_ARRAY = 6;
+ FIXED_LEN_BYTE_ARRAY = 7;
+}
+
+/**
+ * DEPRECATED: Common types used by frameworks (e.g. Hive, Pig) using parquet.
+ * ConvertedType is superseded by LogicalType. This enum should not be extended.
+ *
+ * See LogicalTypes.md for conversion between ConvertedType and LogicalType.
+ */
+enum ConvertedType {
+ /** a BYTE_ARRAY actually contains UTF8 encoded chars */
+ UTF8 = 0;
+
+ /** a map is converted as an optional field containing a repeated key/value pair */
+ MAP = 1;
+
+ /** a key/value pair is converted into a group of two fields */
+ MAP_KEY_VALUE = 2;
+
+ /** a list is converted into an optional field containing a repeated field for its
+ * values */
+ LIST = 3;
+
+ /** an enum is converted into a BYTE_ARRAY field */
+ ENUM = 4;
+
+ /**
+ * A decimal value.
+ *
+ * This may be used to annotate BYTE_ARRAY or FIXED_LEN_BYTE_ARRAY primitive
+ * types. The underlying byte array stores the unscaled value encoded as two's
+ * complement using big-endian byte order (the most significant byte is the
+ * zeroth element). The value of the decimal is the value * 10^{-scale}.
+ *
+ * This must be accompanied by a (maximum) precision and a scale in the
+ * SchemaElement. The precision specifies the number of digits in the decimal
+ * and the scale stores the location of the decimal point. For example 1.23
+ * would have precision 3 (3 total digits) and scale 2 (the decimal point is
+ * 2 digits over).
+ */
+ DECIMAL = 5;
+
+ /**
+ * A Date
+ *
+ * Stored as days since Unix epoch, encoded as the INT32 physical type.
+ *
+ */
+ DATE = 6;
+
+ /**
+ * A time
+ *
+ * The total number of milliseconds since midnight. The value is stored
+ * as an INT32 physical type.
+ */
+ TIME_MILLIS = 7;
+
+ /**
+ * A time.
+ *
+ * The total number of microseconds since midnight. The value is stored as
+ * an INT64 physical type.
+ */
+ TIME_MICROS = 8;
+
+ /**
+ * A date/time combination
+ *
+ * Date and time recorded as milliseconds since the Unix epoch. Recorded as
+ * a physical type of INT64.
+ */
+ TIMESTAMP_MILLIS = 9;
+
+ /**
+ * A date/time combination
+ *
+ * Date and time recorded as microseconds since the Unix epoch. The value is
+ * stored as an INT64 physical type.
+ */
+ TIMESTAMP_MICROS = 10;
+
+
+ /**
+ * An unsigned integer value.
+ *
+ * The number describes the maximum number of meaningful data bits in
+ * the stored value. 8, 16 and 32 bit values are stored using the
+ * INT32 physical type. 64 bit values are stored using the INT64
+ * physical type.
+ *
+ */
+ UINT_8 = 11;
+ UINT_16 = 12;
+ UINT_32 = 13;
+ UINT_64 = 14;
+
+ /**
+ * A signed integer value.
+ *
+ * The number describes the maximum number of meaningful data bits in
+ * the stored value. 8, 16 and 32 bit values are stored using the
+ * INT32 physical type. 64 bit values are stored using the INT64
+ * physical type.
+ *
+ */
+ INT_8 = 15;
+ INT_16 = 16;
+ INT_32 = 17;
+ INT_64 = 18;
+
+ /**
+ * An embedded JSON document
+ *
+ * A JSON document embedded within a single UTF8 column.
+ */
+ JSON = 19;
+
+ /**
+ * An embedded BSON document
+ *
+ * A BSON document embedded within a single BYTE_ARRAY column.
+ */
+ BSON = 20;
+
+ /**
+ * An interval of time
+ *
+ * This type annotates data stored as a FIXED_LEN_BYTE_ARRAY of length 12
+ * This data is composed of three separate little endian unsigned
+ * integers. Each stores a component of a duration of time. The first
+ * integer identifies the number of months associated with the duration,
+ * the second identifies the number of days associated with the duration
+ * and the third identifies the number of milliseconds associated with
+ * the provided duration. This duration of time is independent of any
+ * particular timezone or date.
+ */
+ INTERVAL = 21;
+}
+
+/**
+ * Representation of Schemas
+ */
+enum FieldRepetitionType {
+ /** This field is required (can not be null) and each row has exactly 1 value. */
+ REQUIRED = 0;
+
+ /** The field is optional (can be null) and each row has 0 or 1 values. */
+ OPTIONAL = 1;
+
+ /** The field is repeated and can contain 0 or more values */
+ REPEATED = 2;
+}
+
+/**
+ * A structure for capturing metadata for estimating the unencoded,
+ * uncompressed size of data written. This is useful for readers to estimate
+ * how much memory is needed to reconstruct data in their memory model and for
+ * fine grained filter pushdown on nested structures (the histograms contained
+ * in this structure can help determine the number of nulls at a particular
+ * nesting level and maximum length of lists).
+ */
+struct SizeStatistics {
+ /**
+ * The number of physical bytes stored for BYTE_ARRAY data values assuming
+ * no encoding. This is exclusive of the bytes needed to store the length of
+ * each byte array. In other words, this field is equivalent to the `(size
+ * of PLAIN-ENCODING the byte array values) - (4 bytes * number of values
+ * written)`. To determine unencoded sizes of other types readers can use
+ * schema information multiplied by the number of non-null and null values.
+ * The number of null/non-null values can be inferred from the histograms
+ * below.
+ *
+ * For example, if a column chunk is dictionary-encoded with dictionary
+ * ["a", "bc", "cde"], and a data page contains the indices [0, 0, 1, 2],
+ * then this value for that data page should be 7 (1 + 1 + 2 + 3).
+ *
+ * This field should only be set for types that use BYTE_ARRAY as their
+ * physical type.
+ */
+ 1: optional i64 unencoded_byte_array_data_bytes;
+ /**
+ * When present, there is expected to be one element corresponding to each
+ * repetition (i.e. size=max repetition_level+1) where each element
+ * represents the number of times the repetition level was observed in the
+ * data.
+ *
+ * This field may be omitted if max_repetition_level is 0 without loss
+ * of information.
+ **/
+ 2: optional list repetition_level_histogram;
+ /**
+ * Same as repetition_level_histogram except for definition levels.
+ *
+ * This field may be omitted if max_definition_level is 0 or 1 without
+ * loss of information.
+ **/
+ 3: optional list definition_level_histogram;
+}
+
+/**
+ * Bounding box for GEOMETRY or GEOGRAPHY type in the representation of min/max
+ * value pair of coordinates from each axis.
+ */
+struct BoundingBox {
+ 1: required double xmin;
+ 2: required double xmax;
+ 3: required double ymin;
+ 4: required double ymax;
+ 5: optional double zmin;
+ 6: optional double zmax;
+ 7: optional double mmin;
+ 8: optional double mmax;
+}
+
+/** Statistics specific to Geometry and Geography logical types */
+struct GeospatialStatistics {
+ /** A bounding box of geospatial instances */
+ 1: optional BoundingBox bbox;
+ /** Geospatial type codes of all instances, or an empty list if not known */
+ 2: optional list geospatial_types;
+}
+
+/**
+ * Statistics per row group and per page
+ * All fields are optional.
+ */
+struct Statistics {
+ /**
+ * DEPRECATED: min and max value of the column. Use min_value and max_value.
+ *
+ * Values are encoded using PLAIN encoding, except that variable-length byte
+ * arrays do not include a length prefix.
+ *
+ * These fields encode min and max values determined by signed comparison
+ * only. New files should use the correct order for a column's logical type
+ * and store the values in the min_value and max_value fields.
+ *
+ * To support older readers, these may be set when the column order is
+ * signed.
+ */
+ 1: optional binary max;
+ 2: optional binary min;
+ /**
+ * Count of null values in the column.
+ *
+ * Writers SHOULD always write this field even if it is zero (i.e. no null value)
+ * or the column is not nullable.
+ * Readers MUST distinguish between null_count not being present and null_count == 0.
+ * If null_count is not present, readers MUST NOT assume null_count == 0.
+ */
+ 3: optional i64 null_count;
+ /** count of distinct values occurring */
+ 4: optional i64 distinct_count;
+ /**
+ * Lower and upper bound values for the column, determined by its ColumnOrder.
+ *
+ * These may be the actual minimum and maximum values found on a page or column
+ * chunk, but can also be (more compact) values that do not exist on a page or
+ * column chunk. For example, instead of storing "Blart Versenwald III", a writer
+ * may set min_value="B", max_value="C". Such more compact values must still be
+ * valid values within the column's logical type.
+ *
+ * Values are encoded using PLAIN encoding, except that variable-length byte
+ * arrays do not include a length prefix.
+ */
+ 5: optional binary max_value;
+ 6: optional binary min_value;
+ /** If true, max_value is the actual maximum value for a column */
+ 7: optional bool is_max_value_exact;
+ /** If true, min_value is the actual minimum value for a column */
+ 8: optional bool is_min_value_exact;
+ /**
+ * Count of NaN values in the column; only present if physical type is FLOAT
+ * or DOUBLE, or logical type is FLOAT16.
+ * If this field is not present, readers MUST assume NaNs may be present
+ * (i.e. MUST assume nan_count > 0 and MAY NOT assume nan_count == 0).
+ */
+ 9: optional i64 nan_count;
+}
+
+/** Empty structs to use as logical type annotations */
+struct StringType {} // allowed for BYTE_ARRAY, must be encoded with UTF-8
+struct UUIDType {} // allowed for FIXED[16], must be encoded as raw UUID bytes
+struct MapType {} // see LogicalTypes.md
+struct ListType {} // see LogicalTypes.md
+struct EnumType {} // allowed for BYTE_ARRAY, must be encoded with UTF-8
+struct DateType {} // allowed for INT32
+struct Float16Type {} // allowed for FIXED[2], must be encoded as raw FLOAT16 bytes (see LogicalTypes.md)
+
+/**
+ * Logical type to annotate a column that is always null.
+ *
+ * Sometimes when discovering the schema of existing data, values are always
+ * null and the physical type can't be determined. This annotation signals
+ * the case where the physical type was guessed from all null values.
+ */
+struct NullType {} // allowed for any physical type, only null values stored
+
+/**
+ * Decimal logical type annotation
+ *
+ * Scale must be zero or a positive integer less than or equal to the precision.
+ * Precision must be a non-zero positive integer.
+ *
+ * To maintain forward-compatibility in v1, implementations using this logical
+ * type must also set scale and precision on the annotated SchemaElement.
+ *
+ * Allowed for physical types: INT32, INT64, FIXED_LEN_BYTE_ARRAY, and BYTE_ARRAY.
+ */
+struct DecimalType {
+ 1: required i32 scale
+ 2: required i32 precision
+}
+
+/** Time units for logical types */
+struct MilliSeconds {}
+struct MicroSeconds {}
+struct NanoSeconds {}
+union TimeUnit {
+ 1: MilliSeconds MILLIS
+ 2: MicroSeconds MICROS
+ 3: NanoSeconds NANOS
+}
+
+/**
+ * Timestamp logical type annotation
+ *
+ * Allowed for physical types: INT64
+ */
+struct TimestampType {
+ 1: required bool isAdjustedToUTC
+ 2: required TimeUnit unit
+}
+
+/**
+ * Time logical type annotation
+ *
+ * Allowed for physical types: INT32 (millis), INT64 (micros, nanos)
+ */
+struct TimeType {
+ 1: required bool isAdjustedToUTC
+ 2: required TimeUnit unit
+}
+
+/**
+ * Integer logical type annotation
+ *
+ * bitWidth must be 8, 16, 32, or 64.
+ *
+ * Allowed for physical types: INT32, INT64
+ */
+struct IntType {
+ 1: required i8 bitWidth
+ 2: required bool isSigned
+}
+
+/**
+ * Embedded JSON logical type annotation
+ *
+ * Allowed for physical types: BYTE_ARRAY
+ */
+struct JsonType {
+}
+
+/**
+ * Embedded BSON logical type annotation
+ *
+ * Allowed for physical types: BYTE_ARRAY
+ */
+struct BsonType {
+}
+
+/**
+ * Embedded Variant logical type annotation
+ */
+struct VariantType {
+ // The version of the variant specification that the variant was
+ // written with.
+ 1: optional i8 specification_version
+}
+
+/** Edge interpolation algorithm for Geography logical type */
+enum EdgeInterpolationAlgorithm {
+ SPHERICAL = 0;
+ VINCENTY = 1;
+ THOMAS = 2;
+ ANDOYER = 3;
+ KARNEY = 4;
+}
+
+/**
+ * Embedded Geometry logical type annotation
+ *
+ * Geospatial features in the Well-Known Binary (WKB) format and `edges` interpolation
+ * is always linear/planar.
+ *
+ * A custom CRS can be set by the crs field. If unset, it defaults to "OGC:CRS84",
+ * which means that the geometries must be stored in longitude, latitude based on
+ * the WGS84 datum.
+ *
+ * Allowed for physical type: BYTE_ARRAY.
+ *
+ * See Geospatial.md for details.
+ */
+struct GeometryType {
+ 1: optional string crs;
+}
+
+/**
+ * Embedded Geography logical type annotation
+ *
+ * Geospatial features in the WKB format with an explicit (non-linear/non-planar)
+ * `edges` interpolation algorithm.
+ *
+ * A custom geographic CRS can be set by the crs field, where longitudes are
+ * bound by [-180, 180] and latitudes are bound by [-90, 90]. If unset, the CRS
+ * defaults to "OGC:CRS84".
+ *
+ * An optional algorithm can be set to correctly interpret `edges` interpolation
+ * of the geometries. If unset, the algorithm defaults to SPHERICAL.
+ *
+ * Allowed for physical type: BYTE_ARRAY.
+ *
+ * See Geospatial.md for details.
+ */
+struct GeographyType {
+ 1: optional string crs;
+ 2: optional EdgeInterpolationAlgorithm algorithm;
+}
+
+/**
+ * LogicalType annotations to replace ConvertedType.
+ *
+ * To maintain compatibility, implementations using LogicalType for a
+ * SchemaElement must also set the corresponding ConvertedType (if any)
+ * from the following table.
+ */
+union LogicalType {
+ 1: StringType STRING // use ConvertedType UTF8
+ 2: MapType MAP // use ConvertedType MAP
+ 3: ListType LIST // use ConvertedType LIST
+ 4: EnumType ENUM // use ConvertedType ENUM
+ 5: DecimalType DECIMAL // use ConvertedType DECIMAL + SchemaElement.{scale, precision}
+ 6: DateType DATE // use ConvertedType DATE
+
+ // use ConvertedType TIME_MICROS for TIME(isAdjustedToUTC = *, unit = MICROS)
+ // use ConvertedType TIME_MILLIS for TIME(isAdjustedToUTC = *, unit = MILLIS)
+ 7: TimeType TIME
+
+ // use ConvertedType TIMESTAMP_MICROS for TIMESTAMP(isAdjustedToUTC = *, unit = MICROS)
+ // use ConvertedType TIMESTAMP_MILLIS for TIMESTAMP(isAdjustedToUTC = *, unit = MILLIS)
+ 8: TimestampType TIMESTAMP
+
+ // 9: reserved for INTERVAL
+ 10: IntType INTEGER // use ConvertedType INT_* or UINT_*
+ 11: NullType UNKNOWN // no compatible ConvertedType
+ 12: JsonType JSON // use ConvertedType JSON
+ 13: BsonType BSON // use ConvertedType BSON
+ 14: UUIDType UUID // no compatible ConvertedType
+ 15: Float16Type FLOAT16 // no compatible ConvertedType
+ 16: VariantType VARIANT // no compatible ConvertedType
+ 17: GeometryType GEOMETRY // no compatible ConvertedType
+ 18: GeographyType GEOGRAPHY // no compatible ConvertedType
+}
+
+/**
+ * Represents an element inside a schema definition.
+ * - if it is a group (inner node) then type is undefined and num_children is defined
+ * - if it is a primitive type (leaf) then type is defined and num_children is undefined
+ * the nodes are listed in depth first traversal order.
+ */
+struct SchemaElement {
+ /** Data type for this field. Not set if the current element is a non-leaf node */
+ 1: optional Type type;
+
+ /** If type is FIXED_LEN_BYTE_ARRAY, this is the byte length of the values.
+ * Otherwise, if specified, this is the maximum bit length to store any of the values.
+ * (e.g. a low cardinality INT col could have this set to 3). Note that this is
+ * in the schema, and therefore fixed for the entire file.
+ */
+ 2: optional i32 type_length;
+
+ /** repetition of the field. The root of the schema does not have a repetition_type.
+ * All other nodes must have one */
+ 3: optional FieldRepetitionType repetition_type;
+
+ /** Name of the field in the schema */
+ 4: required string name;
+
+ /** Nested fields. Since thrift does not support nested fields,
+ * the nesting is flattened to a single list by a depth-first traversal.
+ * The children count is used to construct the nested relationship.
+ * This field is not set when the element is a primitive type
+ */
+ 5: optional i32 num_children;
+
+ /**
+ * DEPRECATED: When the schema is the result of a conversion from another model.
+ * Used to record the original type to help with cross conversion.
+ *
+ * This is superseded by logicalType.
+ */
+ 6: optional ConvertedType converted_type;
+
+ /**
+ * DEPRECATED: Used when this column contains decimal data.
+ * See the DECIMAL converted type for more details.
+ *
+ * This is superseded by using the DecimalType annotation in logicalType.
+ */
+ 7: optional i32 scale
+ 8: optional i32 precision
+
+ /** When the original schema supports field ids, this will save the
+ * original field id in the parquet schema
+ */
+ 9: optional i32 field_id;
+
+ /**
+ * The logical type of this SchemaElement
+ *
+ * LogicalType replaces ConvertedType, but ConvertedType is still required
+ * for some logical types to ensure forward-compatibility in format v1.
+ */
+ 10: optional LogicalType logicalType
+}
+
+/**
+ * Encodings supported by Parquet. Not all encodings are valid for all types. These
+ * enums are also used to specify the encoding of definition and repetition levels.
+ * See the accompanying doc for the details of the more complicated encodings.
+ */
+enum Encoding {
+ /** Default encoding.
+ * BOOLEAN - 1 bit per value. 0 is false; 1 is true.
+ * INT32 - 4 bytes per value. Stored as little-endian.
+ * INT64 - 8 bytes per value. Stored as little-endian.
+ * FLOAT - 4 bytes per value. IEEE. Stored as little-endian.
+ * DOUBLE - 8 bytes per value. IEEE. Stored as little-endian.
+ * BYTE_ARRAY - 4 byte length stored as little endian, followed by bytes.
+ * FIXED_LEN_BYTE_ARRAY - Just the bytes.
+ */
+ PLAIN = 0;
+
+ /** Group VarInt encoding for INT32/INT64.
+ * This encoding is deprecated. It was never used.
+ */
+ // GROUP_VAR_INT = 1;
+
+ /**
+ * DEPRECATED: Dictionary encoding. The values in the dictionary are encoded in the
+ * plain type.
+ * For a data page use RLE_DICTIONARY instead.
+ * For a Dictionary page use PLAIN instead.
+ */
+ PLAIN_DICTIONARY = 2;
+
+ /** Group packed run length encoding. Usable for definition/repetition levels
+ * encoding and Booleans (on one bit: 0 is false; 1 is true.)
+ */
+ RLE = 3;
+
+ /** DEPRECATED: Bit packed encoding. This can only be used if the data has a known max
+ * width. Usable for definition/repetition levels encoding.
+ * Superseded by RLE (which is a hybrid of RLE and bit packing); see Encodings.md.
+ */
+ BIT_PACKED = 4;
+
+ /** Delta encoding for integers. This can be used for int columns and works best
+ * on sorted data
+ */
+ DELTA_BINARY_PACKED = 5;
+
+ /** Encoding for byte arrays to separate the length values and the data. The lengths
+ * are encoded using DELTA_BINARY_PACKED
+ */
+ DELTA_LENGTH_BYTE_ARRAY = 6;
+
+ /** Incremental-encoded byte array. Prefix lengths are encoded using DELTA_BINARY_PACKED.
+ * Suffixes are stored as delta length byte arrays.
+ */
+ DELTA_BYTE_ARRAY = 7;
+
+ /** Dictionary encoding: the ids are encoded using the RLE encoding
+ */
+ RLE_DICTIONARY = 8;
+
+ /** Encoding for fixed-width data (FLOAT, DOUBLE, INT32, INT64, FIXED_LEN_BYTE_ARRAY).
+ K byte-streams are created where K is the size in bytes of the data type.
+ The individual bytes of a value are scattered to the corresponding stream and
+ the streams are concatenated.
+ This itself does not reduce the size of the data but can lead to better compression
+ afterwards.
+
+ Added in 2.8 for FLOAT and DOUBLE.
+ Support for INT32, INT64 and FIXED_LEN_BYTE_ARRAY added in 2.11.
+ */
+ BYTE_STREAM_SPLIT = 9;
+}
+
+/**
+ * Supported compression algorithms.
+ *
+ * Codecs added in format version X.Y can be read by readers based on X.Y and later.
+ * Codec support may vary between readers based on the format version and
+ * libraries available at runtime.
+ *
+ * See Compression.md for a detailed specification of these algorithms.
+ */
+enum CompressionCodec {
+ UNCOMPRESSED = 0;
+ SNAPPY = 1;
+ GZIP = 2;
+ LZO = 3;
+ BROTLI = 4; // Added in 2.4
+ LZ4 = 5; // DEPRECATED (Added in 2.4)
+ ZSTD = 6; // Added in 2.4
+ LZ4_RAW = 7; // Added in 2.9
+}
+
+enum PageType {
+ DATA_PAGE = 0;
+ INDEX_PAGE = 1;
+ DICTIONARY_PAGE = 2;
+ DATA_PAGE_V2 = 3;
+}
+
+/**
+ * Enum to annotate whether lists of min/max elements inside ColumnIndex
+ * are ordered and if so, in which direction.
+ */
+enum BoundaryOrder {
+ UNORDERED = 0;
+ ASCENDING = 1;
+ DESCENDING = 2;
+}
+
+/** Data page header */
+struct DataPageHeader {
+ /**
+ * Number of values, including NULLs, in this data page.
+ *
+ * If an OffsetIndex is present, a page must begin at a row
+ * boundary (repetition_level = 0). Otherwise, pages may begin
+ * within a row (repetition_level > 0).
+ **/
+ 1: required i32 num_values
+
+ /** Encoding used for this data page **/
+ 2: required Encoding encoding
+
+ /** Encoding used for definition levels **/
+ 3: required Encoding definition_level_encoding;
+
+ /** Encoding used for repetition levels **/
+ 4: required Encoding repetition_level_encoding;
+
+ /** Optional statistics for the data in this page **/
+ 5: optional Statistics statistics;
+}
+
+struct IndexPageHeader {
+ // TODO
+}
+
+/**
+ * The dictionary page must be placed at the first position of the column chunk
+ * if it is partly or completely dictionary encoded. At most one dictionary page
+ * can be placed in a column chunk.
+ **/
+struct DictionaryPageHeader {
+ /** Number of values in the dictionary **/
+ 1: required i32 num_values;
+
+ /** Encoding using this dictionary page **/
+ 2: required Encoding encoding
+
+ /** If true, the entries in the dictionary are sorted in ascending order **/
+ 3: optional bool is_sorted;
+}
+
+/**
+ * Alternate page format allowing reading levels without decompressing the data
+ * Repetition and definition levels are uncompressed
+ * The remaining section containing the data is compressed if is_compressed is true
+ *
+ * Implementation note - this header is not necessarily a strict improvement over
+ * `DataPageHeader` (in particular the original header might provide better compression
+ * in some scenarios). Page indexes require pages to start and end at row boundaries,
+ * regardless of which page header is used.
+ **/
+struct DataPageHeaderV2 {
+ /** Number of values, including NULLs, in this data page. **/
+ 1: required i32 num_values
+ /** Number of NULL values, in this data page.
+ Number of non-null = num_values - num_nulls which is also the number of values in the data section **/
+ 2: required i32 num_nulls
+ /**
+ * Number of rows in this data page. Every page must begin at a
+ * row boundary (repetition_level = 0): rows must **not** be
+ * split across page boundaries when using V2 data pages.
+ **/
+ 3: required i32 num_rows
+ /** Encoding used for data in this page **/
+ 4: required Encoding encoding
+
+ // repetition levels and definition levels are always using RLE (without size in it)
+
+ /** Length of the definition levels */
+ 5: required i32 definition_levels_byte_length;
+ /** Length of the repetition levels */
+ 6: required i32 repetition_levels_byte_length;
+
+ /** Whether the values are compressed.
+ Which means the section of the page between
+ definition_levels_byte_length + repetition_levels_byte_length and compressed_page_size (included)
+ is compressed with the compression_codec.
+ If missing it is considered compressed */
+ 7: optional bool is_compressed = true;
+
+ /** Optional statistics for the data in this page **/
+ 8: optional Statistics statistics;
+}
+
+/** Block-based algorithm type annotation. **/
+struct SplitBlockAlgorithm {}
+/** The algorithm used in Bloom filter. **/
+union BloomFilterAlgorithm {
+ /** Block-based Bloom filter. **/
+ 1: SplitBlockAlgorithm BLOCK;
+}
+
+/** Hash strategy type annotation. xxHash is an extremely fast non-cryptographic hash
+ * algorithm. It uses 64 bits version of xxHash.
+ **/
+struct XxHash {}
+
+/**
+ * The hash function used in Bloom filter. This function takes the hash of a column value
+ * using plain encoding.
+ **/
+union BloomFilterHash {
+ /** xxHash Strategy. **/
+ 1: XxHash XXHASH;
+}
+
+/**
+ * The compression used in the Bloom filter.
+ **/
+struct Uncompressed {}
+union BloomFilterCompression {
+ 1: Uncompressed UNCOMPRESSED;
+}
+
+/**
+ * Bloom filter header is stored at beginning of Bloom filter data of each column
+ * and followed by its bitset.
+ **/
+struct BloomFilterHeader {
+ /** The size of bitset in bytes **/
+ 1: required i32 numBytes;
+ /** The algorithm for setting bits. **/
+ 2: required BloomFilterAlgorithm algorithm;
+ /** The hash function used for Bloom filter. **/
+ 3: required BloomFilterHash hash;
+ /** The compression used in the Bloom filter **/
+ 4: required BloomFilterCompression compression;
+}
+
+struct PageHeader {
+ /** the type of the page: indicates which of the *_header fields is set **/
+ 1: required PageType type
+
+ /** Uncompressed page size in bytes (not including this header) **/
+ 2: required i32 uncompressed_page_size
+
+ /** Compressed (and potentially encrypted) page size in bytes, not including this header **/
+ 3: required i32 compressed_page_size
+
+ /** The 32-bit CRC checksum for the page, to be calculated as follows:
+ *
+ * - The standard CRC32 algorithm is used (with polynomial 0x04C11DB7,
+ * the same as in e.g. GZIP).
+ * - All page types can have a CRC (v1 and v2 data pages, dictionary pages,
+ * etc.).
+ * - The CRC is computed on the serialization binary representation of the page
+ * (as written to disk), excluding the page header. For example, for v1
+ * data pages, the CRC is computed on the concatenation of repetition levels,
+ * definition levels and column values (optionally compressed, optionally
+ * encrypted).
+ * - The CRC computation therefore takes place after any compression
+ * and encryption steps, if any.
+ *
+ * If enabled, this allows for disabling checksumming in HDFS if only a few
+ * pages need to be read.
+ */
+ 4: optional i32 crc
+
+ // Headers for page specific data. One only will be set.
+ 5: optional DataPageHeader data_page_header;
+ 6: optional IndexPageHeader index_page_header;
+ 7: optional DictionaryPageHeader dictionary_page_header;
+ 8: optional DataPageHeaderV2 data_page_header_v2;
+}
+
+/**
+ * Wrapper struct to store key values
+ */
+ struct KeyValue {
+ 1: required string key
+ 2: optional string value
+}
+
+/**
+ * Sort order within a RowGroup of a leaf column
+ */
+struct SortingColumn {
+ /** The ordinal position of the column (in this row group) **/
+ 1: required i32 column_idx
+
+ /** If true, indicates this column is sorted in descending order. **/
+ 2: required bool descending
+
+ /** If true, nulls will come before non-null values, otherwise,
+ * nulls go at the end. */
+ 3: required bool nulls_first
+}
+
+/**
+ * statistics of a given page type and encoding
+ */
+struct PageEncodingStats {
+
+ /** the page type (data/dic/...) **/
+ 1: required PageType page_type;
+
+ /** encoding of the page **/
+ 2: required Encoding encoding;
+
+ /** number of pages of this type with this encoding **/
+ 3: required i32 count;
+
+}
+
+/**
+ * Description for column metadata
+ */
+struct ColumnMetaData {
+ /** Type of this column **/
+ 1: required Type type
+
+ /** Set of all encodings used for this column. The purpose is to validate
+ * whether we can decode those pages. **/
+ 2: required list encodings
+
+ /** Path in schema **/
+ 3: required list path_in_schema
+
+ /** Compression codec **/
+ 4: required CompressionCodec codec
+
+ /** Number of values in this column **/
+ 5: required i64 num_values
+
+ /** total byte size of all uncompressed pages in this column chunk (including the headers) **/
+ 6: required i64 total_uncompressed_size
+
+ /** total byte size of all compressed, and potentially encrypted, pages
+ * in this column chunk (including the headers) **/
+ 7: required i64 total_compressed_size
+
+ /** Optional key/value metadata **/
+ 8: optional list key_value_metadata
+
+ /** Byte offset from beginning of file to first data page **/
+ 9: required i64 data_page_offset
+
+ /** Byte offset from beginning of file to root index page **/
+ 10: optional i64 index_page_offset
+
+ /** Byte offset from the beginning of file to first (only) dictionary page **/
+ 11: optional i64 dictionary_page_offset
+
+ /** optional statistics for this column chunk */
+ 12: optional Statistics statistics;
+
+ /** Set of all encodings used for pages in this column chunk.
+ * This information can be used to determine if all data pages are
+ * dictionary encoded for example **/
+ 13: optional list encoding_stats;
+
+ /** Byte offset from beginning of file to Bloom filter data. **/
+ 14: optional i64 bloom_filter_offset;
+
+ /** Size of Bloom filter data including the serialized header, in bytes.
+ * Added in 2.10 so readers may not read this field from old files and
+ * it can be obtained after the BloomFilterHeader has been deserialized.
+ * Writers should write this field so readers can read the bloom filter
+ * in a single I/O.
+ */
+ 15: optional i32 bloom_filter_length;
+
+ /**
+ * Optional statistics to help estimate total memory when converted to in-memory
+ * representations. The histograms contained in these statistics can
+ * also be useful in some cases for more fine-grained nullability/list length
+ * filter pushdown.
+ */
+ 16: optional SizeStatistics size_statistics;
+
+ /** Optional statistics specific for Geometry and Geography logical types */
+ 17: optional GeospatialStatistics geospatial_statistics;
+}
+
+struct EncryptionWithFooterKey {
+}
+
+struct EncryptionWithColumnKey {
+ /** Column path in schema **/
+ 1: required list path_in_schema
+
+ /** Retrieval metadata of column encryption key **/
+ 2: optional binary key_metadata
+}
+
+union ColumnCryptoMetaData {
+ 1: EncryptionWithFooterKey ENCRYPTION_WITH_FOOTER_KEY
+ 2: EncryptionWithColumnKey ENCRYPTION_WITH_COLUMN_KEY
+}
+
+struct ColumnChunk {
+ /** File where column data is stored. If not set, assumed to be same file as
+ * metadata. This path is relative to the current file.
+ *
+ * As of December 2025, the only known use-case for this field is writing summary
+ * parquet files (i.e. "_metadata" files). These files consolidate footers from
+ * multiple parquet files to allow for efficient reading of footers to avoid file
+ * listing costs and prune out files that do not need to be read based on statistics.
+ *
+ * These files do not appear to have ever been formally specified in the specification.
+ * and are potentially problematic from a correctness perspective [1].
+ *
+ * [1] https://lists.apache.org/thread/ootf2kmyg3p01b1bvplpvp4ftd1bt72d
+ *
+ * There is no other known usage of this field. Specifically, there are no known
+ * reference implementations that will read externally stored column data if this field is populated
+ * within a standard parquet file. Making use of the field for this purpose is
+ * not considered part of the Parquet specification.
+ **/
+ 1: optional string file_path
+
+ /** DEPRECATED: Byte offset in file_path to the ColumnMetaData
+ *
+ * Past use of this field has been inconsistent, with some implementations
+ * using it to point to the ColumnMetaData and some using it to point to
+ * the first page in the column chunk. In many cases, the ColumnMetaData at this
+ * location is wrong. This field is now deprecated and should not be used.
+ * Writers should set this field to 0 if no ColumnMetaData has been written outside
+ * the footer.
+ */
+ 2: required i64 file_offset = 0
+
+ /** Column metadata for this chunk. Some writers may also replicate this at the
+ * location pointed to by file_path/file_offset.
+ * Note: while marked as optional, this field is in fact required by most major
+ * Parquet implementations. As such, writers MUST populate this field.
+ **/
+ 3: optional ColumnMetaData meta_data
+
+ /** File offset of ColumnChunk's OffsetIndex **/
+ 4: optional i64 offset_index_offset
+
+ /** Size of ColumnChunk's OffsetIndex, in bytes **/
+ 5: optional i32 offset_index_length
+
+ /** File offset of ColumnChunk's ColumnIndex **/
+ 6: optional i64 column_index_offset
+
+ /** Size of ColumnChunk's ColumnIndex, in bytes **/
+ 7: optional i32 column_index_length
+
+ /** Crypto metadata of encrypted columns **/
+ 8: optional ColumnCryptoMetaData crypto_metadata
+
+ /** Encrypted column metadata for this chunk **/
+ 9: optional binary encrypted_column_metadata
+}
+
+struct RowGroup {
+ /** Metadata for each column chunk in this row group.
+ * This list must have the same order as the SchemaElement list in FileMetaData.
+ **/
+ 1: required list columns
+
+ /** Total byte size of all the uncompressed column data in this row group **/
+ 2: required i64 total_byte_size
+
+ /** Number of rows in this row group **/
+ 3: required i64 num_rows
+
+ /** If set, specifies a sort ordering of the rows in this RowGroup.
+ * The sorting columns can be a subset of all the columns.
+ */
+ 4: optional list sorting_columns
+
+ /** Byte offset from beginning of file to first page (data or dictionary)
+ * in this row group **/
+ 5: optional i64 file_offset
+
+ /** Total byte size of all compressed (and potentially encrypted) column data
+ * in this row group **/
+ 6: optional i64 total_compressed_size
+
+ /** Row group ordinal in the file **/
+ 7: optional i16 ordinal
+}
+
+/** Empty struct to signal the order defined by the physical or logical type */
+struct TypeDefinedOrder {}
+
+/** Empty struct to signal IEEE 754 total order for floating point types */
+struct IEEE754TotalOrder {}
+
+/**
+ * Union to specify the order used for the min_value and max_value fields for a
+ * column. This union takes the role of an enhanced enum that allows rich
+ * elements (which will be needed for a collation-based ordering in the future).
+ *
+ * Possible values are:
+ * * TypeDefinedOrder - the column uses the order defined by its logical or
+ * physical type (if there is no logical type).
+ * * IEEE754TotalOrder - the floating point column uses IEEE 754 total order.
+ *
+ * If the reader does not support the value of this union, min and max stats
+ * for this column should be ignored.
+ */
+union ColumnOrder {
+
+ /**
+ * The sort orders for logical types are:
+ * UTF8 - unsigned byte-wise comparison
+ * INT8 - signed comparison
+ * INT16 - signed comparison
+ * INT32 - signed comparison
+ * INT64 - signed comparison
+ * UINT8 - unsigned comparison
+ * UINT16 - unsigned comparison
+ * UINT32 - unsigned comparison
+ * UINT64 - unsigned comparison
+ * DECIMAL - signed comparison of the represented value
+ * DATE - signed comparison
+ * FLOAT16 - signed comparison of the represented value (*)
+ * TIME_MILLIS - signed comparison
+ * TIME_MICROS - signed comparison
+ * TIMESTAMP_MILLIS - signed comparison
+ * TIMESTAMP_MICROS - signed comparison
+ * INTERVAL - undefined
+ * JSON - unsigned byte-wise comparison
+ * BSON - unsigned byte-wise comparison
+ * ENUM - unsigned byte-wise comparison
+ * LIST - undefined
+ * MAP - undefined
+ * VARIANT - undefined
+ * GEOMETRY - undefined
+ * GEOGRAPHY - undefined
+ *
+ * In the absence of logical types, the sort order is determined by the physical type:
+ * BOOLEAN - false, true
+ * INT32 - signed comparison
+ * INT64 - signed comparison
+ * INT96 (only used for legacy timestamps) - undefined(+)
+ * FLOAT - signed comparison of the represented value (*)
+ * DOUBLE - signed comparison of the represented value (*)
+ * BYTE_ARRAY - unsigned byte-wise comparison
+ * FIXED_LEN_BYTE_ARRAY - unsigned byte-wise comparison
+ *
+ * (+) While the INT96 type has been deprecated, at the time of writing it is
+ * still used in many legacy systems. If a Parquet implementation chooses
+ * to write statistics for INT96 columns, it is recommended to order them
+ * according to the legacy rules:
+ * - compare the last 4 bytes (days) as a little-endian 32-bit signed integer
+ * - if equal last 4 bytes, compare the first 8 bytes as a little-endian
+ * 64-bit signed integer (nanos)
+ * See https://github.com/apache/parquet-format/issues/502 for more details
+ *
+ * (*) Because TYPE_ORDER is ambiguous for floating point types due to
+ * underspecified handling of NaN and -0/+0, it is recommended that writers
+ * use IEEE_754_TOTAL_ORDER for these types.
+ *
+ * If TYPE_ORDER is used for floating point types, then the following
+ * compatibility rules should be applied when reading statistics:
+ * - If the min is a NaN, it should be ignored.
+ * - If the max is a NaN, it should be ignored.
+ * - If the nan_count field is set, a reader can compute
+ * nan_count + null_count == num_values to deduce whether all non-null
+ * values are NaN.
+ * - If the min is +0, the row group may contain -0 values as well.
+ * - If the max is -0, the row group may contain +0 values as well.
+ * - When looking for NaN values, min and max should be ignored.
+ * If the nan_count field is set, it can be used to check whether
+ * NaNs are present.
+ *
+ * When writing page or column chunk statistics for columns with
+ * TYPE_ORDER order, the following rules must be followed:
+ * - The nan_count field must be set for floating point types, even if
+ * it is zero.
+ * - If the nan_count field is set, min and max statistics fields, when
+ * present, must not contain NaN values and must be computed from
+ * non-NaN values only. This signals to readers that the min and max
+ * statistics are reliable for non-NaN values.
+ * - If all non-null values are NaN, min and max statistics must not be
+ * written.
+ * - If the computed max value is zero (whether negative or positive),
+ * `+0.0` should be written into the max statistics field.
+ * - If the computed min value is zero (whether negative or positive),
+ * `-0.0` should be written into the min statistics field.
+ *
+ * When writing column indexes for columns with TYPE_ORDER order, the
+ * following rules must be followed:
+ * - NaNs must not be written to min_values or max_values.
+ * - If all non-null values of a page are NaN, a column index must not
+ * be written for this column chunk because min_values and max_values
+ * are required.
+ * - If the computed max value is zero (whether negative or positive),
+ * `+0.0` should be written into the corresponding max_values entry.
+ * - If the computed min value is zero (whether negative or positive),
+ * `-0.0` should be written into the corresponding min_values entry.
+ */
+ 1: TypeDefinedOrder TYPE_ORDER;
+
+ /*
+ * The floating point type is ordered according to the totalOrder predicate,
+ * as defined in section 5.10 of IEEE-754 (2008 revision). Only columns of
+ * physical type FLOAT or DOUBLE, or logical type FLOAT16 may use this ordering.
+ *
+ * Intuitively, this orders floats mathematically, but defines -0 to be less
+ * than +0, -NaN to be less than anything else, and +NaN to be greater than
+ * anything else. It also defines an order between different bit representations
+ * of the same value.
+ *
+ * When writing statistics for columns with IEEE_754_TOTAL_ORDER order, then
+ * following rules must be followed:
+ * - Writing the nan_count field is mandatory when using this ordering.
+ * - Min and max statistics must contain the smallest and largest non-NaN
+ * values respectively, or if all non-null values are NaN, the smallest and
+ * largest NaN values as defined by IEEE 754 total order.
+ *
+ * When reading statistics for columns with this order, the following rules
+ * should be followed:
+ * - Readers should consult the nan_count field to determine whether NaNs
+ * are present.
+ * - A reader can compute nan_count + null_count == num_values to deduce
+ * whether all non-null values are NaN. In the page index, which does not
+ * have a num_values field, the presence of a NaN value in min_values
+ * or max_values indicates that all non-null values are NaN.
+ */
+ 2: IEEE754TotalOrder IEEE_754_TOTAL_ORDER;
+}
+
+struct PageLocation {
+ /** Offset of the page in the file **/
+ 1: required i64 offset
+
+ /**
+ * Size of the page, including header. Equal to the sum of the page's
+ * PageHeader.compressed_page_size and the size of the serialized PageHeader.
+ */
+ 2: required i32 compressed_page_size
+
+ /**
+ * Index within the RowGroup of the first row of the page. When an
+ * OffsetIndex is present, pages must begin on row boundaries
+ * (repetition_level = 0).
+ */
+ 3: required i64 first_row_index
+}
+
+/**
+ * Optional offsets for each data page in a ColumnChunk.
+ *
+ * Forms part of the page index, along with ColumnIndex.
+ *
+ * OffsetIndex may be present even if ColumnIndex is not.
+ */
+struct OffsetIndex {
+ /**
+ * PageLocations, ordered by increasing PageLocation.offset. It is required
+ * that page_locations[i].first_row_index < page_locations[i+1].first_row_index.
+ */
+ 1: required list page_locations
+ /**
+ * Unencoded/uncompressed size for BYTE_ARRAY types.
+ *
+ * See documentation for unencoded_byte_array_data_bytes in SizeStatistics for
+ * more details on this field.
+ */
+ 2: optional list unencoded_byte_array_data_bytes
+}
+
+/**
+ * Optional statistics for each data page in a ColumnChunk.
+ *
+ * Forms part the page index, along with OffsetIndex.
+ *
+ * If this structure is present, OffsetIndex must also be present.
+ *
+ * For each field in this structure, [i] refers to the page at
+ * OffsetIndex.page_locations[i]
+ */
+struct ColumnIndex {
+ /**
+ * A list of Boolean values to determine the validity of the corresponding
+ * min and max values. If true, a page contains only null values, and writers
+ * have to set the corresponding entries in min_values and max_values to
+ * byte[0], so that all lists have the same length. If false, the
+ * corresponding entries in min_values and max_values must be valid.
+ */
+ 1: required list null_pages
+
+ /**
+ * Two lists containing lower and upper bounds for the values of each page
+ * determined by the ColumnOrder of the column. These may be the actual
+ * minimum and maximum values found on a page, but can also be (more compact)
+ * values that do not exist on a page. For example, instead of storing "Blart
+ * Versenwald III", a writer may set min_values[i]="B", max_values[i]="C".
+ * Such more compact values must still be valid values within the column's
+ * logical type. Readers must make sure that list entries are populated before
+ * using them by inspecting null_pages.
+ *
+ * For columns of physical type FLOAT or DOUBLE, or logical type FLOAT16,
+ * NaN values are not to be included in these bounds. If all non-null values
+ * of a page are NaN, then a writer must do the following:
+ * - If the order of this column is TYPE_ORDER, then a column index must
+ * not be written for this column chunk. While this is unfortunate for
+ * performance, it is necessary to avoid conflict with legacy files that
+ * still included NaN in min_values and max_values even if the page had
+ * non-NaN values. To mitigate this, IEEE754_TOTAL_ORDER is recommended.
+ * - If the order of this column is IEEE754_TOTAL_ORDER, then min_values[i]
+ * and max_values[i] of that page must be set to the smallest and largest
+ * NaN values as defined by IEEE 754 total order.
+ */
+ 2: required list min_values
+ 3: required list max_values
+
+ /**
+ * Stores whether both min_values and max_values are ordered and if so, in
+ * which direction. This allows readers to perform binary searches in both
+ * lists. Readers cannot assume that max_values[i] <= min_values[i+1], even
+ * if the lists are ordered.
+ */
+ 4: required BoundaryOrder boundary_order
+
+ /**
+ * A list containing the number of null values for each page
+ *
+ * Writers SHOULD always write this field even if no null values
+ * are present or the column is not nullable.
+ * Readers MUST distinguish between null_counts not being present
+ * and null_count being 0.
+ * If null_counts are not present, readers MUST NOT assume all
+ * null counts are 0.
+ */
+ 5: optional list null_counts
+
+ /**
+ * Contains repetition level histograms for each page
+ * concatenated together. The repetition_level_histogram field on
+ * SizeStatistics contains more details.
+ *
+ * When present the length should always be (number of pages *
+ * (max_repetition_level + 1)) elements.
+ *
+ * Element 0 is the first element of the histogram for the first page.
+ * Element (max_repetition_level + 1) is the first element of the histogram
+ * for the second page.
+ **/
+ 6: optional list repetition_level_histograms;
+ /**
+ * Same as repetition_level_histograms except for definitions levels.
+ **/
+ 7: optional list definition_level_histograms;
+
+ /**
+ * A list containing the number of NaN values for each page. Only present
+ * for columns of physical type FLOAT or DOUBLE, or logical type FLOAT16.
+ * If this field is not present, readers MUST assume that there might be
+ * NaN values in any page.
+ */
+ 8: optional list nan_counts
+
+}
+
+struct AesGcmV1 {
+ /** AAD prefix **/
+ 1: optional binary aad_prefix
+
+ /** Unique file identifier part of AAD suffix **/
+ 2: optional binary aad_file_unique
+
+ /** In files encrypted with AAD prefix without storing it,
+ * readers must supply the prefix **/
+ 3: optional bool supply_aad_prefix
+}
+
+struct AesGcmCtrV1 {
+ /** AAD prefix **/
+ 1: optional binary aad_prefix
+
+ /** Unique file identifier part of AAD suffix **/
+ 2: optional binary aad_file_unique
+
+ /** In files encrypted with AAD prefix without storing it,
+ * readers must supply the prefix **/
+ 3: optional bool supply_aad_prefix
+}
+
+union EncryptionAlgorithm {
+ 1: AesGcmV1 AES_GCM_V1
+ 2: AesGcmCtrV1 AES_GCM_CTR_V1
+}
+
+/**
+ * Description for file metadata
+ */
+struct FileMetaData {
+ /** Version of this file
+ *
+ * As of December 2025, there is no agreed upon consensus of what constitutes
+ * version 2 of the file. For maximum compatibility with readers, writers should
+ * always populate "1" for version. For maximum compatibility with writers,
+ * readers should accept "1" and "2" interchangeably. All other versions are
+ * reserved for potential future use-cases.
+ */
+ 1: required i32 version
+
+ /** Parquet schema for this file. This schema contains metadata for all the columns.
+ * The schema is represented as a tree with a single root. The nodes of the tree
+ * are flattened to a list by doing a depth-first traversal.
+ * The column metadata contains the path in the schema for that column which can be
+ * used to map columns to nodes in the schema.
+ * The first element is the root **/
+ 2: required list schema;
+
+ /** Number of rows in this file **/
+ 3: required i64 num_rows
+
+ /** Row groups in this file **/
+ 4: required list row_groups
+
+ /** Optional key/value metadata **/
+ 5: optional list key_value_metadata
+
+ /** String for application that wrote this file. This should be in the format
+ * version (build ).
+ * e.g. impala version 1.0 (build 6cf94d29b2b7115df4de2c06e2ab4326d721eb55)
+ **/
+ 6: optional string created_by
+
+ /**
+ * Sort order used for the min_value and max_value fields in the Statistics
+ * objects and the min_values and max_values fields in the ColumnIndex
+ * objects of each column in this file. Sort orders are listed in the order
+ * matching the columns in the schema. The indexes are not necessarily the same
+ * though, because only leaf nodes of the schema are represented in the list
+ * of sort orders.
+ *
+ * Without column_orders, the meaning of the min_value and max_value fields
+ * in the Statistics object and the ColumnIndex object is undefined. To ensure
+ * well-defined behaviour, if these fields are written to a Parquet file,
+ * column_orders must be written as well.
+ *
+ * The obsolete min and max fields in the Statistics object are always sorted
+ * by signed comparison regardless of column_orders.
+ */
+ 7: optional list column_orders;
+
+ /**
+ * Encryption algorithm. This field is set only in encrypted files
+ * with plaintext footer. Files with encrypted footer store algorithm id
+ * in FileCryptoMetaData structure.
+ */
+ 8: optional EncryptionAlgorithm encryption_algorithm
+
+ /**
+ * Retrieval metadata of key used for signing the footer.
+ * Used only in encrypted files with plaintext footer.
+ */
+ 9: optional binary footer_signing_key_metadata
+}
+
+/** Crypto metadata for files with encrypted footer **/
+struct FileCryptoMetaData {
+ /**
+ * Encryption algorithm. This field is only used for files
+ * with encrypted footer. Files with plaintext footer store algorithm id
+ * inside footer (FileMetaData structure).
+ */
+ 1: required EncryptionAlgorithm encryption_algorithm
+
+ /** Retrieval metadata of key used for encryption of footer,
+ * and (possibly) columns **/
+ 2: optional binary key_metadata
+}
diff --git a/pom.xml b/pom.xml
index 35d09f60d7..29ccaa1c3e 100644
--- a/pom.xml
+++ b/pom.xml
@@ -84,7 +84,6 @@
shaded.parquet
3.3.0
- 2.13.0
1.17.0
thrift
${thrift.executable}
@@ -577,6 +576,7 @@
thrift-${thrift.version}.tar.gz
**/dependency-reduced-pom.xml
**/*.rej
+ **/src/main/thrift/parquet-format.version