From 006ea5be325f2bc90cca6ed074e85f20dbd0544f Mon Sep 17 00:00:00 2001 From: Krisztian Szucs Date: Mon, 5 Oct 2026 21:15:04 +0200 Subject: [PATCH] GH-3817: Content defined chunking of data pages Add opt-in content defined chunking (CDC) to the writer. Data page boundaries are chosen by a rolling gear hash over each column's (definition level, repetition level, value) stream instead of by position, so an insert or delete only changes the pages around it and files deduplicate in content addressable storage. The chunker is a port of Arrow C++ (parquet/chunker_internal) and arrow-rs (column/chunker/cdc.rs): same gear tables, mask derivation, defaults and cut rules. Each column keeps one chunker for the whole file, as arrow-rs does, so chunk boundaries continue across row groups instead of restarting in each. Arrow C++ builds a new chunker for every row group, contrary to its own documentation, so its chunks match ours in the first row group but generally not after it. - CdcOptions: min/max chunk size and normalization level, defaulting to 256 KiB / 1 MiB / 0 as in Arrow. - ParquetProperties and ParquetWriter.Builder: withContentDefinedChunkingEnabled(boolean) and withContentDefinedChunking(CdcOptions). - ParquetOutputFormat: parquet.page.content-defined-chunking.* keys. - ChunkingColumnWriter wraps the column writer only when chunking is enabled. It ends pages at chunk boundaries and, as Arrow C++ does, applies the page size and row count limits counted from the page start, so they cut a chunk in the same places after an edit. - ParquetProperties holds the chunkers, so every column write store made with them continues the chunking. - With chunking on, a dictionary falls back to plain only when it outgrows its size limit, as in Arrow C++, not on its first page, which chunking cuts by content. Pages of nulls alone get an empty dictionary page, as Arrow C++ writes one. - CdcWriteBenchmarks measures the write cost. Off by default, and with it off the writer behaves as before. --- .../benchmarks/CdcWriteBenchmarks.java | 159 +++++ .../org/apache/parquet/column/CdcOptions.java | 157 +++++ .../parquet/column/ParquetProperties.java | 79 +++ .../parquet/column/impl/CdcChunker.java | 183 ++++++ .../parquet/column/impl/CdcChunkers.java | 38 ++ .../column/impl/ChunkingColumnWriter.java | 135 +++++ .../column/impl/ColumnWriteStoreBase.java | 26 +- .../parquet/column/impl/ColumnWriterBase.java | 4 + .../parquet/column/impl/GearHashTable.java | 571 ++++++++++++++++++ .../dictionary/DictionaryValuesWriter.java | 25 +- .../factory/DefaultValuesWriterFactory.java | 8 +- .../values/fallback/FallbackValuesWriter.java | 19 +- .../column/chunking/RollingHashMask.java | 76 +++ .../apache/parquet/column/TestCdcOptions.java | 72 +++ .../parquet/column/TestParquetProperties.java | 45 ++ .../column/impl/ChunkingTestSupport.java | 158 +++++ .../parquet/column/impl/TestCdcChunker.java | 171 ++++++ .../impl/TestCdcChunkerGoldenBoundaries.java | 233 +++++++ .../parquet/column/impl/TestCdcWrite.java | 500 +++++++++++++++ .../column/impl/TestGearHashTable.java | 54 ++ .../values/dictionary/TestDictionary.java | 67 ++ .../column/chunking/TestRollingHashMask.java | 112 ++++ parquet-hadoop/README.md | 36 ++ .../parquet/hadoop/ParquetOutputFormat.java | 23 + .../apache/parquet/hadoop/ParquetWriter.java | 25 + .../apache/parquet/hadoop/TestCdcWriter.java | 478 +++++++++++++++ 26 files changed, 3439 insertions(+), 15 deletions(-) create mode 100644 parquet-benchmarks/src/main/java/org/apache/parquet/benchmarks/CdcWriteBenchmarks.java create mode 100644 parquet-column/src/main/java/org/apache/parquet/column/CdcOptions.java create mode 100644 parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunker.java create mode 100644 parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunkers.java create mode 100644 parquet-column/src/main/java/org/apache/parquet/column/impl/ChunkingColumnWriter.java create mode 100644 parquet-column/src/main/java/org/apache/parquet/column/impl/GearHashTable.java create mode 100644 parquet-column/src/main/java/org/apache/parquet/internal/column/chunking/RollingHashMask.java create mode 100644 parquet-column/src/test/java/org/apache/parquet/column/TestCdcOptions.java create mode 100644 parquet-column/src/test/java/org/apache/parquet/column/impl/ChunkingTestSupport.java create mode 100644 parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunker.java create mode 100644 parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunkerGoldenBoundaries.java create mode 100644 parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcWrite.java create mode 100644 parquet-column/src/test/java/org/apache/parquet/column/impl/TestGearHashTable.java create mode 100644 parquet-column/src/test/java/org/apache/parquet/internal/column/chunking/TestRollingHashMask.java create mode 100644 parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestCdcWriter.java diff --git a/parquet-benchmarks/src/main/java/org/apache/parquet/benchmarks/CdcWriteBenchmarks.java b/parquet-benchmarks/src/main/java/org/apache/parquet/benchmarks/CdcWriteBenchmarks.java new file mode 100644 index 0000000000..51e839ec48 --- /dev/null +++ b/parquet-benchmarks/src/main/java/org/apache/parquet/benchmarks/CdcWriteBenchmarks.java @@ -0,0 +1,159 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.benchmarks; + +import java.io.IOException; +import java.util.ArrayList; +import java.util.List; +import java.util.Random; +import java.util.concurrent.TimeUnit; +import org.apache.parquet.example.data.Group; +import org.apache.parquet.example.data.simple.SimpleGroupFactory; +import org.apache.parquet.hadoop.ParquetFileWriter; +import org.apache.parquet.hadoop.ParquetWriter; +import org.apache.parquet.hadoop.example.ExampleParquetWriter; +import org.apache.parquet.hadoop.metadata.CompressionCodecName; +import org.apache.parquet.io.api.Binary; +import org.apache.parquet.schema.MessageType; +import org.apache.parquet.schema.MessageTypeParser; +import org.openjdk.jmh.annotations.Benchmark; +import org.openjdk.jmh.annotations.BenchmarkMode; +import org.openjdk.jmh.annotations.Fork; +import org.openjdk.jmh.annotations.Level; +import org.openjdk.jmh.annotations.Measurement; +import org.openjdk.jmh.annotations.Mode; +import org.openjdk.jmh.annotations.OutputTimeUnit; +import org.openjdk.jmh.annotations.Param; +import org.openjdk.jmh.annotations.Scope; +import org.openjdk.jmh.annotations.Setup; +import org.openjdk.jmh.annotations.State; +import org.openjdk.jmh.annotations.Warmup; + +/** + * The write cost of content defined chunking: the same rows written with and without it, for rows of + * several shapes, since the cost is the hashing of every value's and level's bytes. + * + *

Both variants lift the page row count limit and turn dictionary encoding off: chunking changes + * where pages end and when a dictionary falls back, either of which would dwarf the hashing. Rows are + * built once and written to {@link BlackHoleOutputFile}, so neither data generation nor I/O is + * measured. The cost while disabled is {@link WriteBenchmarks} compared across revisions. + */ +@BenchmarkMode(Mode.AverageTime) +@Fork(1) +@Warmup(iterations = 3) +@Measurement(iterations = 5) +@OutputTimeUnit(TimeUnit.MILLISECONDS) +@State(Scope.Thread) +public class CdcWriteBenchmarks { + + private static final int ROW_COUNT = 100_000; + + @Param({"false", "true"}) + public boolean chunking; + + /** + * mixed: a long, a 128-byte binary and a list of eight ints; numbers: an int, a long and a nullable + * double; strings: short strings, one of them nullable; lists: nullable lists of zero to eight + * nullable longs. + */ + @Param({"mixed", "numbers", "strings", "lists"}) + public String data; + + private MessageType schema; + private List rows; + + @Setup(Level.Trial) + public void setup() { + Random random = new Random(TestDataFactory.DEFAULT_SEED); + switch (data) { + case "mixed": + schema = MessageTypeParser.parseMessageType( + "message m { required int64 l; required binary b; required group g { repeated int32 i; } }"); + break; + case "numbers": + schema = MessageTypeParser.parseMessageType( + "message m { required int32 i; required int64 l; optional double d; }"); + break; + case "strings": + schema = MessageTypeParser.parseMessageType( + "message m { required binary s (STRING); optional binary t (STRING); }"); + break; + case "lists": + schema = MessageTypeParser.parseMessageType( + "message m { optional group l (LIST) { repeated group list { optional int64 element; } } }"); + break; + default: + throw new IllegalArgumentException("unknown data " + data); + } + Binary[] binaries = TestDataFactory.generateBinaryData(ROW_COUNT, 128, 0, TestDataFactory.DEFAULT_SEED); + SimpleGroupFactory factory = new SimpleGroupFactory(schema); + rows = new ArrayList<>(ROW_COUNT); + for (int i = 0; i < ROW_COUNT; i++) { + Group row = factory.newGroup(); + switch (data) { + case "mixed": + row.append("l", (long) i).append("b", binaries[i]); + Group g = row.addGroup("g"); + for (int j = 0; j < 8; j++) { + g.append("i", random.nextInt()); + } + break; + case "numbers": + row.append("i", random.nextInt()).append("l", random.nextLong()); + if (random.nextInt(10) > 0) { + row.append("d", random.nextDouble()); + } + break; + case "strings": + row.append("s", "s" + random.nextInt(1_000_000)); + if (random.nextInt(10) > 0) { + row.append("t", Long.toString(random.nextLong(), 36)); + } + break; + default: + if (random.nextInt(10) > 0) { + Group list = row.addGroup("l"); + for (int n = random.nextInt(9); n > 0; n--) { + Group element = list.addGroup("list"); + if (random.nextInt(10) > 0) { + element.append("element", random.nextLong()); + } + } + } + } + rows.add(row); + } + } + + @Benchmark + public void write() throws IOException { + try (ParquetWriter writer = ExampleParquetWriter.builder(BlackHoleOutputFile.INSTANCE) + .withWriteMode(ParquetFileWriter.Mode.OVERWRITE) + .withType(schema) + .withCompressionCodec(CompressionCodecName.UNCOMPRESSED) + .withPageRowCountLimit(Integer.MAX_VALUE) + .withDictionaryEncoding(false) + .withContentDefinedChunkingEnabled(chunking) + .build()) { + for (Group row : rows) { + writer.write(row); + } + } + } +} diff --git a/parquet-column/src/main/java/org/apache/parquet/column/CdcOptions.java b/parquet-column/src/main/java/org/apache/parquet/column/CdcOptions.java new file mode 100644 index 0000000000..b074880bae --- /dev/null +++ b/parquet-column/src/main/java/org/apache/parquet/column/CdcOptions.java @@ -0,0 +1,157 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column; + +import org.apache.parquet.Preconditions; +import org.apache.parquet.internal.column.chunking.RollingHashMask; + +/** + * EXPERIMENTAL: The size envelope and normalization level of content defined chunking (CDC), which + * ends data pages at boundaries derived from the column's values, so that files sharing a run of + * values share byte-identical pages. Sizes are measured on values and levels before encoding. + * Dictionary ids follow the order values first appear in, so an edit that adds new values renumbers + * the later ones within its row group: columns of many distinct values deduplicate best without + * dictionary encoding. + * + *

The settings and defaults match {@code CdcOptions} in Arrow C++ and arrow-rs, and so do the + * chunk boundaries for the same physical values. Arrow types converted before writing, such as + * narrow integers, coerced timestamps and decimals, are hashed differently there. Chunk boundaries + * continue across row groups, as in arrow-rs; Arrow C++ restarts them in every row group, so its + * chunks match only in the first. + * + * @see ParquetProperties.Builder#withContentDefinedChunking(CdcOptions) + */ +public final class CdcOptions { + + /** 256 KiB minimum, 1 MiB maximum and normalization level 0. */ + public static final CdcOptions DEFAULT = builder().build(); + + private final long minChunkSize; + private final long maxChunkSize; + private final int normLevel; + + private CdcOptions(Builder builder) { + this.minChunkSize = builder.minChunkSize; + this.maxChunkSize = builder.maxChunkSize; + this.normLevel = builder.normLevel; + // Validate now rather than when the first column writer is built. + RollingHashMask.calculate(minChunkSize, maxChunkSize, normLevel); + } + + /** + * @return the minimum chunk size in bytes + */ + public long getMinChunkSize() { + return minChunkSize; + } + + /** + * @return the maximum chunk size in bytes + */ + public long getMaxChunkSize() { + return maxChunkSize; + } + + /** + * @return the normalization level of the rolling hash mask + */ + public int getNormLevel() { + return normLevel; + } + + @Override + public String toString() { + return "CdcOptions{minChunkSize=" + minChunkSize + ", maxChunkSize=" + maxChunkSize + ", normLevel=" + normLevel + + '}'; + } + + /** + * @return a builder holding the default options + */ + public static Builder builder() { + return new Builder(); + } + + /** EXPERIMENTAL: Builds {@link CdcOptions}. */ + public static class Builder { + private long minChunkSize = 256 * 1024L; + private long maxChunkSize = 1024 * 1024L; + private int normLevel = 0; + + private Builder() {} + + /** + * Set the minimum chunk size in bytes, 256 KiB by default. The rolling hash is not updated + * until a chunk reaches this size, so no chunk is shorter but a file's last; pages can be, where + * a row group or a page limit ends one. + * + * @param minChunkSize the minimum chunk size in bytes + * @return this builder for method chaining + */ + public Builder withMinChunkSize(long minChunkSize) { + Preconditions.checkArgument( + minChunkSize >= 0, + "Invalid content defined chunking minimum chunk size (negative): %s", + minChunkSize); + this.minChunkSize = minChunkSize; + return this; + } + + /** + * Set the maximum chunk size in bytes, 1 MiB by default. A chunk ends when it reaches this size, + * whatever the rolling hash says. {@link ParquetProperties.Builder#withPageSize(int)} separately + * limits the page size; below this it splits chunks into more pages, in the same places after + * an edit, so it does not cost deduplication. + * + * @param maxChunkSize the maximum chunk size in bytes + * @return this builder for method chaining + */ + public Builder withMaxChunkSize(long maxChunkSize) { + Preconditions.checkArgument( + maxChunkSize > 0, + "Invalid content defined chunking maximum chunk size (not positive): %s", + maxChunkSize); + this.maxChunkSize = maxChunkSize; + return this; + } + + /** + * Set the normalization level of the rolling hash mask, 0 by default. Raising it makes a + * boundary more likely, which tightens the chunk size distribution and improves deduplication + * at the cost of more small pages; lowering it does the reverse. Values outside + * {@code [-3, 3]} are not useful. + * + * @param normLevel the normalization level + * @return this builder for method chaining + */ + public Builder withNormLevel(int normLevel) { + this.normLevel = normLevel; + return this; + } + + /** + * @return the options + * @throws IllegalArgumentException if the maximum chunk size is not greater than the minimum, + * or the envelope is too narrow for the normalization level + */ + public CdcOptions build() { + return new CdcOptions(this); + } + } +} diff --git a/parquet-column/src/main/java/org/apache/parquet/column/ParquetProperties.java b/parquet-column/src/main/java/org/apache/parquet/column/ParquetProperties.java index 8fe45e01ef..5556744843 100644 --- a/parquet-column/src/main/java/org/apache/parquet/column/ParquetProperties.java +++ b/parquet-column/src/main/java/org/apache/parquet/column/ParquetProperties.java @@ -29,6 +29,7 @@ import org.apache.parquet.bytes.ByteBufferAllocator; import org.apache.parquet.bytes.CapacityByteArrayOutputStream; import org.apache.parquet.bytes.HeapByteBufferAllocator; +import org.apache.parquet.column.impl.CdcChunkers; import org.apache.parquet.column.impl.ColumnWriteStoreV1; import org.apache.parquet.column.impl.ColumnWriteStoreV2; import org.apache.parquet.column.page.PageWriteStore; @@ -70,6 +71,8 @@ public class ParquetProperties { public static final boolean DEFAULT_PAGE_WRITE_CHECKSUM_ENABLED = true; + public static final boolean DEFAULT_CONTENT_DEFINED_CHUNKING_ENABLED = false; + /** * @deprecated This shared instance can cause thread safety issues when used by multiple builders concurrently. * Use {@code new DefaultValuesWriterFactory()} instead to create individual instances. @@ -138,6 +141,9 @@ public static WriterVersion fromString(String name) { private final ColumnProperty sizeStatistics; private final ColumnProperty columnCodecs; private final ColumnProperty columnCompressionLevels; + private final boolean cdcEnabled; + private final CdcOptions cdcOptions; + private final CdcChunkers cdcChunkers = new CdcChunkers(); private ParquetProperties(Builder builder) { this.pageSizeThreshold = builder.pageSize; @@ -172,6 +178,8 @@ private ParquetProperties(Builder builder) { this.sizeStatistics = builder.sizeStatistics.build(); this.columnCodecs = builder.columnCodecs.build(); this.columnCompressionLevels = builder.columnCompressionLevels.build(); + this.cdcEnabled = builder.cdcEnabled; + this.cdcOptions = builder.cdcOptions; } public static Builder builder() { @@ -319,6 +327,35 @@ public int getRowGroupRowCountLimit() { return rowGroupRowCountLimit; } + /** + * EXPERIMENTAL: Whether data page boundaries are derived from the content of the data. + * + * @return {@code true} if content defined chunking is enabled + */ + public boolean isContentDefinedChunkingEnabled() { + return cdcEnabled; + } + + /** + * EXPERIMENTAL: The content defined chunking options, which only apply while + * {@link #isContentDefinedChunkingEnabled()} is {@code true}. + * + * @return the chunking options, never {@code null} + */ + public CdcOptions getCdcOptions() { + return cdcOptions; + } + + /** + * Internal: the content defined chunking state, which every column write store made with these + * properties continues. + * + * @return the chunkers of the file written with these properties + */ + public CdcChunkers getCdcChunkers() { + return cdcChunkers; + } + public int getPageRowCountLimit() { return pageRowCountLimit; } @@ -413,6 +450,9 @@ public String toString() { + "Bloom filter expected number of distinct values are: " + bloomFilterNDVs + '\n' + "Bloom filter false positive probabilities are: " + bloomFilterFPPs + '\n' + "Page row count limit to " + getPageRowCountLimit() + '\n' + + "Content defined chunking is: " + + (cdcEnabled ? cdcOptions.toString() : "off") + + '\n' + "Writing page checksums is: " + (getPageWriteChecksumEnabled() ? "on" : "off") + '\n' + "Statistics enabled: " + statisticsEnabled + '\n' + "Size statistics enabled: " + sizeStatisticsEnabled; @@ -460,6 +500,8 @@ public static class Builder { private final ColumnProperty.Builder sizeStatistics; private final ColumnProperty.Builder columnCodecs; private final ColumnProperty.Builder columnCompressionLevels; + private boolean cdcEnabled = DEFAULT_CONTENT_DEFINED_CHUNKING_ENABLED; + private CdcOptions cdcOptions = CdcOptions.DEFAULT; private Builder() { enableDict = ColumnProperty.builder().withDefaultValue(DEFAULT_IS_DICTIONARY_ENABLED); @@ -511,6 +553,8 @@ private Builder(ParquetProperties toCopy) { this.sizeStatisticsEnabled = toCopy.sizeStatisticsEnabled; this.columnCodecs = ColumnProperty.builder(toCopy.columnCodecs); this.columnCompressionLevels = ColumnProperty.builder(toCopy.columnCompressionLevels); + this.cdcEnabled = toCopy.cdcEnabled; + this.cdcOptions = toCopy.cdcOptions; } /** @@ -755,6 +799,41 @@ public Builder withPageRowCountLimit(int rowCount) { return this; } + /** + * EXPERIMENTAL: Enable or disable content defined chunking of data pages, using + * {@link CdcOptions#DEFAULT} unless {@link #withContentDefinedChunking(CdcOptions)} sets others. + * Disabled by default. + * + *

As in Arrow C++, {@link #withPageSize(int)} and {@link #withPageRowCountLimit(int)} still + * cut pages inside a chunk, counted from the page start, so they add pages without moving any + * after an edit; and a column falls back from dictionary encoding only when its dictionary + * outgrows {@link #withDictionaryPageSize(int)}, not on its first page. + * + *

The chunking state lives in the built properties and continues across every column write + * store made with them, so chunk boundaries carry over row groups: build them for each file, and + * use them for one file at a time. + * + * @param enabled whether to derive data page boundaries from the content + * @return this builder for method chaining. + */ + public Builder withContentDefinedChunkingEnabled(boolean enabled) { + this.cdcEnabled = enabled; + return this; + } + + /** + * EXPERIMENTAL: Enable content defined chunking with the given options. A later + * {@code withContentDefinedChunkingEnabled(false)} disables it again. + * + * @param options the chunking options + * @return this builder for method chaining. + */ + public Builder withContentDefinedChunking(CdcOptions options) { + this.cdcOptions = Objects.requireNonNull(options, "CdcOptions cannot be null"); + this.cdcEnabled = true; + return this; + } + public Builder withPageWriteChecksumEnabled(boolean val) { this.pageWriteChecksumEnabled = val; return this; diff --git a/parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunker.java b/parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunker.java new file mode 100644 index 0000000000..45cbb28e3b --- /dev/null +++ b/parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunker.java @@ -0,0 +1,183 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import java.nio.ByteBuffer; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.internal.column.chunking.RollingHashMask; +import org.apache.parquet.io.api.Binary; + +/** + * Decides content defined data page boundaries for one column. + * + *

The column writer offers every {@code (definitionLevel, repetitionLevel, value)} triplet, and + * each {@code offer} says whether a page should end before it. + * + *

Ported from Arrow C++ {@code parquet/chunker_internal} and arrow-rs + * {@code column/chunker/cdc.rs}. Pages only deduplicate across implementations while all of them + * place the same boundaries, which {@code TestCdcChunkerGoldenBoundaries} pins against Arrow C++. + * + *

Not thread safe. One instance serves one leaf column for a whole file: it is never reset, at a + * page or a row group boundary, but carried to the next row group's store. + */ +final class CdcChunker { + + private final long minChunkSize; + private final long maxChunkSize; + private final long mask; + private final int maxDef; + private final int maxRep; + + private long rollingHash; + private boolean hasMatched; + private int nthRun; + private long chunkSize; + + CdcChunker(CdcOptions options, ColumnDescriptor path) { + this.minChunkSize = options.getMinChunkSize(); + this.maxChunkSize = options.getMaxChunkSize(); + this.mask = RollingHashMask.calculate(minChunkSize, maxChunkSize, options.getNormLevel()); + this.maxDef = path.getMaxDefinitionLevel(); + this.maxRep = path.getMaxRepetitionLevel(); + } + + boolean offer(int value, int repetitionLevel, int definitionLevel) { + return offerFixedWidth(value, 4, repetitionLevel, definitionLevel); + } + + boolean offer(long value, int repetitionLevel, int definitionLevel) { + return offerFixedWidth(value, 8, repetitionLevel, definitionLevel); + } + + boolean offer(float value, int repetitionLevel, int definitionLevel) { + // Raw bits: floatToIntBits canonicalizes NaN payloads and would diverge from the references. + return offerFixedWidth(Float.floatToRawIntBits(value), 4, repetitionLevel, definitionLevel); + } + + boolean offer(double value, int repetitionLevel, int definitionLevel) { + return offerFixedWidth(Double.doubleToRawLongBits(value), 8, repetitionLevel, definitionLevel); + } + + boolean offer(boolean value, int repetitionLevel, int definitionLevel) { + // One byte per boolean, not one bit, as the references hash it. + return offerFixedWidth(value ? 1 : 0, 1, repetitionLevel, definitionLevel); + } + + private boolean offerFixedWidth(long bits, int width, int repetitionLevel, int definitionLevel) { + rollLevels(repetitionLevel, definitionLevel); + rollFixedWidth(bits, width); + return endsPageHere(repetitionLevel); + } + + boolean offer(Binary value, int repetitionLevel, int definitionLevel) { + rollLevels(repetitionLevel, definitionLevel); + rollBinary(value); + return endsPageHere(repetitionLevel); + } + + /** + * Rolls the levels only, even when {@code definitionLevel == maxDef}: an omitted top-level + * required field arrives here as {@code (0, 0)} and has no value to hash. + */ + boolean offerNull(int repetitionLevel, int definitionLevel) { + rollLevels(repetitionLevel, definitionLevel); + return endsPageHere(repetitionLevel); + } + + /** + * Definition level first, and each level as two bytes: the references hash the levels as int16_t + * in this order. + */ + private void rollLevels(int repetitionLevel, int definitionLevel) { + if (maxDef > 0) { + rollFixedWidth(definitionLevel, 2); + } + if (maxRep > 0) { + rollFixedWidth(repetitionLevel, 2); + } + } + + /** + * Asks {@link #needNewChunk()} only at a record start, because asking consumes a match; a match + * inside a record carries over to the next record start. + */ + private boolean endsPageHere(int repetitionLevel) { + return repetitionLevel == 0 && needNewChunk(); + } + + /** The value's bytes without a length prefix, as the references hash them. */ + private void rollBinary(Binary value) { + chunkSize += value.length(); + if (chunkSize < minChunkSize) { + return; + } + ByteBuffer bytes = value.toByteBuffer(); + long hash = rollingHash; + boolean matched = hasMatched; + long[] table = GearHashTable.TABLE[nthRun]; + for (int i = bytes.position(); i < bytes.limit(); ++i) { + hash = (hash << 1) + table[bytes.get(i) & 0xFF]; + matched |= (hash & mask) == 0; + } + rollingHash = hash; + hasMatched = matched; + } + + /** + * The {@code width} low-order bytes of {@code bits}, little-endian. Like the references, this + * checks the skip window once per value rather than per byte. + */ + private void rollFixedWidth(long bits, int width) { + chunkSize += width; + if (chunkSize < minChunkSize) { + return; + } + long hash = rollingHash; + boolean matched = hasMatched; + long[] table = GearHashTable.TABLE[nthRun]; + for (int i = 0; i < width; ++i) { + hash = (hash << 1) + table[(int) ((bits >>> (8 * i)) & 0xFF)]; + matched |= (hash & mask) == 0; + } + rollingHash = hash; + hasMatched = matched; + } + + /** + * A chunk ends after eight matches, each against the next gear hash table, which approximates a + * normal chunk size distribution; or at {@code maxChunkSize}. As in the references, neither + * resets the rolling hash, and the maximum size cut leaves the run counter alone. + */ + private boolean needNewChunk() { + if (hasMatched) { + hasMatched = false; + if (++nthRun >= GearHashTable.TABLE.length) { + nthRun = 0; + chunkSize = 0; + return true; + } + } + if (chunkSize >= maxChunkSize) { + chunkSize = 0; + return true; + } + return false; + } +} diff --git a/parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunkers.java b/parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunkers.java new file mode 100644 index 0000000000..d721cb4d95 --- /dev/null +++ b/parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunkers.java @@ -0,0 +1,38 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import java.util.HashMap; +import java.util.Map; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; + +/** + * Internal: the content defined chunkers of one {@link org.apache.parquet.column.ParquetProperties}, + * one per leaf column. Every store made with those properties takes its chunkers from here, so chunk + * boundaries continue across row groups instead of restarting in each, as in arrow-rs. + */ +public final class CdcChunkers { + + private final Map chunkers = new HashMap<>(); + + CdcChunker chunker(ColumnDescriptor path, CdcOptions options) { + return chunkers.computeIfAbsent(path, p -> new CdcChunker(options, p)); + } +} diff --git a/parquet-column/src/main/java/org/apache/parquet/column/impl/ChunkingColumnWriter.java b/parquet-column/src/main/java/org/apache/parquet/column/impl/ChunkingColumnWriter.java new file mode 100644 index 0000000000..496e030607 --- /dev/null +++ b/parquet-column/src/main/java/org/apache/parquet/column/impl/ChunkingColumnWriter.java @@ -0,0 +1,135 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import org.apache.parquet.column.ColumnWriter; +import org.apache.parquet.column.ParquetProperties; +import org.apache.parquet.io.api.Binary; + +/** + * Ends the wrapped writer's page wherever its {@link CdcChunker} places a boundary, and applies the + * page limits the way Arrow C++ does under content defined chunking: measured from the page start + * rather than on the store's schedule, so that they cut a chunk in the same places after an edit. + * + *

A decorator rather than a change to {@link ColumnWriterBase}, so that a writer with content + * defined chunking disabled runs exactly the code it always has. + */ +final class ChunkingColumnWriter implements ColumnWriter { + + // Arrow's default write_batch_size: the size limits are checked once per batch of this many + // levels, counted from the page start, as Arrow C++ and arrow-rs check them. + private static final int BATCH_SIZE = 1024; + + private final ColumnWriterBase writer; + private final CdcChunker chunker; + private final int pageSizeThreshold; + private final int pageValueCountThreshold; + private final int pageRowCountLimit; + + private int levelsInBatch; + + ChunkingColumnWriter(ColumnWriterBase writer, CdcChunker chunker, ParquetProperties props) { + this.writer = writer; + this.chunker = chunker; + this.pageSizeThreshold = props.getPageSizeThreshold(); + this.pageValueCountThreshold = props.getPageValueCountThreshold(); + this.pageRowCountLimit = props.getPageRowCountLimit(); + } + + @Override + public void write(int value, int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offer(value, repetitionLevel, definitionLevel), repetitionLevel); + writer.write(value, repetitionLevel, definitionLevel); + } + + @Override + public void write(long value, int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offer(value, repetitionLevel, definitionLevel), repetitionLevel); + writer.write(value, repetitionLevel, definitionLevel); + } + + @Override + public void write(boolean value, int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offer(value, repetitionLevel, definitionLevel), repetitionLevel); + writer.write(value, repetitionLevel, definitionLevel); + } + + @Override + public void write(Binary value, int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offer(value, repetitionLevel, definitionLevel), repetitionLevel); + writer.write(value, repetitionLevel, definitionLevel); + } + + @Override + public void write(float value, int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offer(value, repetitionLevel, definitionLevel), repetitionLevel); + writer.write(value, repetitionLevel, definitionLevel); + } + + @Override + public void write(double value, int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offer(value, repetitionLevel, definitionLevel), repetitionLevel); + writer.write(value, repetitionLevel, definitionLevel); + } + + @Override + public void writeNull(int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offerNull(repetitionLevel, definitionLevel), repetitionLevel); + writer.writeNull(repetitionLevel, definitionLevel); + } + + @Override + public void close() { + writer.close(); + } + + @Override + public long getBufferedSizeInMemory() { + return writer.getBufferedSizeInMemory(); + } + + /** + * Ends the page before this triplet at a chunk boundary or where a page limit is reached. Pages + * only end at a record start: the row count limit at any, the size limits at the first once a + * batch is full. + */ + private void beforeTriplet(boolean chunkBoundary, int repetitionLevel) { + if (repetitionLevel == 0) { + if (chunkBoundary || writer.getPageRowCount() >= pageRowCountLimit) { + endPage(); + } else if (levelsInBatch >= BATCH_SIZE) { + if (writer.getCurrentPageBufferedSize() >= pageSizeThreshold + || writer.getValueCount() >= pageValueCountThreshold) { + endPage(); + } else { + levelsInBatch = 0; + } + } + } + levelsInBatch++; + } + + /** A boundary on the first value of a page has no page to end. */ + private void endPage() { + if (writer.getValueCount() > 0) { + writer.writePage(); + } + levelsInBatch = 0; + } +} diff --git a/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriteStoreBase.java b/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriteStoreBase.java index 9bc7726491..08558d2dc6 100644 --- a/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriteStoreBase.java +++ b/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriteStoreBase.java @@ -23,6 +23,7 @@ import static java.util.Collections.unmodifiableMap; import java.util.Arrays; +import java.util.HashMap; import java.util.Map; import java.util.Map.Entry; import java.util.Set; @@ -44,7 +45,7 @@ abstract class ColumnWriteStoreBase implements ColumnWriteStore { // Used to support the deprecated workflow of ColumnWriteStoreV1 (lazy init of ColumnWriters) private interface ColumnWriterProvider { - ColumnWriter getColumnWriter(ColumnDescriptor path); + ColumnWriterBase getColumnWriter(ColumnDescriptor path); } private final ColumnWriterProvider columnWriterProvider; @@ -53,6 +54,8 @@ private interface ColumnWriterProvider { private static final float THRESHOLD_TOLERANCE_RATIO = 0.1f; // 10 % private final Map columns; + // Content defined chunking wraps each writer once, with a chunker that can outlive the store. + private final Map chunkingColumns = new HashMap<>(); private final ParquetProperties props; private final long thresholdTolerance; private long rowCount; @@ -71,7 +74,7 @@ private interface ColumnWriterProvider { columnWriterProvider = new ColumnWriterProvider() { @Override - public ColumnWriter getColumnWriter(ColumnDescriptor path) { + public ColumnWriterBase getColumnWriter(ColumnDescriptor path) { ColumnWriterBase column = columns.get(path); if (column == null) { column = createColumnWriterBase(path, pageWriteStore.getPageWriter(path), null, props); @@ -96,7 +99,7 @@ public ColumnWriter getColumnWriter(ColumnDescriptor path) { columnWriterProvider = new ColumnWriterProvider() { @Override - public ColumnWriter getColumnWriter(ColumnDescriptor path) { + public ColumnWriterBase getColumnWriter(ColumnDescriptor path) { return columns.get(path); } }; @@ -126,7 +129,7 @@ public ColumnWriter getColumnWriter(ColumnDescriptor path) { columnWriterProvider = new ColumnWriterProvider() { @Override - public ColumnWriter getColumnWriter(ColumnDescriptor path) { + public ColumnWriterBase getColumnWriter(ColumnDescriptor path) { return columns.get(path); } }; @@ -147,7 +150,13 @@ abstract ColumnWriterBase createColumnWriter( @Override public ColumnWriter getColumnWriter(ColumnDescriptor path) { - return columnWriterProvider.getColumnWriter(path); + ColumnWriterBase column = columnWriterProvider.getColumnWriter(path); + if (column == null || !props.isContentDefinedChunkingEnabled()) { + return column; + } + return chunkingColumns.computeIfAbsent( + path, + p -> new ChunkingColumnWriter(column, props.getCdcChunkers().chunker(p, props.getCdcOptions()), props)); } public Set getColumnDescriptors() { @@ -224,7 +233,12 @@ public void close() { public void endRecord() { ++rowCount; if (rowCount >= rowCountForNextSizeCheck) { - sizeCheck(); + if (props.isContentDefinedChunkingEnabled()) { + // ChunkingColumnWriter applies the page limits; the schedule only flushes the cached nulls. + rowCountForNextSizeCheck = rowCount + props.getMinRowCountForPageSizeCheck(); + } else { + sizeCheck(); + } } } diff --git a/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriterBase.java b/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriterBase.java index 408627404d..6ba49fa614 100644 --- a/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriterBase.java +++ b/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriterBase.java @@ -365,6 +365,10 @@ int getValueCount() { return this.valueCount; } + int getPageRowCount() { + return this.pageRowCount; + } + /** * Writes the current data to a new page in the page store */ diff --git a/parquet-column/src/main/java/org/apache/parquet/column/impl/GearHashTable.java b/parquet-column/src/main/java/org/apache/parquet/column/impl/GearHashTable.java new file mode 100644 index 0000000000..c0c961fd42 --- /dev/null +++ b/parquet-column/src/main/java/org/apache/parquet/column/impl/GearHashTable.java @@ -0,0 +1,571 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +/** + * The gear hash tables used by {@link CdcChunker}, copied from Arrow C++ + * ({@code cpp/src/parquet/chunker_internal_generated.h}) -- do not edit. + * + *

Eight tables of 256 64-bit values. Table {@code r}, byte {@code b} is {@code TABLE[r][b]}. + * arrow-rs ({@code parquet/src/column/chunker/cdc_generated.rs}) ships the same tables. + * {@code TestGearHashTable} pins them to the MD5 specification they were generated from. + */ +final class GearHashTable { + + private GearHashTable() {} + + static final long[][] TABLE = { + // seed = 0 + { + 0xf09f35a563783945L, 0x0dcc5b3bc5ae410aL, 0x63f1ea8d22554270L, 0xfbe5ee7bd05a7b61L, + 0x3f692ed5e9934abaL, 0xaab3755952250eb8L, 0xdefb168dc2888fa5L, 0x501b36f7c77a7d47L, + 0xd2fff45d1989642dL, 0x80217c1c600e30a6L, 0xb9469ee2e43df7acL, 0x3654b76a61999706L, + 0x6ea73dfe5de0c6b6L, 0xdfd662e1937a589dL, 0x0dbe0cc74b188a68L, 0xde45f4e6d73ffc6fL, + 0xcdf7a7759e70d87eL, 0x5d6a951b8d38c310L, 0xdc9423c3813fcf2cL, 0x25dc2976e167ffceL, + 0xc2555baa1d031c84L, 0x115bc3f2230a3ab6L, 0xd4b10260f350bedeL, 0xdfd3501ab447d723L, + 0x022e79217edaf167L, 0x1635e2255c5a7526L, 0xa0a750350cc77102L, 0xc027133e05d39f56L, + 0xd949459779cf0387L, 0xb92f1464f5c688c2L, 0xd9ac5f3e8b42f2f3L, 0xdf02bb6f5ecaac21L, + 0x8156f988fac7bfa4L, 0xe4580f97bede2ec8L, 0x44fe7d17a76fca32L, 0x885f59bd54c2014cL, + 0x435e63ec655ffae9L, 0x5ebc51930967b1f1L, 0x5428c2084ac29e47L, 0x9465938fec30e36bL, + 0xc7cb3de4977772cdL, 0x15692d7c201e8c3aL, 0x505ee65cdc4b17f4L, 0x7d9839a0a7aead6bL, + 0xeef5f5b6a0105291L, 0x76c2fb232ce7f5bfL, 0x5c13893c1c3ff3a9L, 0x65b6b547d4442f98L, + 0xb8ad7487c8c96fceL, 0x906bcf51c99974f8L, 0x2f56e48bb943a48cL, 0xbc9ab109f82d3a44L, + 0xcd5160cdc8c7e735L, 0xbe9acb9df3427732L, 0x386b91d477d7fadeL, 0x36be463621dd5af2L, + 0xcbe6a2faffd627a8L, 0x9c8fd528463a2f5aL, 0xb9b88c6bb802b184L, 0xb414b4e665c597c7L, + 0xbedb142568209556L, 0x5360d81c25429dceL, 0x63a69a960a952f37L, 0xc900d63899e1b503L, + 0x1abc63a8b37c7728L, 0xa8b3a8b6409080ebL, 0x495e391f662959f6L, 0xdf1e136f3e12229bL, + 0x33d5fc526b0dd38dL, 0x321221ae2abfac63L, 0x7fde18351fda7395L, 0xed79fe5c3a6aa4c3L, + 0x2dd6965a4867d8d4L, 0x54813ca20fe8799bL, 0x5d59ea6456465c39L, 0x0de0c294d1936b81L, + 0x4aaf0755002c588cL, 0x3530a1857ad04c6dL, 0xb8a64f4ce184442bL, 0xe0def10bceedfa17L, + 0x46e38d0a443757ecL, 0x9795a1c645ee16d7L, 0x7e531def245eac8aL, 0x683b25c43a0716cfL, + 0x884583d372da219dL, 0x5b06b62c910416e5L, 0x54b6902fbebd3dbeL, 0x931198d40a761a75L, + 0xead7d8e830013590L, 0x80b4d5dc99bfacedL, 0xf98272c8108a1ad2L, 0x1adce054289a0ec6L, + 0x7d53a1143c56b465L, 0x497fbe4f00c92b52L, 0x525e4cc2e81ebd69L, 0xc94478e0d5508ff6L, + 0xb8a5da83c196d07cL, 0x7667a921b65b0603L, 0xf236fabbdefe6cd1L, 0x53da978d19a92b98L, + 0xc604f6e97087124dL, 0x2cbd27221924b094L, 0x65cd1102c985b1d2L, 0x08c0755dc1a97eb4L, + 0x5e0419e921c0fef1L, 0x282d2c1196f84a29L, 0xe21117fcfc5793f7L, 0xcf4e985dc38e6c2eL, + 0xd521f4f264d55616L, 0xde69b04c485f2a10L, 0x59410e245305178aL, 0xceab1d477c943601L, + 0xa9805732d71ee5e9L, 0x054cd443896974f6L, 0xf2b517717a423a3eL, 0x09517937fa9fac95L, + 0x4938233e9ca871e3L, 0x9132cbaf56f83ec0L, 0x4703421ed1dd027dL, 0xfd9933f4e6f1ec4eL, + 0xf237c7fded2274a8L, 0xdf4616efe68cd7b4L, 0x5e46de0f39f0a380L, 0x3d41e0c6d8e095b0L, + 0xc5272f8a5bb2df09L, 0x68aa78e8301fb964L, 0xbf5b5b52c8e32ae0L, 0xbf28ed3df74bdcf7L, + 0xd6198f64c833815aL, 0x8cd99d2974267544L, 0xd90560ea4465ff2cL, 0x571d65ad7ad59261L, + 0x309453518baa367aL, 0xa60538377bc79fb2L, 0xace515da1ab4183cL, 0xf56d3c8d891d1c5bL, + 0x5b0d8370b59def49L, 0x775866ce7c83c762L, 0x3d76085695c8e18aL, 0xba064d1a9af1b114L, + 0xc84ef7cd7b98b521L, 0x90b9231681c2bc37L, 0x37e2b13e6f585b6bL, 0x1d0a34e55e0f369fL, + 0x86bb8019cf41447cL, 0x4b95c6ef55b3f71fL, 0x3b6ed1660732b310L, 0x617eee603d137f21L, + 0xf4f6278b464f3bbcL, 0xdfb763b720da205aL, 0x353478899b871cb7L, 0xe45fbbff574cc41eL, + 0x1a94b60847907d72L, 0xb10eef051eff67a5L, 0xf0e012ec6a284d40L, 0xcc1cd1a11b926d7cL, + 0xcf9d9c5453e19cadL, 0x270febcc0fc0e86bL, 0xd6567568778b781eL, 0x7323b98965eeb46bL, + 0xccecd374567086ffL, 0xef7b44bfc497a704L, 0xebc479c051a9f0a5L, 0xc9b7410e3e00a235L, + 0x1d084f7ecdf83dabL, 0xc8a9a97e33ba8ba3L, 0x8c75318f5b2350d6L, 0xaa3cd5d0c684bddaL, + 0xa81125fe0901bedfL, 0xf7bcd76020edfc93L, 0x834ee4c12e75874fL, 0xb2bb8a7beb44fa14L, + 0x32cd26f50a4f4e4dL, 0x0fc5817ca55d959aL, 0xd6e4ae2e3ae10718L, 0x074abdcceb8d6e38L, + 0xc0cc5f4f9b3a9c43L, 0x1115d364363595b2L, 0x69861db2eb19f2e8L, 0x59b8d804cf92bc67L, + 0x9bac9785e5e4b863L, 0x7fa0e17a41869561L, 0x10d3c9633f0c709cL, 0x534a03deee6bc44aL, + 0x73b1f7201257f581L, 0x46fd6a11e2e0706bL, 0x494abb554946e67aL, 0xb5d6da317864dc8eL, + 0x402ded9238f39687L, 0xd8fa37d2cbd6d290L, 0xcc818293fcb06791L, 0x6482ab344806cd4dL, + 0x0956e6ee9d8eb60bL, 0x01fee622d8465ac8L, 0xae7ece370cbd9c35L, 0x7ff09e937a177279L, + 0xa2c29ee7a33ca5f1L, 0x990e8dbee083923bL, 0x4a819b72f610863aL, 0xddecfad79d3f08beL, + 0x627372480fac20a7L, 0x802154d6eca2db4cL, 0x8fcf02e42f805e55L, 0x040a911ff8cea977L, + 0xbb544485bc64d0d4L, 0xaddde1aeb406d0fbL, 0xf6b35fae23dce66fL, 0xc07a9fb3645d2f9bL, + 0xccd113907e9c0fedL, 0xd17af369984fd213L, 0x9223823c59a083e7L, 0xe19d475606b81013L, + 0xe181ac116a90e57aL, 0x71f7b6258c6def4cL, 0x2246f34b45964f7cL, 0xd74aedaea2d31751L, + 0xb1add86e5dd305d1L, 0xeb9ba881f16d6471L, 0xef7600e036f5c6ffL, 0x1d50bc9735b8fb85L, + 0xe63942bd1f3e2969L, 0x9241ba9f8b3f4e72L, 0xee8bb2bca07d35b6L, 0x55cd55dab522654eL, + 0x94d0cfa7c1a6845dL, 0x02f9845d559884c3L, 0x8ce70ea21063b560L, 0xd70998028ef08b74L, + 0xdfdb5bbee310876bL, 0x4e21b2e348256d16L, 0xde007a981c13debcL, 0xe51950cbbddabfddL, + 0xd223301dbe9957c1L, 0x084b8634cc2cce4bL, 0x90e551378aa9d70cL, 0x833b533ac633e448L, + 0x7891e232882da57fL, 0xa1bf26f0163ce2b3L, 0xf33a0171eb9c68d5L, 0x2e7de18ca69b3fa2L, + 0x666fd6f175619199L, 0x1239d37edb5feb9fL, 0xfa9fc9382e61ff5cL, 0x3ca4ad427e3c126fL, + 0x37c6dd4c2c31ae6eL, 0x1f1bacb619d427b2L, 0x7dd09f5d10759afeL, 0xc8d941432327d733L, + 0x2b389ba25e1d43a7L, 0xa4e3030c3740ff21L, 0xcc56dae13fd37463L, 0x2481457c175b560fL, + 0x9deb35bde77c5c41L, 0x847aa6ea5549a0c3L, 0xcde01bb48b6e7f02L, 0x15a28844e64cb211L, + }, + // seed = 1 + { + 0xecfcba92fe5691a3L, 0x71377799fea34699L, 0xb284c9096fa614e5L, 0x54534170f40de6c8L, + 0xbbd804d45884fba3L, 0x44929a896388c8a1L, 0x79b712508e0fa3b1L, 0xeb53ab280af31054L, + 0x351ea23a6319da7aL, 0x2fbe55d9819d85a2L, 0x34f4b6568dcd28b1L, 0x8c94ea5e5d82967aL, + 0x09068d333a46d3c5L, 0x762ad4f64cb73381L, 0xd5c6db5ef0e22640L, 0x36d8ab5a36175680L, + 0xd41fe333cdc3525aL, 0xa1f51dbdf20ce781L, 0x1410a95e786c8be6L, 0x96b7499a670c2b41L, + 0x3912e1037835d893L, 0x272c5bd83e1e9115L, 0x2ea7f91cad82a0d6L, 0xcd10e85662ce9931L, + 0xedad49be8d5e8b74L, 0x7ccd8fe0f37d12bcL, 0xfac0482005eed593L, 0x4513991681f6c8b0L, + 0x2804d612eb0ad37dL, 0x7cca9e8412b81d34L, 0x85ffd6707192b7b8L, 0xea0560aeea954411L, + 0x0122d28226102bbaL, 0xf51c47cdbd22fdd1L, 0x3707d851183ff17cL, 0xaef5a1465f3e902dL, + 0xbcb38c2d8736a04fL, 0x4025317e864bef15L, 0x8d3f66d86e1ea58fL, 0xc16759a3d97ed79aL, + 0x1c62abdc0659f2f5L, 0x23b3eb4e699bd28fL, 0x5083c4fceed3ccafL, 0xa65bf34562cc989cL, + 0xaa5865932fd79064L, 0xf24d08d268c24593L, 0x7fbd00a215196999L, 0x7812cd366d752964L, + 0x62e8dcb27ef3d945L, 0xf08b7984e1b946dcL, 0x547d23ad9a5c1dcfL, 0x496b1fb249b27fb7L, + 0xcd692e1db5f3b3baL, 0x41931e39f1e1bc61L, 0x286c6a7d7edae82bL, 0x17ef6638b6c4ca6eL, + 0x609beb5a2576a934L, 0xcc5e16fe4a69b83cL, 0xbbd14d08b078fc24L, 0x2a617680f481cb94L, + 0x81dbbd5f86e6d039L, 0xeb8205e1fc8ecc3cL, 0xe5e3bb576faa8042L, 0x5d6f1eb9d9df01b5L, + 0x9a47b8739c10fb44L, 0x398a7caad7ea7696L, 0x9c0fc1d7c46adde6L, 0x67cd6de0a51978a6L, + 0x68ccc4b77a21cca4L, 0x1e067066b82f415cL, 0xf7ddade6535e1819L, 0xf2185c884291751bL, + 0xc322b7381fcbe34fL, 0x242f593e88290b9bL, 0x8e11ccc0ea5e84a3L, 0x40e3a2e3346db8a2L, + 0xf18bfc3ad2931a2cL, 0x2468397394b00144L, 0xeae199cce14e6817L, 0x05b462686c75a1aeL, + 0xda096cb859c51673L, 0xd87aeb967a906befL, 0xaabc74493cb02fe6L, 0x74d48fc2e7da143eL, + 0x6ec1c8fed3f2c1fdL, 0xe01e0704b463f18eL, 0xc3d88a4d3a8056e4L, 0xd01ae0ffab6c8f3fL, + 0x881ba052620ae7c7L, 0xcea033aef0a823a5L, 0x8d2cad91d83df1e3L, 0x18746d205e66dbe9L, + 0x3061f8e58d046650L, 0xd819c59f0ce2cf8bL, 0x144e89e93635e870L, 0x3415e88279b21651L, + 0xd6f7ab944b86c3faL, 0x45f1dd15d0f67bdcL, 0xbf0d97c7f4fa24f4L, 0x34a7de520a57fcd2L, + 0x4ba86fda03e9e2bcL, 0xa7995265a025b552L, 0x698f6819d5f51cf7L, 0xd07dbe9d8a156981L, + 0x2683945373857fc1L, 0x116f8a84f96167deL, 0x8bc832bd85595ebfL, 0xb206519d74fdfafaL, + 0xde9519b2e9b5cc5fL, 0x16fdd6f2da1d8163L, 0x7ba32bd48ef56f11L, 0x6f4e4d7ee8b29717L, + 0xd31576dde7468aadL, 0x023bb08848676045L, 0xf6dcc083178160b7L, 0x42035f426250e683L, + 0x343732993cfed89fL, 0x0640a870a22d3d58L, 0x65cff80b53b4ae6aL, 0x27996fa17ab05215L, + 0xfd5db01401b21a04L, 0x894508784bc1673cL, 0x5bfcf43a2380e27dL, 0x4cd6dcc2715583b7L, + 0xa43b3763e7d4c902L, 0x6da83e12ef0c1257L, 0xfe80a602b0335affL, 0x293a7d8f4ff344deL, + 0xb4ae7c2b8956bf5aL, 0x6b45432d38254b4dL, 0xd086acbdf15d9455L, 0xa4d19e43f41ea87bL, + 0xf01f13ba4bb87fbfL, 0xca582cf301a299ffL, 0x0ddad3d45298fa7dL, 0x0646a130459c3999L, + 0xc08e3af3747e2ceeL, 0xfc7db8aa9ed67295L, 0x783b329e7bd79d5fL, 0x732dbc607957af7bL, + 0x8e446ac19fb26555L, 0xff1dfa4d61dc89a5L, 0xb6fbc46bd8d011d8L, 0x185147ec5779f0d7L, + 0x6eb2cf6149a5380fL, 0xb0e773df803a1eaeL, 0xc07706c5519bfce5L, 0xc35abcf54fa95f14L, + 0x40a01d99a38608eaL, 0x776dcd6f603c277fL, 0x6ae12389b1d6d0bbL, 0x8bd981448df92bb9L, + 0x426a6a7ca21a2c16L, 0x87efd5b71c1bad26L, 0x71fb7fc4cd41de48L, 0xdd9033c45619d463L, + 0x40eaab322654cef7L, 0xe077fffed6f3e3a2L, 0x375a4dbef9384447L, 0x2066b009d2c4a100L, + 0xeca4a5794a068447L, 0x2128f64bddf341a1L, 0x738b4bb1be90bd61L, 0x433772cf3813d52eL, + 0x9540c88add8e4474L, 0x0b6d5decd21d3519L, 0x654ead966745642dL, 0xe1bfb03c3b4bdb4cL, + 0x0b977a9937515b1fL, 0x0a4587509ef63870L, 0xe89f0de1d9cfd44aL, 0x23a91390272e7f68L, + 0xd92defbc9096b8d8L, 0x004db87174612539L, 0xc88ecaabdd1a71f1L, 0x050de38393073346L, + 0x8af1426d7964e038L, 0xf352c4fef8ad5c87L, 0x6f26bc7408e26548L, 0x0d41543fd9bf3084L, + 0xfc4e07553a840fc6L, 0x5ef117de86a555a9L, 0x1f11c42dffb5ae1bL, 0x4147648f07490fa5L, + 0x09b35fd7671b21aaL, 0x1453b14f7ccca481L, 0x944f6fcce4c9b2baL, 0x5b08dd2e3583dc06L, + 0xe0220df78dc9c22dL, 0x1c200b9506cbf666L, 0x8a0b7465eadb523bL, 0xfbcb43a91a1e2d80L, + 0xe697f44be3c36a58L, 0x2f8a8e48fb7e350dL, 0x7baba71b8920d55fL, 0x10edc0216105bc96L, + 0x52db07c79d7a7a63L, 0x1916e8cef9452ac3L, 0x5cbbbf21f867b6ccL, 0xadd583365a690a4bL, + 0x4e4ca2c8bffc2fdbL, 0xf5fe3416d2eebcfeL, 0x839af8b85e452476L, 0x8496c0c54ad44e16L, + 0x6c46f1ecad4482bfL, 0xb794cad76ae18715L, 0x67b762eec7c62985L, 0x52dc9e68df5b3a53L, + 0x0cc7e444b422a5f9L, 0xadbfe90841c112b0L, 0xfe37b136f0ca5c34L, 0xcfe9e47948a8d73eL, + 0xee90572b86a30d91L, 0x549e72d8262830aaL, 0x3361564b469f32c6L, 0x1e6eba9e0d2648e2L, + 0x5f8e2b2ac5fcb4ebL, 0xe4224fa5f71f7cc6L, 0x7357a9230c76757bL, 0xcad70f74aaf6b702L, + 0xeef28ced23894cc2L, 0x753fdd3352aefd68L, 0x1fed6ba90bbeb9d2L, 0x05316f4ab4034b4bL, + 0x3396df022b9f63d6L, 0x82d7125a7cfd0935L, 0x3519a71caf1f87f0L, 0xd1dfb7a5cc3974beL, + 0xbfae40ecbdbbcc2aL, 0x152c11778e08dd54L, 0x4a96566a6c848554L, 0x3a84d621c340cdd7L, + 0xfd47aa1887e2fb03L, 0xa63cae94b2f1d099L, 0xed61783f3e5b75e0L, 0xefd44864106019beL, + 0x145ff78b80b081aaL, 0x34670e5fcea9230eL, 0x876ef976328db371L, 0x4221f3a5269942a6L, + 0x95315cbd85c648f4L, 0x3ca344dc7c3b1600L, 0x38421ea39ff28780L, 0x31dbeee967c0435cL, + 0x27437c3e268402e7L, 0xdd0cf8343312a654L, 0x965ab9dad1d8aa29L, 0xf871706dd3e23509L, + 0xce23d06c7a25e699L, 0x1b37d59382b27589L, 0x3407f004723d6324L, 0x56efb69cdb5deaa1L, + 0xf46cdd2b9fd604e0L, 0xcad3ca79fdac69bdL, 0x7252802a574e63cbL, 0xc281fb8acc6ec1d3L, + }, + // seed = 2 + { + 0xdd16cb672ba6979cL, 0x3954eaa9ec41ae41L, 0x52cb802771d2966dL, 0xf57ed8eb0d0294f2L, + 0x768be23c71da2219L, 0x6131e22d95a84ad3L, 0xd849e4e49bb15842L, 0x18e8e5c4978cf00dL, + 0x3af5e5867ce1f9bdL, 0x06c75a9fffe83d63L, 0xe8de75a00b58a065L, 0x0a773251bc0d755aL, + 0x629dc21e54548329L, 0x2a168f5e5a883e70L, 0x33547375f0996c86L, 0xdfcb4c7680451322L, + 0x55c1ecaaaa57e397L, 0x4546c346c24f5a31L, 0x6f8f0401dfabc86cL, 0x7760d2d36ee340b4L, + 0xf6448e48bdeb229dL, 0xba70e1633b4dba65L, 0x069cda561e273054L, 0xa010b6a84aebf340L, + 0x5c23b8229eee34b6L, 0xea63c926d90153afL, 0x7d7de27b3e43ec1bL, 0xea119541eddc3491L, + 0xf1259daeddfc724cL, 0x2873ca9a67730647L, 0xa1e7710dade32607L, 0x758de030b61d43fdL, + 0xd2c9bcbfa475edb4L, 0x18ade47bb8a0aa29L, 0xf7a74af0ff1aea88L, 0x6f8873274a987162L, + 0x6963e8d876f4d282L, 0xd435d4fe448c6c5bL, 0x93ec80ba404cafffL, 0xcf90d24c509e41e7L, + 0x5f0fc8a62923e36eL, 0x9224878fe458f3a4L, 0xd9a039edf1945bcdL, 0x0877d1892c288441L, + 0x75205491f4b4740bL, 0x30f9d2d523a9085bL, 0x4b7f4029fa097c99L, 0x170bb013745709d4L, + 0x7087af537f11ef2eL, 0x28c62b88e08fc464L, 0x84bbcb3e0bb56271L, 0x485a4b099165c681L, + 0x357c63357caa9292L, 0x819eb7d1aee2d27eL, 0xdaa759eb9c0f8c9dL, 0x42cdc36729cc3db5L, + 0x9489aa852eddbb06L, 0x8161e4f85a84e6d4L, 0xa964863fdad3eb29L, 0xcc095ddbce1a6702L, + 0x3ecfadbb8dc2ce58L, 0x971316509b95a231L, 0xc8f484d1dbc38427L, 0xae9c510c463574c0L, + 0xdf2b31179600c21aL, 0x440de87bada4dfa3L, 0xbd8d30f3f6fb7522L, 0x84e6d7f678a0e2d0L, + 0x0ec4d74323e15975L, 0xf6947610dad6d9abL, 0x73a55a95d73fe3a5L, 0x3e5f623024d37edaL, + 0x8d99a728d95d9344L, 0x8b82a7956c4acdc4L, 0x7faeaea4385b27f6L, 0x540625ff4aa2ff21L, + 0x4aa43b3ebd92ce2bL, 0x899646a6df2da807L, 0x49225115780942d7L, 0xe16606636af89525L, + 0xb980bcf893888e33L, 0xf9ed57695291b0d8L, 0x5c6dd14464619afaL, 0x50606d69b733d4f3L, + 0x7fb1af465b990f97L, 0x3fab2634c8bbd936L, 0x556da6168838b902L, 0x0f15975902a30e1fL, + 0xb29d782ae9e1991fL, 0xae00e26ff8f7e739L, 0xd3da86458bb292d5L, 0x4528ee0afb27e4ceL, + 0x49882d5ba49fabadL, 0x7e873b6a7cf875eeL, 0x777edd535113c912L, 0x94ed05e7ff149594L, + 0x0b8f95fc4211df43L, 0x9135c2b42426fef2L, 0x411e6c2b47307073L, 0x503207d1af0c8cf8L, + 0xd76f8619059f9a79L, 0x64d24617855dee45L, 0xf7bc7a877923196aL, 0xd6cc42ed6a65be79L, + 0xe3912ff09d4fc574L, 0x4192d03b2bc2460aL, 0xa0dcc37dad98af85L, 0xfc59049b2a5818a4L, + 0x2128bae90a5b975fL, 0xbe7067ca05ea3294L, 0x5bab7e7753064c4fL, 0x42cbf0949ef88443L, + 0x564df4bbd017492cL, 0xf2c2eb500cf80564L, 0x5b92e67eb00e92afL, 0x8c4103eef59c0341L, + 0x83412122b8284998L, 0x888daf2da0636b6dL, 0x4d54b10303dd07d6L, 0x201190e7c1e7b5edL, + 0x3797510bb53a5771L, 0x03f7bc598b570b79L, 0xdc1e15d67d94f73eL, 0x721e8b499ebe02c1L, + 0x71f954f606d13fa0L, 0x0c7a2e408c168bf0L, 0x07df2ef14f69c89dL, 0xe295096f46b4baafL, + 0x7a2037916438737eL, 0xd1e861aeaf8676eaL, 0xb36ebdce368b8108L, 0xb7e53b090ddb5d25L, + 0x5a606607b390b1aaL, 0x475e52994f4a2471L, 0xbcc2038ba55b2078L, 0x28b8a6b6c80df694L, + 0xb5f0130ec972c9a2L, 0x7a87cd2a93276b54L, 0x4d0eec7ecf92d625L, 0xac1a8ce16269a42eL, + 0xa4ca0237ca9637b8L, 0xd8dc8ff91202b6ffL, 0x75b29846799d7678L, 0x761b11a5edd9c757L, + 0xf2581db294ef3307L, 0xe3173c2b6a48e20fL, 0xe46fd7d486d65b3cL, 0x1352024303580d1fL, + 0x2d665dae485c1d6dL, 0x4e0905c825d74d3bL, 0x14ff470c331c229eL, 0xbdc656b8613d8805L, + 0x36de38e396345721L, 0xaae682c1aa8ff13bL, 0x57eb28d7b85a1052L, 0xf3145290231d443aL, + 0xd0f68095e23cbe39L, 0x67f99b3c2570b33dL, 0x54575285f3017a83L, 0x9b2f7bb03d836a79L, + 0xa57b209d303367a9L, 0x7ccb545dd0939c79L, 0x1392b79a37f4716dL, 0x6e81bb91a3c79bcdL, + 0x2c2cd80307dddf81L, 0xb949e119e2a16cbbL, 0x69625382c4c7596fL, 0xf19c6d97204fb95cL, + 0x1b2ea42a24b6b05eL, 0x8976f83cd43d20acL, 0x7149dd3de44c9872L, 0xc79f1ae2d2623059L, + 0xca17a4f143a414e1L, 0x66d7a1a21b6f0185L, 0xed2c6198fe73f113L, 0x16a5f0295cbe06afL, + 0x5f27162e38d98013L, 0xf54d9f295bdc0f76L, 0x9ba7d562073ef77bL, 0xa4a24daaa2cfc571L, + 0x49884cf486da43cdL, 0x74c641c0e2148a24L, 0xbff9dcbff504c482L, 0xf8fc2d9403c837abL, + 0x6ccc44828af0bb1eL, 0xbcf0d69b4c19dfdbL, 0x8fe0d962d47abf8fL, 0xa65f1d9d5514271dL, + 0x26ff393e62ef6a03L, 0xc7153500f283e8fcL, 0xea5ed99cdd9d15cdL, 0xfc16ac2ba8b48bb7L, + 0xf49694b70041c67aL, 0xbd35dd30f5d15f72L, 0xcf10ad7385f83f98L, 0x709e52e27339cdc2L, + 0xe9505cb3ec893b71L, 0x2ffa610e4a229af7L, 0x12e1bc774d1f0e52L, 0xe301a3bb7eacccc8L, + 0x1fdd3b6dcd877ebfL, 0x56a7e8bda59c05aaL, 0x99acd421035d6ab4L, 0xfd21e401cecd2808L, + 0x9a89d23df8b8d46fL, 0x4e26b1f1eb297b9cL, 0x9df24d973e1eae07L, 0xe6cdc74da62a6318L, + 0xfc360d74df992db0L, 0xf4eca0a739514c98L, 0x481c515ba9bf5215L, 0xce89cce80f5f3022L, + 0xf487a10fc80e4777L, 0x235b379a87e41832L, 0x76f72e028371f194L, 0xd044d4a201325a7dL, + 0x47d8e855e0ffbddeL, 0x268ae196fe7334b0L, 0x123f2b26db46faa8L, 0x11741175b86eb083L, + 0x72ee185a423e6e31L, 0x8da113dfe6f6df89L, 0x286b72e338bbd548L, 0xa922246204973592L, + 0x7237b4f939a6b629L, 0x31babda9bedf039aL, 0xb2e8f18c6aeec258L, 0x0f5f6ce6dd65a45eL, + 0x8f9071a0f23e57d3L, 0x71307115ba598423L, 0xcbe70264c0e1768cL, 0x1c23729f955681a8L, + 0xfbc829099bc2fc24L, 0x9619355cbc37d5d6L, 0xea694d4e59b59a74L, 0xb41cf8d3a7c4f638L, + 0xae1e792df721cd0bL, 0x7cd855d28aac11f6L, 0xca11ba0efec11238L, 0x7c433e554ce261d8L, + 0xe3140366f042b6baL, 0x8a59d68642b3b18cL, 0x094fcdd5d7bccac2L, 0x9517d80356362c37L, + 0x4a20a9949c6c74e8L, 0xc25bcf1699d3b326L, 0xa8893f1d1ed2f340L, 0x9b58986e0e8a886eL, + 0x29d78c647587ce41L, 0x3b210181df471767L, 0xd45e8e807627849dL, 0x1ec56bc3f2b653e3L, + 0x974ff23068558b00L, 0xdb72bdac5d34262cL, 0x23225143bb206b57L, 0xd0a34cfe027cbb7eL, + }, + // seed = 3 + { + 0x39209fb3eb541043L, 0xee0cd3754563088fL, 0x36c05fc545bf8abeL, 0x842cb6381a9d396bL, + 0xd5059dcb443ce3bfL, 0xe92545a8dfa7097eL, 0xb9d47558d8049174L, 0xc6389e426f4c2fc0L, + 0xd8e0a6e4c0b850d3L, 0x7730e54360bd0d0dL, 0x6ecb4d4c50d050d5L, 0x07a16584d4eb229fL, + 0x13305d05f4a92267L, 0xb278ddd75db4baecL, 0x32381b774138608fL, 0x61fe7a7163948057L, + 0x460c58a9092efee6L, 0x553bf895d9b5ff62L, 0x899daf2dabfd0189L, 0xf388ab9c1c4b6f70L, + 0xd600fe47027ea4cdL, 0x16d527ec2b5ef355L, 0x5ac1f58ff6908c81L, 0xa08d79ff8ee9ffe8L, + 0xc1060a80b7a5e117L, 0x14b2c23118c60bdaL, 0x8cc0defbb890df8fL, 0xe29540fd94c6d28bL, + 0xa604f003f82d5b71L, 0xa67583d4eb066d18L, 0xd62cbd796322b3fcL, 0x070cfe244cdcccf3L, + 0x73557c30b3af47e5L, 0x2e544e31153a2163L, 0x996eef7464d5beadL, 0xbc71cb5ab0586cdcL, + 0x0bfcb6c1b517ed69L, 0x62b4f1fcc82e8ca0L, 0x0edbc68f544965c5L, 0x40fa39baa24af412L, + 0xf39aeb2413dab165L, 0x17e6013e7afee738L, 0x8109bff1c8d42a9dL, 0x3cd99863390989b5L, + 0x02021a4cc9c336c8L, 0xa06060778cb60aa4L, 0xd96591db60bc1e06L, 0xd2727175183f4022L, + 0xcdc1f1c5bce3e7ceL, 0xb393ccc447872a37L, 0xdf6efe63257ead3aL, 0x20729d0340dbceb6L, + 0x9f3d2d26fc0ea0d7L, 0xf392e0885189bd79L, 0xdf2ee01eb212b8b6L, 0x6e103a0c0f97e2c3L, + 0x96c604a763bd841bL, 0x9fc590c43bba0169L, 0xf92dcd5ddc248c40L, 0x113a8b54446941dcL, + 0x5943eda146b46bb8L, 0xbf657901a36a39a7L, 0x5a4e0e7ea6568971L, 0xb94c635bae9f9117L, + 0x2626fb65b3a4ef81L, 0xa59bfd5478ce97deL, 0x79112ba9cc1a1c63L, 0xf41f102f002cf39cL, + 0x0a589bcbfb7ff1c8L, 0xa1478c53540c4fa1L, 0x60d55e72c86dfacaL, 0x312e7b6840ea7a39L, + 0x8aae72dcccfe1f75L, 0xff2f51f55bf0247aL, 0x3c2e4b109edb4a90L, 0x5c6d73f6525c7637L, + 0xe49acb04a199f61cL, 0x27860642d966df7fL, 0x541ce75fb1e21c30L, 0xd9fcd6f90806c7ccL, + 0xb87c27bc93a7969bL, 0x92f77a1179b8f8dcL, 0xb1f29379deb89ed4L, 0x7e63ead35808efe7L, + 0x13545183d7fa5420L, 0x575f593e34cf029dL, 0x27f1199fb07344aeL, 0xe67f95f7dc741455L, + 0x49b478b761ab850bL, 0xd7bedf794adfc21eL, 0xdc788dcd2dda40aeL, 0x14673eb9f4d8ad35L, + 0x0cced3c71ecf5eb1L, 0xe62d4e6c84471180L, 0xdfe1b9e2cb4ada7dL, 0x70185a8fce980426L, + 0x0ce2db5e8f9553d6L, 0x1fedc57bb37b7264L, 0xb9310a2e970b3760L, 0x989ff8ab9805e87dL, + 0x0b912d7eb712d9eeL, 0x1fe272830379e67cL, 0x16e6a73aff4738fbL, 0xeed196d98ba43866L, + 0x7088ca12d356cbe2L, 0x23539aa43a71eee0L, 0xed52f0311fa0f7adL, 0xa12b16233f302eeaL, + 0xc477786f0870ecb4L, 0xd603674717a93920L, 0x4abe0ae17fa62a4cL, 0xa18f1ad79e4edc8dL, + 0xc49fe6db967c6981L, 0xcc154d7e3c1271e9L, 0xdd075d640013c0c0L, 0xc026cd797d10922aL, + 0xead7339703f95572L, 0x4342f6f11739eb4bL, 0x9862f4657d15c197L, 0x4f3cb1d4d392f9ffL, + 0xe35bffa018b97d03L, 0x600c755031939ad3L, 0xb8c6557ffea83abfL, 0x14c9e7f2f8a122eaL, + 0x0a2eb9285ee95a7cL, 0x8823fec19840c46fL, 0x2c4c445c736ed1d0L, 0x83181dff233449f1L, + 0x15ed3fca3107bef5L, 0x305e9adb688a4c71L, 0x7dbef196f68a3e2eL, 0x93e47ece3e249187L, + 0x8353c5e890ead93cL, 0xea8a7ae66abafdf7L, 0xf956dbb6becf7f74L, 0x9f37c494fbfdb6e4L, + 0x11c6cbaa2485dd32L, 0x206f336fcca11320L, 0x9befe9a59135d8feL, 0x5f3ef8b8db92c7dbL, + 0xbb305e556ce0ce9aL, 0xf26bdafb1305887fL, 0xcbf28abe23f08c61L, 0x0bc64173b914e00bL, + 0x9168da52e983f54aL, 0x6ea41d09c3574a3eL, 0x78aa44d4a74459aeL, 0x2931422878387bf5L, + 0x018f64a3a92c2d9cL, 0x9be43f6752e66b34L, 0xae378890decd1152L, 0x07325329a1cb7623L, + 0x3b96f4ee3dd9c525L, 0x2d6ebcdbe77d61a3L, 0x10e32b0e975f510cL, 0xffc007b9da959bf9L, + 0x38bf66c6559e5d90L, 0xbe22bdf0bf8899feL, 0x87807d7a991632a8L, 0x149a0d702816766aL, + 0x026f723db057e9abL, 0xeeecb83625ec6798L, 0xcec2ed5984208148L, 0xd985a78e97f03c84L, + 0xf96c279e7927b116L, 0x99d5027b3204f6e2L, 0x13a84878c3d34c55L, 0x5cf5ec96229e9676L, + 0x0bc36b07e4f8e289L, 0xbed33b80a069914dL, 0x2fbfbdd1ff4b9396L, 0xab352bb6982da90fL, + 0x154d219e4fa3f62bL, 0x4d087512bb6b9be7L, 0xc582e31775ee400eL, 0x7dadb002ae8c4a4eL, + 0xaae2957375c1aee2L, 0x5f36ca643356625bL, 0xf87cf8eb76e07fb7L, 0x46f432a755e02cc3L, + 0x36087e07aba09642L, 0xe5642c1e4ebb9939L, 0xb9152d22338eefadL, 0xf7ba44278a22cf7fL, + 0xd3b8013502acd838L, 0x7761511da6482659L, 0xb0857621638e8e50L, 0x552eddb4a8b1d5f5L, + 0xc43d9861e812c3eaL, 0xd765c2aada47910cL, 0x21c935b68f552b19L, 0x6256d5641a2b47dcL, + 0xab711d8e6c94bc79L, 0xa8d0b91a2a01ab81L, 0x5e6d66141e8d632aL, 0x7638285124d5d602L, + 0x794876dbca3e471fL, 0x951937d8682670ceL, 0x0f99cb1f52ed466aL, 0x8c7cd205543b804cL, + 0x2fd24d74a9c33783L, 0xe5dcb7b7762e5af1L, 0x45e6749cca4af77cL, 0x540ac7ee61f2259fL, + 0x89c505c72802ce86L, 0xeab83b9d2d8000d1L, 0x9f01d5e76748d005L, 0xc740aaef3035b6d0L, + 0x49afcd31d582d054L, 0xcba5dc4c1efb5ddcL, 0xc0a4c07434350ca1L, 0xfc8dfaddcc65ee80L, + 0x157c9780f6e4b2d9L, 0x9762a872e1797617L, 0xc4afae2cf3c7e1bdL, 0x71cde14591b595d4L, + 0x8843c3e0e641f3b9L, 0xd92ecd91dce28750L, 0x1474e7a1742cb19fL, 0xec198e22764fa06bL, + 0x39394edb47330c7dL, 0x00ba1d925242533dL, 0xaed8702536c6fb30L, 0x6d3618e531c2967aL, + 0x77f7cedcd7cc0411L, 0xbc1e2ab82be5b752L, 0x07b0cf9223676977L, 0x596c693b099edd53L, + 0xbb7f570f5b9b2811L, 0x96bfdad3c4a6840cL, 0x668015e79b60c534L, 0x3ad38d72123f1366L, + 0x6b994d81d2fcbb09L, 0x70885f022c5052d8L, 0xc891ee79d9306a7bL, 0x2c4df05c0ed02497L, + 0x19ebc13816898be2L, 0xea7c64df11c392a2L, 0xb7663e88dd12e1bdL, 0x79f768cb8e154c21L, + 0x1fb21b12e945933bL, 0xe6a9045643f6906eL, 0x544c47acd7e15371L, 0xb7709b14f727e3d1L, + 0x326ee36a46942971L, 0x477f1cf7b0e2d847L, 0x88b8f6b82b3b0c24L, 0x18bc357b80e3cd5cL, + 0x3333de70e4d66e0bL, 0x4fd4c5e148583cf6L, 0xae1b62f3008c0af3L, 0xc49f419b6ab29cf5L, + 0x2c29fa65afc3fa28L, 0x4b19d93734d03009L, 0x7dd6c09e589276adL, 0x1cece97f30de48adL, + }, + // seed = 4 + { + 0x58bdf4338602e4fbL, 0x71a5620b02c926d5L, 0x3811c960129c2d9fL, 0x29c2fb11fccac567L, + 0x0d6b1ea7780f1352L, 0xcc4d3ddfae3f87b3L, 0xfdd30257362a586bL, 0xabc948fde69f25f1L, + 0x51b3523469d30f7bL, 0xe0f0322724405aceL, 0xd3729266d896da1eL, 0xb10c37e5147915bfL, + 0x8b577039f9fa32a3L, 0xe677c6a9cbfb44b3L, 0x7317a756ebb51a03L, 0xf8e988ef37359485L, + 0x600fc1ef3f469ff3L, 0xbf0b8f8520444e01L, 0x3711168b08b63d73L, 0x34146f2944a6cb36L, + 0x717feb263862cddeL, 0x7185f8347db00412L, 0x900798d82127e693L, 0x84089e976a473268L, + 0x10f8308c0d293719L, 0xf62a618d4e5719b8L, 0x8bdbd257a1a9516fL, 0xf49f666fd7a75110L, + 0xbaf45e2db7864339L, 0xe4efa1ea0c627697L, 0x3e71d4c82a09fe10L, 0x54a2a51cf12127bbL, + 0xa0592c9f54ba14cdL, 0x27dd627a101c7a42L, 0x3d2ceb44b3d20d72L, 0x7ee1f94a68ca8f5dL, + 0x7e8cb8651b006c36L, 0xbd9fa7ca3a475259L, 0x856de173586a7b34L, 0xcedb291b594cb1b5L, + 0xa3d6e462fd21cddcL, 0x74561d10af9118e4L, 0x13a3d389fc2d4b36L, 0xeea8594a4a054856L, + 0xf56d7474d9ba4b13L, 0x25ddce2f6490b2fdL, 0x920653ff3a8d830bL, 0xcd8c0c9cdac740d1L, + 0x2c348a738db9c4a0L, 0x2967ccbe8ea44c22L, 0x47963f69adb049f8L, 0xf9d01eb5b4cf7eb6L, + 0x7a5c26eb63a86bd2L, 0x62ad8b7a71fa0566L, 0xb373213179f250aeL, 0x589d4e9a88245a4dL, + 0x433dafebe2d558a8L, 0x521fbef2c8fe4399L, 0x62a31f9ff9ccd46bL, 0x51602203eba7c1a6L, + 0x9afc8c451b06c99fL, 0xb529085bdbaffceaL, 0xac251825cc75892bL, 0x94976a5bce23d58eL, + 0xdd17925b6c71b515L, 0x568fd07a57bce92eL, 0xefac31200d8bd340L, 0x716c3e466b540ef9L, + 0x3d2c9e380063c69bL, 0x14168f9a3662dd83L, 0xd298c7504dbc412fL, 0x74490a94f016719fL, + 0x0e0da431e1ab80c8L, 0xe321f63dc6b169aeL, 0xf08671544febc95aL, 0x39324450cc394b3bL, + 0xea6e3d35f1aa3a70L, 0x8ef8a886508ce486L, 0xdc1a631ef0a17f06L, 0xfda2b3fbcd79e87bL, + 0xd75bcae936403b10L, 0xf88b5bd9f035f875L, 0xc43efec2e3792dd4L, 0xe9fac21a9d47cd94L, + 0xc2876f0c4b7d47c3L, 0xaba156cf49f368b4L, 0x5ccda2170fa58bf9L, 0xadc92c879ed18df7L, + 0x110c1b227354e6c8L, 0x298ee7a603249200L, 0xde92142ede0e8ee7L, 0x88e4a4610644ba9eL, + 0xbb62d277e7641d3aL, 0xb9be1985b7bf8073L, 0x29024e5426cdb0d1L, 0xf6aefd01f3092ab8L, + 0x2a07087b313133aaL, 0x6d71f445d6dfc839L, 0x1e2412ff12e5526bL, 0xed5cdeba6617b9e1L, + 0x20b1d0d5e5f8760eL, 0x12ff15705c368260L, 0x7bf4338b7c387203L, 0x34ff25f00cd06185L, + 0x1148c706c518cf28L, 0x5c04f0623388f025L, 0xcb9d649275d87d79L, 0x9b5f0c24fabc42ecL, + 0x1a7b5e7964e33858L, 0x2a81bbd8efdc6793L, 0x8d05431ffe42752eL, 0x83915cd511002677L, + 0x580ed4d791837b31L, 0x5982e041d19ff306L, 0xcad0d08fa5d864caL, 0x867bee6efe1afa63L, + 0x26467b0320f23009L, 0xd842414dfda4ec36L, 0x047fcdcbc0a76725L, 0xbddb340a3768aecaL, + 0xef4ce6fa6e99ab45L, 0x88c5b66c7762bf9bL, 0x5679f1c51ffb225dL, 0xdab79048317d77eeL, + 0xf14e9b8a8ba03803L, 0xe77f07f7731184c1L, 0x4c2aab9a108c1ef5L, 0xa137795718e6ad97L, + 0x8d6c7cc73350b88bL, 0x5c34e2ae74131a49L, 0xd4828f579570a056L, 0xb7862594da5336fcL, + 0x6fd590a4a2bed7a5L, 0x138d327de35e0ec1L, 0xe8290eb33d585b0bL, 0xcee01d52cdf88833L, + 0x165c7c76484f160eL, 0x7232653da72fc7f6L, 0x66600f13445ca481L, 0x6bbdf0a01f7b127dL, + 0xd7b71d6a1992c73bL, 0xcf259d37ae3fda4aL, 0xf570c70d05895acfL, 0x1e01e6a3e8f60155L, + 0x2dacbb83c2bd3671L, 0x9c291f5a5bca81afL, 0xd976826c68b4ee90L, 0x95112eec1f6310a2L, + 0x11ebc7f623bc4c9aL, 0x18471781b1122b30L, 0x48f7c65414b00187L, 0x6834b03efa2f5c30L, + 0x0875ef5c2c56b164L, 0x45248d4f2a60ba71L, 0x5a7d466e7f7ba830L, 0x2bebe6a5e42c4a1dL, + 0xd871d8483db51d10L, 0x6ee37decd2fd392fL, 0x7d724392010cede3L, 0x8e96ef11e1c9bcc8L, + 0x804a61d86b89d178L, 0xbb1b83ce956055ecL, 0xcb44e107410ff64fL, 0xc426bb09ee0ba955L, + 0x057c08f42c3dd7f1L, 0x40ea1ec148602bdfL, 0xc24688deeb65d7f1L, 0xd8bcc53c768ba4e4L, + 0x16e0e3af65c1106cL, 0xfc12f7e7d647218bL, 0x70d6e1d3ee93cef4L, 0x01d2a505c4541ef9L, + 0x1ef79e16e764d5c3L, 0x0363d14d13870b98L, 0xb56ef64345d06b11L, 0xe653d557ebb7c346L, + 0x8304a8597c2b2706L, 0x1536e1322ce7e7bbL, 0x525aec08a65af822L, 0x91f66d6e98d28e43L, + 0xe65af12c0b5c0274L, 0xdf6ae56b7d5ea4c2L, 0x5cef621cedf3c81cL, 0x41e8b1ffd4889944L, + 0xb5c0f452c213c3e5L, 0x77af86f3e67e499bL, 0xe20e76ea5b010704L, 0xbdc205ab0c889ec0L, + 0xc76d93eb0469cd83L, 0x17ac27f65cab0034L, 0xd49ec4531fd62133L, 0x07a873ea2f1b9984L, + 0xbff270dfef0032eeL, 0x1764dbe91592f255L, 0xe40363126f79e859L, 0xa06cad3ab46971f6L, + 0x0be596e90dedd875L, 0x3387cce5c1658461L, 0x44246acf88a9585eL, 0xe0ad82b92d5ecb2cL, + 0x2177491c9a1600a6L, 0x16e7c4aac0f02422L, 0x75792eeeec15c4e1L, 0x2309cd359d08ee30L, + 0x7cd9831dd1b83b0aL, 0x374914a7c4ee8cf0L, 0x0dd17765c9ac2e54L, 0xb7847470ba9a7688L, + 0xfba4f4bbe2991173L, 0x422b203fc3de040eL, 0x63bfcaf2ecf2ab0eL, 0x0c5559f3a192946eL, + 0xfdf80675c1847695L, 0xf5f570accab842c9L, 0x65cc5a448767afeaL, 0x1efeb0a7ee234f2fL, + 0x9b05f03d81e7b5d2L, 0xe7c31317a8626cf4L, 0x620f2a53081d0398L, 0x1b6de96cdd9943aeL, + 0x8c226a436777d303L, 0xa08fbbd50fafb10dL, 0x6a64c5ec20104883L, 0x9c9c653502c0f671L, + 0x678a02b2174f52a0L, 0x68e008ba16bbad4bL, 0xa317c16d2efb860fL, 0xeab2075d17ed714cL, + 0x565eeeddf0c4ea15L, 0x8ec8e94d242a6c19L, 0x139e8e27d9000faeL, 0xc977a7ff1b33d2f5L, + 0x1d0accca84420346L, 0xc9e82602cd436e03L, 0x6a2231da53d2ccd3L, 0xb44b12d917826e2aL, + 0x4f4567c6a74cf0b9L, 0xd8e115a42fc6da8fL, 0xb6bbe79d95742a74L, 0x5686c647f1707dabL, + 0xa70d58eb6c008fc5L, 0xaaedc2dbe4418026L, 0x6661e2267bdcfd3dL, 0x4882a6eda7706f9eL, + 0xf6c2d2c912dafdd0L, 0x2f2298c142fd61f9L, 0x31d75afeb17143a8L, 0x1f9b96580a2a982fL, + 0xa6cd3e5604a8ad49L, 0x0dae2a80aad17419L, 0xdb9a9d12868124acL, 0x66b6109f80877facL, + 0x9a81d9c703a94029L, 0xbd3b381b1e03c647L, 0xe88bc07b70f31083L, 0x4e17878356a55822L, + }, + // seed = 5 + { + 0xb3c58c2483ad5eadL, 0x6570847428cdcf6cL, 0x2b38adbf813ac866L, 0x8cb9945d37eb9ad3L, + 0xf5b409ec3d1aed1cL, 0xa35f4bffc9bb5a93L, 0x5db89cde3c9e9340L, 0xff1225231b2afb2bL, + 0x157b0b212b9cc47dL, 0xf03faf97a2b2e04dL, 0x86fdab8544a20f87L, 0xfcb8732744ae5c1cL, + 0xd91744c0787986d5L, 0x5f8db2a76d65ad05L, 0xcff605cbed17a90dL, 0xf80284980a3164e7L, + 0x59cc24e713fccc7dL, 0x268982cada117ce4L, 0xcd020e63896e730eL, 0xe760dc46e9fe9885L, + 0x6aaece8ab49c6b5dL, 0x7451194d597aae3eL, 0x35d4385900332457L, 0xa40fb563a096583dL, + 0xa797b612f7f11b76L, 0x2fed6eb68e6a2b9bL, 0x2f06ee64aeffd943L, 0x9dd0e49d9ca45330L, + 0x97d48f08bd7f1d8fL, 0x1cfa7fe3ebe4d8eeL, 0x2a2ba076bd397d42L, 0x68c4344f7472f333L, + 0xce21ec31987d74b5L, 0xb73dabdc91d84088L, 0x801aadee592222feL, 0xaf41345398ebc3f5L, + 0x8a8f653d7f15ee46L, 0xce2d065ff2ba2965L, 0x4e05da515da2adb7L, 0xa6dbdb8aa25f0fd4L, + 0xca9f9666bbd2d5a9L, 0x6b917ce50bd46408L, 0x1550cc564ba6c84dL, 0xb3063ae043506504L, + 0x84e5f96bb796653dL, 0xe2364798096cf6e3L, 0x3b0dfedf6d3a53d0L, 0xb7e4c7c77bde8d93L, + 0xe99545bac9ab418aL, 0xa0e31f96889507bbL, 0x883c74f80c346885L, 0xf674ae0b039fd341L, + 0x8bb6ce2d5e8d1c75L, 0x0c48737966a7ed7cL, 0x04fcdf897b34c61cL, 0xe96ac181bacbd4d6L, + 0x5a9c55a6106a9c01L, 0x2520f020de4f45d3L, 0x935730955e94d208L, 0xce5ad4d7f3f67d3bL, + 0xa4b6d107fe2d81caL, 0x4f0033f50ae7944eL, 0x32c5d28dd8a645a7L, 0x57ce018223ef1039L, + 0x2cbab15a661ab68eL, 0x6de08798c0b5bec2L, 0xee197fb2c5c007c6L, 0x31b630ac63e7bda2L, + 0xab98785aefe9efe3L, 0xa36006158a606bf7L, 0x7b20376b9f4af635L, 0xa40762fdc3c08680L, + 0x943b5faffd0ebee2L, 0x7f39f41d0b81f06eL, 0x7c4b399b116a90f8L, 0x24e1662ac92bc9f3L, + 0xcf586fc4e8e6c7dbL, 0xe46e0d047eeb12d7L, 0xe8021076e4ea9958L, 0x11fc13492e3ca22aL, + 0xd61eae01410397e3L, 0x7e8c4a58036a8e9fL, 0x068a6de267970745L, 0x64faab129bef1a41L, + 0xb4a6f720943dad01L, 0x631491058d73a9d5L, 0xdad4fe95eab3ec02L, 0x0a8b141c5c3a44f6L, + 0x9fc69d4c2b335b98L, 0x94d5f84a07d6e4cdL, 0x1b73965de143c608L, 0x443932c2dda54bccL, + 0x7397818fb0b04cd2L, 0xef4ab03a1202b277L, 0xf3d2ee459c0c2b92L, 0x182d4daf8b058a87L, + 0x90e63035d7b51368L, 0xba4cd8b9a95d45fdL, 0x12a7392c76731090L, 0x890d264ec5d082d2L, + 0xeeaf5c363da4994eL, 0xd6aad756902123fbL, 0xb531ebebdb28f191L, 0xe71ce659fc59babdL, + 0x37c1b94f63f2dcb5L, 0xe4e3abeb311f9b96L, 0x4a31b72ccb8695d3L, 0x52cae1f0629fdce4L, + 0xe5b0475e2ed71369L, 0x2724e8c3506414fbL, 0xbab0367920672debL, 0x0161a781c305449fL, + 0x37b70f40f5bb60beL, 0xddd1094c50251a01L, 0x3b28283afd17224eL, 0x06dec0cfe889fc6bL, + 0x47608ea95bb4902dL, 0xad883ebc12c00e82L, 0x9e8d7ae0f7a8df29L, 0xa79443e9f7c013a1L, + 0xcfa26f68b7c68b71L, 0x33ae6cc19bda1f23L, 0xd9741e22b407887fL, 0xf2bff78066d46b1cL, + 0x794123191c9d32d4L, 0x56cb6b903764ec76L, 0x98775d0ef91e1a5aL, 0xae7b713bc15c1db9L, + 0x3b4c1a7870ed7a0dL, 0x46666965f305cc34L, 0x0ea0c3b2e9c6b3cdL, 0x4dc387039a143bffL, + 0x5f38bb9229ef9477L, 0xea5d39ba72af7850L, 0x69a5ed0174ce2b6dL, 0x06969a36bfe7594dL, + 0x0adee8e4065ccaa3L, 0x908a581d57113718L, 0x64822d6c5a8190edL, 0x8c5068b56ace4e4cL, + 0x88ba3b4fb4e30befL, 0xa6ec0b8bb5896cfeL, 0x4e23fcc6b47996fdL, 0xe18e75b0dd549c7aL, + 0xcd90f17e106cf939L, 0x1666fdfb2ef7c52fL, 0x4fae325f206dd88cL, 0xe7bc1160e25b062dL, + 0x3cc999cb246db950L, 0xc5930a7326cd5c37L, 0xb008a48a211367bdL, 0xc5559da145a88fd4L, + 0x1e3ad46655fac69cL, 0x7834266b4841bfd7L, 0xa764450fbffc58ccL, 0x54d8cf93a939c667L, + 0x93c51f11b21b2d9dL, 0x0964112082ed65ccL, 0x4c2df21213e7fb03L, 0xf0405bc877468615L, + 0x17b4fc835d116ab4L, 0xa6b112ae5f3cb4efL, 0x23cfc8a7fd38a46eL, 0x8e0a360dc2774808L, + 0x24ca9c8092105ad5L, 0xafd3f75524f2e0d5L, 0x4f39ed7dbaddc24cL, 0xe5e362c7679a7875L, + 0x00914a916b07b389L, 0xdfe1119b7d5ab5daL, 0xabd6ed9940e46161L, 0x630ed2044171e22cL, + 0xdecc244157dd1601L, 0x777e6d5b4b4868d5L, 0x9b3530bee67017d8L, 0xd2faf08b291fdcb9L, + 0x006e99455d6523deL, 0xd559b5817f6955b5L, 0xefcc1063b0088c61L, 0xed73145ae0f00ae7L, + 0xab2af402cf5b7421L, 0x897767f537644926L, 0x26c9c0473ca83695L, 0x192e34e1881b2962L, + 0xf7cf666ec3b3d020L, 0x27f9b79c7404afb7L, 0xe533e8bed3010767L, 0xe5817838e11d05d3L, + 0x65659c531bd36517L, 0xd427c5e0a23836fdL, 0xf3eab7ea58fa3528L, 0x07683adae1289f35L, + 0x201d6af7e896dd32L, 0xd5da938b9a21ad88L, 0x843fb73ad67bc316L, 0x1782ec7d5feef21bL, + 0x943f66f6ec772877L, 0x7e9112e7b26da097L, 0xeac8161f8663c2c7L, 0xe8600db480a9ebf4L, + 0x07807fc90f6eaf5fL, 0xe0e4c9deb41abf83L, 0xbdf533db271f9c15L, 0xb398411b0497afe2L, + 0xdebb45ef25448940L, 0xe7a5decefcd376c4L, 0xaf1ef3c728c83735L, 0xb8b83a99355cb15aL, + 0x6444a0344f1611e4L, 0xe8bb7f5cf3c60179L, 0x77ab5c5177e75ff7L, 0xc38fd6fa849d585dL, + 0x390d57d53029060aL, 0xa66327eb7b8b593cL, 0x6350a14f6fcd5ac9L, 0x2c08125bcd7008b4L, + 0x2d00c299a6a6bf8eL, 0x6b0039c1f68d1445L, 0x0035150c5d06f143L, 0xa34d01628cc927e1L, + 0xdf5b3164d7b2ede1L, 0x8167db1d0583d72eL, 0x4e13b341cd2ae8bcL, 0xa693d9b1f416e306L, + 0xc15ed7ca0bc67609L, 0xdc344313c1c4f0afL, 0x88b6887ccf772bb4L, 0x6326d8f93ca0b20eL, + 0x6964fad667dc2f11L, 0xe9783dd38fc6d515L, 0x359ed258fa022718L, 0x27ac934d1f7fd60aL, + 0xd68130437294dbccL, 0xaf5f869921f8f416L, 0x2b8f149b4ab4bf9fL, 0xc41caca607e421cbL, + 0x7746976904238ef9L, 0x604cb5529b1532f0L, 0x1c94cd17c4c4e4abL, 0xe833274b734d6bbeL, + 0xe9f1d3ef674539ceL, 0x64f56ed68d193c6aL, 0xe34192343d8ecfc1L, 0xcb162f6c3aa71fe8L, + 0x99eaf25f4c0f8fa4L, 0x92f11e7361cb8d02L, 0xb89170cddff37197L, 0x4f86e68a51e071e3L, + 0x31abf6afd911a75bL, 0x6d20cf259c269333L, 0x4150b9f88fcb6513L, 0x705063989ebf7451L, + 0x559231d927c84410L, 0x1ca8ec4b098bc687L, 0xebed22405c9180e0L, 0xaa815b37d052af59L, + }, + // seed = 6 + { + 0x946ac62246e04460L, 0x9cebee264fcbc1aeL, 0x8af54943a415652bL, 0x2b327ed3b17b8682L, + 0x983fde47b3c3847eL, 0x10a3013f99a2ad33L, 0x6e230bb92d2721efL, 0x1cf8b8369e5c5c50L, + 0x7f64017f2b7b3738L, 0xd393248a62417fa1L, 0x9ff01c0b20a372c5L, 0xb0e44abce7e7c220L, + 0xcebb9f88d48a815fL, 0xdb7df6bd09033886L, 0x7844fc82b6fa9091L, 0x72d095449863b8ecL, + 0xc13e678c89da2c7eL, 0x6caf4d5ad231d12fL, 0x2e0ab7b5fcf35c49L, 0xf410720cb932a70fL, + 0xd66ea581f16fce06L, 0x175c9f002f57dc98L, 0xccbcfd0d32988775L, 0xfde4c407d3b0a232L, + 0x5db2931ae7e97223L, 0x6e07e2173085809fL, 0x6e1d1ec0f9cad73cL, 0xb2fc251a7f802619L, + 0xbc1fc17f04f342deL, 0x8de8f21ec658e078L, 0x72c0f40cbee53fd6L, 0x0678244411fc17a1L, + 0x1d5837ca166b9bbdL, 0xc8cada003c554345L, 0x6a2fe2bfb2e58652L, 0xfca9d797a6f7988bL, + 0x6699e24ac737948bL, 0x69623ffcb05789baL, 0x946429c529d95b75L, 0x0d14df0b2a13970fL, + 0x593d8592c440dfecL, 0x2ee176f3d7e74b94L, 0xae003f1da3be9e26L, 0x0c7b02c4c0f6764aL, + 0x3117e2fa1f632462L, 0xf0f23265b6f1eaebL, 0x3111255d9b10c137L, 0xc82745e509a00397L, + 0xbd1d04037005fea7L, 0xe104ab0dd22a9036L, 0x51b27ce50851ac7aL, 0xb2cb9fb21b471b15L, + 0x29d298074c5a3e26L, 0x6ebdf2058b737418L, 0xc4a974041431b96fL, 0x1ec5a30ccb6bdaacL, + 0xe818beede9bf4425L, 0x4b69b1bce67a5555L, 0xf5c35f1eb0d62698L, 0xf4509bbd8e99867cL, + 0xb17206debd52e1bcL, 0x35785668c770b3beL, 0xe9343987ff5863bcL, 0x2ee768499ac73114L, + 0x5132bb3426eeaaf4L, 0x471bce2c6833c5ffL, 0xbb9a2d5428e6f6f9L, 0xd5678943c595792dL, + 0xab2a65e7f81e479cL, 0xa82407bb23990b31L, 0xdae321383984923cL, 0x01823bb22648e6f1L, + 0xda6e8df4214a8b04L, 0x0e172bb88e03d94fL, 0x552da6c22e362777L, 0x7ce67329fb0e90cbL, + 0x7b2d7f287ede7ebfL, 0xd44f8222500651bdL, 0x4acca1ef58fbb8abL, 0x428ecf058df9656bL, + 0xd7e1ec6a8987c185L, 0x365be6a54b253246L, 0x168849be1e271ee8L, 0x6a00f3c4151a8db2L, + 0x37602727ca94b33dL, 0xf6b50f18504fa9ceL, 0x1c10817f6bc872deL, 0x4bfe1fe42b0f3638L, + 0x135fad4b8ef6143bL, 0x1b25ad2bafc25f58L, 0x41e37f85cf321f92L, 0xfc73f75d9d5b9beaL, + 0x9eb3694d1e9cb7e1L, 0x601d51f08fa83b90L, 0x234a2a9b88366f41L, 0x63fe903e16f2c3bfL, + 0x1cdbd34fa751c0b0L, 0x0ce4fc6747c0558cL, 0x51ed72afb8bb49aaL, 0x20313ba13ca12c96L, + 0x271fa38f9ebd54c1L, 0x3696a5ac03a8eddeL, 0x05602be7df625702L, 0x11f1ac73790f7a9fL, + 0xa2836c099f0810bdL, 0xe5ac2e47caa532faL, 0xd9c000a66d39f681L, 0xd93d900e6f3d9d5fL, + 0x792c81c65b7900f2L, 0x5c5dce790ee20da1L, 0x74ff1950edec1aeeL, 0x71fc85fa1e277d8fL, + 0x0e77df17d6546cbcL, 0x07debad44816c3b4L, 0xbafa721581e92a70L, 0x8ab6fbe2ed27bba8L, + 0xe83243a20dea304aL, 0xaa85a63a84c00a07L, 0xde0e79917fc4153aL, 0x21bb445e83537896L, + 0xeedcac49fc0b433aL, 0xffb2926a810ae57aL, 0xf724be1f41d28702L, 0x79cb95746039bb3bL, + 0x5a54fe3742a00900L, 0xda4768d64922c04fL, 0x420396a84a339daeL, 0xa171e26ee5e8724eL, + 0x4c8da7c5d289c20aL, 0x9ebd79a1a8e94742L, 0x39235232b97e9782L, 0xb75df0be9bba7d80L, + 0x0c1d204dd87d48fcL, 0x8f81f3e7177266e8L, 0xe4a460b39e78d72bL, 0x50b98fa151e65351L, + 0xb7cb585c3ee1eddcL, 0x11cdad9a76ee1dc4L, 0xa38054a78595dc1cL, 0x92f09e2ec4978edcL, + 0xa8f0061b5efdabaaL, 0x04bcc4abc224d230L, 0xc58606738e692d46L, 0xdd2b27b565952433L, + 0x19e6ed1b740beec0L, 0xceadd49b2ef9891fL, 0x328178c28fe95cadL, 0xe5ad4c43afe02848L, + 0x03c0cb538cd967c0L, 0xec4352526d19a630L, 0x4c7e99389d39b031L, 0xf65dd05362c2deb6L, + 0xd1e70daf6879d28dL, 0xbe9f57db6309b265L, 0xa4b66f370b872bb7L, 0xe26896fbc6ee1fd5L, + 0xac705e661bfcf7c5L, 0xab4d0d07d7f09940L, 0x976417c06aeb6267L, 0x8161c684a6bd468cL, + 0xf77b6b9976dc4601L, 0xc6489b779a39c12cL, 0xb2aa58d5681cea1aL, 0x043b1b40f8c3e04cL, + 0x681fcbfadc845430L, 0xab8896c921ba8defL, 0x57aaf172606f37b2L, 0xc3735048cd5eb8d7L, + 0xa7078b96955631bdL, 0xdd6b3543aa187f33L, 0xc7103ea4a2a697fdL, 0x8d7b95f6ff1f7407L, + 0xe44f419e84709530L, 0xf340caa9132cbb0aL, 0x2ba407283143c66cL, 0xe1be240ca636c844L, + 0x90d32f2877ac08bcL, 0x5d26e6294b2c8673L, 0x4a6b2f5b27c87a44L, 0x961fb9043f76d34fL, + 0x0afee02d8d3c55d2L, 0x6228e3f48c42e5dcL, 0xc338e69ee6593675L, 0x853f74b16efb7bddL, + 0xd062f40bdd22e687L, 0x647164b9ab4c4190L, 0xf94689f67d598369L, 0x8e4b29d87a5012d7L, + 0xaf02b8b925656fbdL, 0x7a722a767179a630L, 0xb5c8afe937a75aceL, 0xfdb8e8d02d279372L, + 0x887ef700cb25fae1L, 0xcfe9bd912f72cabeL, 0xb1d4dedc24f978deL, 0x517522d38319cc2aL, + 0x7dd87b2b36aab798L, 0x579c4ff3046b5a04L, 0xf5c5975c5028b7a7L, 0x7094579d1000ec84L, + 0xbc8d5b1ea70a5291L, 0x161b2d783be8855cL, 0xd26d0b0d6d18279fL, 0x0be1945f02a78bd5L, + 0xb822a5a9e045415bL, 0x2fe9d68b1ccc3562L, 0xb2e375960033d14fL, 0x26aca04e49b4ff22L, + 0x732a81c862112aeaL, 0x8bd901ed6e4260b8L, 0xe839532c561ad5b0L, 0x8fb6e4d517a79b12L, + 0x0dd37f8c0be9b429L, 0xc8ad87ad12f1b1b0L, 0xc51f3aa62b90318bL, 0x031a7e8b86c1cefcL, + 0xa95547af2b70fc76L, 0x9cb3615c5a98801eL, 0xa387e3c3341d7032L, 0xa087ea52a1debaefL, + 0x16325ec9a2e6e835L, 0x587944a484c585ebL, 0xc8879033bde22eccL, 0xa39dbfce709c464aL, + 0x7acc010f99208774L, 0x98dd2973a096c5adL, 0x26458b51139f198cL, 0x2f5d19575e8c4f02L, + 0x726643f0d38af352L, 0x44d879b6d73e6e94L, 0xa68a03885c980abeL, 0x06048acd161c40c0L, + 0xa4dab8f89d405d28L, 0x7120c880cb04be18L, 0xa062ace22a1cf0cfL, 0x3901a9daf29704f4L, + 0xff08f3ed989db30aL, 0x6d22b13e874c67e9L, 0x80c6f35518d73f4dL, 0xc23c2a521aac6f29L, + 0x2e708fd83aaa42e0L, 0x7fc3780f55f1b0fdL, 0xabb3075c98cf87f2L, 0xb4df3f40f7c61143L, + 0x2a04418098a76d75L, 0x0d9eeee9509b2d37L, 0x6be8ae51f4b59cdcL, 0xe746cc7c00e4a2abL, + 0x785bc6df9cac597cL, 0x33cb6620ce8adc48L, 0xc1ba30739bffcef7L, 0x6d95771f18e503f7L, + 0xf7be3ae2e62652ffL, 0xc8d82ffd2a73c62bL, 0x8725a3ba5b110973L, 0x67ed6b9c724757ecL, + }, + // seed = 7 + { + 0xc0272d42c19ff3aeL, 0x4694228b43ea043bL, 0x5709a6ef8a462841L, 0xc9210a1e538805c9L, + 0x279b171196113ec2L, 0x859b769fc2d9e815L, 0x0d5d3125a2bf14d3L, 0x22bca1cfefa878baL, + 0x481b6bf58037bd83L, 0x4933ba8647728d22L, 0xf08c7b6b56f6e1b6L, 0x374e8af5a15407c7L, + 0xa95c4dc3d2487a5cL, 0x9b832808ff11e751L, 0xf2048507e9da01d5L, 0xa9c576189f544a4aL, + 0xf6c2a45b2e9d2b41L, 0x9b9874c9f10ecc2fL, 0x37d9b5f51f8c149eL, 0x93aead54c9de9467L, + 0x59cf0b4af262da23L, 0xe7e9929af18194b2L, 0x9df2644e33eb0178L, 0xde4122d6f0671938L, + 0xf005786c07f4800bL, 0xb1fc9d254b5d1039L, 0x0bf1088631f6dd7bL, 0x665623f0a4b8f0c7L, + 0x60f0113a9187db7cL, 0xfd7cceda4f0d23a6L, 0x26c01e9d89955940L, 0x33afa1dfc0f5a6a0L, + 0xeb77daf215e9283cL, 0xc7575214bf85edb4L, 0xeb0d804bf297e616L, 0x84bff4ffd564f747L, + 0xc4ac33189246f620L, 0x43ef61213ecc1005L, 0xcbbb0dea6cd96acdL, 0x8ed27abfa8cfcb05L, + 0x543b61529cb996b6L, 0xa5f987ca41ea5e59L, 0x3c50e0ac5254cb7aL, 0x4192b0446c06d1e6L, + 0x3e86592e21b45388L, 0xdb766f06fcc6e51eL, 0x0448ee36efe632dbL, 0x663c9db689253e35L, + 0x72e0bd4985331dd4L, 0xff501b5bf7d94e74L, 0xe911ce758e2113a8L, 0xec3a8d03a75a6ba4L, + 0xaf6b4b72f56edc83L, 0xf284857936c0a391L, 0x5ba6feff407d46f4L, 0x9d689c26de9d6702L, + 0x28c04a9083726b5dL, 0x2ccf4a627a029730L, 0x7b4719500d4f0c71L, 0x76470a9a7da250a8L, + 0xcc48409404a1c890L, 0xccefbdc7ec9a8055L, 0xe0db91bff3cc42d3L, 0x0532436426141254L, + 0xf2ee9325e6f0ff0bL, 0x149c20a5fbb28d9dL, 0xe71624cd8d2d14d4L, 0x8f01d4dc8cc2dd77L, + 0x29cf409b333015b7L, 0xba8bebd211884dd1L, 0xc3396635e8c8db1dL, 0x8ed0f6208d0528b8L, + 0x0d90b43fdd0ee334L, 0xd73c9a3333a044c7L, 0xa2595cd208dbdc38L, 0xae93cb264f940c09L, + 0x8e0538d8afb07a97L, 0x19115ec881385ba2L, 0xa886f9e6a8039c6aL, 0xcd5d62147ce3ecacL, + 0xaecdf9e0bb4969f7L, 0x2ddd631c53dcad10L, 0x73ad1c97b3412054L, 0xb08915fa2722efc6L, + 0x97966047e5067eb0L, 0x337f1675ed91445cL, 0xb3a833d150b96a0dL, 0x5940a98fe35e5e2eL, + 0xfd03cc354ed0d8ffL, 0x4e65b98291a8644aL, 0x14a259f2852a60b2L, 0x7648e3478c1e8e5fL, + 0xbc0fbef6d9a919b4L, 0xbec4302081346cf1L, 0x57d2ce7aa1c7c511L, 0x234c209d8f4e1ac3L, + 0x87cf80cc933ce443L, 0x7c262c616931e94eL, 0xc5e33b049cf9eddfL, 0x1a80790ed03ae51bL, + 0xf2e8b9494f7220cfL, 0x124cb59c14fff3ffL, 0xa8a06cbfdb86ce18L, 0x9068ef1f80b37653L, + 0x0c55417b8d90338fL, 0xcd579a523f6bcd30L, 0xa31bfe2476a8d2a9L, 0x1f8d142208094223L, + 0x332dc40a5203cfadL, 0xf8792fe5b2d33b4cL, 0x443bd9668bf9461eL, 0xc9019db0ace1409eL, + 0x781bea919a113e8bL, 0xb0f11d866abfbeecL, 0xcfe139a60db0c26aL, 0x869ab8721e6aa39eL, + 0xdb48a4977717837aL, 0x588a5ff151065b18L, 0xe4a251ea0028864dL, 0x7f0e43ba408a77c3L, + 0x65f66dd50a536135L, 0x6f49e934d9331c3eL, 0xb8d742e0f0fa6b09L, 0xe4e9b272deca2348L, + 0xaee132ff902f773cL, 0x43f658f7c2a0c90aL, 0x28cb4dbc76cc53eaL, 0x7d92253aa99ac39bL, + 0x4fea3d832370baabL, 0xb29e36936e51d78eL, 0xea10778712321064L, 0xff4f21f8ef274be2L, + 0x84eff18ddfa0933fL, 0xd0ec6a9f86c758a0L, 0xaf82e5973c431ae0L, 0x352023c00c045425L, + 0xad34d7bc4a2f8961L, 0xbdb4a02a24d4dee0L, 0x354a4846d97447cfL, 0x331a8b944d5bc19fL, + 0x5ce04f8e17909035L, 0x6497581bad8f4aabL, 0x07c503bba647111eL, 0x85f412ba78e1f7ffL, + 0x7f3b920fd20f4cffL, 0x424e1a9a4ce34e2fL, 0x3035e2d62e1b9f0aL, 0xef63114bff7b729aL, + 0xe86a05889ab6bb60L, 0xee0830cf095585a1L, 0x4a54f7fa47d9c94bL, 0x17daeece9fcb556aL, + 0xc506d3f391834c6fL, 0xb3f24be362e1af64L, 0xc435e4e23608efddL, 0xeeba9caaa4cc1768L, + 0x5a71f306daddc22dL, 0x18e5205f41eba1a0L, 0x7b29b4d1f6610925L, 0x065cb65a0258d9a9L, + 0x3e5ac8faa9fd1f95L, 0x3b362362c1ea0470L, 0xce0e4f6434db7a2eL, 0xf327341098de52f2L, + 0xcfca3b9e2a1992c3L, 0x7483bf9401233e41L, 0xbafbac531c6f9281L, 0x4b52dd71b2c106f8L, + 0xdf73b66e50b5a1f7L, 0x237aec0202a20283L, 0x23dd5be23dffdf2bL, 0xea9730731ee122efL, + 0x5cb3f846014fbcd3L, 0xc3b21c8ffdce9201L, 0x06a99a02f91a8760L, 0x721a81fa8fd7b7a3L, + 0x6aafcdddc53cbcd8L, 0xd03b464005a93bccL, 0x8212edc1b1669dcbL, 0x71f4c31364c31bc7L, + 0xfeeec0eba8772307L, 0x1948d00a13d88cf1L, 0x19064fd6d943ada8L, 0x4ec8d31722697bfdL, + 0x596d9a953a516609L, 0xc4cb4bff53507da2L, 0x1d59f3c5be36e4caL, 0xe5b4fc5bf6044c9bL, + 0x1bb74e052232f735L, 0x04e8a0db611ddd5dL, 0x8d04eaa009b421bfL, 0xa7878ae0ac0e6d58L, + 0x28c1030217cab2b3L, 0x827943767e56a883L, 0x28fce5fa02d22809L, 0xb30c322fffc8c58eL, + 0x1ca5a6a9f8066c5bL, 0xb24db5f1462b2513L, 0x02f653b89b7e5f6cL, 0xe31f8fb5d5f78eeeL, + 0x266acc514ed93501L, 0x936879d1c6fddcc4L, 0xcd51be3636af1952L, 0x3fdbb6fc332c78c8L, + 0x9eb656379fa73094L, 0x056146cc92fa0f96L, 0xed6c4f1836c027c3L, 0x021e0bb5d2113f2aL, + 0x8983e42ec1c626b3L, 0x73ea9bc6513ad9c9L, 0x0c904903b24f4247L, 0xacbac1e6243e2525L, + 0x0b1069a0c230fb06L, 0x77d709fca3fc1ce5L, 0x87ad0f65020947e6L, 0x555302641c53f4e6L, + 0x65ea87871fa9aaeeL, 0x58aaf4ecc1067bb4L, 0x1a66c48cc4c65b3fL, 0xca96aca48b2ea969L, + 0xa68eb70bad14de2bL, 0x5ccdb3d7e00a6f6eL, 0xe178fbfec73fe72fL, 0x2b63d6a16b83e890L, + 0x32fdb7a5330fbae0L, 0x2ab5803c8d1bf32cL, 0xda838388c1527c94L, 0x16a50bdc4de24acbL, + 0xe561301f134c074aL, 0xd7ae63d2816b4db1L, 0x036aabd4df0dd741L, 0xc5e0db8783435b9dL, + 0x9c4386cf0a07f3b2L, 0x6a72ac1aa56a13a1L, 0x299bbdb04bb20a23L, 0x138c1018fda16b81L, + 0x0e354f0b3bda49dfL, 0x9f4c295b23127437L, 0xd133ceb2bd561341L, 0xd8b4bfd5a526ac29L, + 0xcdd0a70ddc1c7bbdL, 0x81dce595bf572225L, 0x1c6f925c05f6efd7L, 0x8ae5097553856ea0L, + 0x3aabeaeef248f60dL, 0xd9005809d19a69e2L, 0x2a3a1a314311cc27L, 0x89bb2dc76b2b624aL, + 0x50a2a95d0412e289L, 0x9def8df564e68581L, 0xf49010a9b2e2ea5cL, 0x8602ae175d9ff3f0L, + 0xbf037e245369a618L, 0x8038164365f6e2b5L, 0xe2e1f6163b4e8d08L, 0x8df9314914f0857eL, + }, + }; +} diff --git a/parquet-column/src/main/java/org/apache/parquet/column/values/dictionary/DictionaryValuesWriter.java b/parquet-column/src/main/java/org/apache/parquet/column/values/dictionary/DictionaryValuesWriter.java index f4ed350e3a..0dcac0f716 100644 --- a/parquet-column/src/main/java/org/apache/parquet/column/values/dictionary/DictionaryValuesWriter.java +++ b/parquet-column/src/main/java/org/apache/parquet/column/values/dictionary/DictionaryValuesWriter.java @@ -82,6 +82,9 @@ public abstract class DictionaryValuesWriter extends ValuesWriter implements Req /* size in items of the dictionary at the end of last dictionary encoded page (in case the current page falls back to PLAIN) */ protected int lastUsedDictionarySize; + /* whether a page was dictionary encoded since the dictionary was reset, so one is due even if empty */ + protected boolean encodedAPage; + /* dictionary encoded values */ protected IntList encodedValues = new IntList(); @@ -113,6 +116,14 @@ protected DictionaryPage dictPage(ValuesWriter dictPageWriter) { return ret; } + /** + * A dictionary page without entries where pages were dictionary encoded with nulls alone, as Arrow + * C++ writes one, since such a page still needs a dictionary to be read. + */ + protected DictionaryPage emptyDictPage() { + return encodedAPage ? new DictionaryPage(BytesInput.empty(), 0, encodingForDictionaryPage) : null; + } + @Override public boolean shouldFallBack() { // if the dictionary reaches the max byte size or the values can not be encoded on 4 bytes anymore. @@ -173,6 +184,7 @@ public BytesInput getBytes() { // remember size of dictionary when we last wrote a page lastUsedDictionarySize = getDictionarySize(); lastUsedDictionaryByteSize = Math.toIntExact(dictionaryByteSize); + encodedAPage = true; return bytes; } catch (IOException e) { throw new ParquetEncodingException("could not encode the values", e); @@ -201,6 +213,7 @@ public void close() { public void resetDictionary() { lastUsedDictionaryByteSize = 0; lastUsedDictionarySize = 0; + encodedAPage = false; dictionaryTooBig = false; dictionaryByteSize = 0; clearDictionaryContent(); @@ -264,7 +277,7 @@ public DictionaryPage toDictPageAndClose() { } return dictPage(dictionaryEncoder); } - return null; + return emptyDictPage(); } @Override @@ -334,7 +347,7 @@ public DictionaryPage toDictPageAndClose() { } return dictPage(dictionaryEncoder); } - return null; + return emptyDictPage(); } } @@ -376,7 +389,7 @@ public DictionaryPage toDictPageAndClose() { } return dictPage(dictionaryEncoder); } - return null; + return emptyDictPage(); } @Override @@ -448,7 +461,7 @@ public DictionaryPage toDictPageAndClose() { } return dictPage(dictionaryEncoder); } - return null; + return emptyDictPage(); } @Override @@ -523,7 +536,7 @@ public DictionaryPage toDictPageAndClose() { } return dictPage(dictionaryEncoder); } - return null; + return emptyDictPage(); } @Override @@ -596,7 +609,7 @@ public DictionaryPage toDictPageAndClose() { } return dictPage(dictionaryEncoder); } - return null; + return emptyDictPage(); } @Override diff --git a/parquet-column/src/main/java/org/apache/parquet/column/values/factory/DefaultValuesWriterFactory.java b/parquet-column/src/main/java/org/apache/parquet/column/values/factory/DefaultValuesWriterFactory.java index 4c03e6b65e..80566ab47e 100644 --- a/parquet-column/src/main/java/org/apache/parquet/column/values/factory/DefaultValuesWriterFactory.java +++ b/parquet-column/src/main/java/org/apache/parquet/column/values/factory/DefaultValuesWriterFactory.java @@ -110,8 +110,12 @@ static ValuesWriter dictWriterWithFallBack( Encoding dataPageEncoding, ValuesWriter writerToFallBackTo) { if (parquetProperties.isDictionaryEnabled(path)) { - return FallbackValuesWriter.of( - dictionaryWriter(path, parquetProperties, dictPageEncoding, dataPageEncoding), writerToFallBackTo); + // Under content defined chunking the first page is cut by content, not size, so it is no + // sample to judge a dictionary on; like Arrow C++, fall back on the dictionary size alone. + return new FallbackValuesWriter<>( + dictionaryWriter(path, parquetProperties, dictPageEncoding, dataPageEncoding), + writerToFallBackTo, + !parquetProperties.isContentDefinedChunkingEnabled()); } else { return writerToFallBackTo; } diff --git a/parquet-column/src/main/java/org/apache/parquet/column/values/fallback/FallbackValuesWriter.java b/parquet-column/src/main/java/org/apache/parquet/column/values/fallback/FallbackValuesWriter.java index 41fe484f37..2b6a3ed2a5 100644 --- a/parquet-column/src/main/java/org/apache/parquet/column/values/fallback/FallbackValuesWriter.java +++ b/parquet-column/src/main/java/org/apache/parquet/column/values/fallback/FallbackValuesWriter.java @@ -62,11 +62,28 @@ public static Ported from Arrow C++ {@code parquet::internal::CalculateMask} and arrow-rs + * {@code CdcChunker::calculate_mask}; the arithmetic has to stay bit-identical to theirs. + */ +public final class RollingHashMask { + + /** A boundary needs this many rolling hash matches, one per gear hash table. */ + private static final int NUM_GEARHASH_TABLES = 8; + + private RollingHashMask() {} + + /** + * Derives the mask, validating the envelope. + * + *

A mask with the top {@code n} bits set matches a uniform hash with probability 1/2^n. The + * target is the average chunk size minus the skipped {@code minChunkSize}, divided by the + * {@value #NUM_GEARHASH_TABLES} matches a boundary needs. + * + * @param minChunkSize the minimum chunk size in bytes + * @param maxChunkSize the maximum chunk size in bytes + * @param normLevel the normalization level + * @return the rolling hash mask + * @throws IllegalArgumentException if the arguments cannot produce a usable mask + */ + public static long calculate(long minChunkSize, long maxChunkSize, int normLevel) { + Preconditions.checkArgument( + minChunkSize >= 0, "Invalid content defined chunking minimum chunk size (negative): %s", minChunkSize); + Preconditions.checkArgument( + maxChunkSize > minChunkSize, + "Invalid content defined chunking size range: maximum chunk size (%s) must be greater than minimum chunk size (%s)", + maxChunkSize, + minChunkSize); + + // Halve before adding, so that the sum cannot overflow; the references do overflow there. + long avgChunkSize = minChunkSize / 2 + maxChunkSize / 2 + (minChunkSize % 2 + maxChunkSize % 2) / 2; + long targetSize = (avgChunkSize - minChunkSize) / NUM_GEARHASH_TABLES; + int targetBits = Long.SIZE - Long.numberOfLeadingZeros(targetSize); + int maskBits = targetBits == 0 ? 0 : targetBits - 1; + int effectiveBits = maskBits - normLevel; + + // Java takes a long's shift count modulo 64, so an out-of-range width must be rejected here. + Preconditions.checkArgument( + effectiveBits >= 1 && effectiveBits <= 63, + "The content defined chunking mask must be between 1 and 63 bits but was %s" + + " (minimum chunk size %s, maximum chunk size %s, normalization level %s)", + effectiveBits, + minChunkSize, + maxChunkSize, + normLevel); + return -1L << (Long.SIZE - effectiveBits); + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/TestCdcOptions.java b/parquet-column/src/test/java/org/apache/parquet/column/TestCdcOptions.java new file mode 100644 index 0000000000..3a8f7775c5 --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/column/TestCdcOptions.java @@ -0,0 +1,72 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatThrownBy; + +import org.junit.jupiter.api.Test; + +public class TestCdcOptions { + + /** Boundaries only match Arrow C++ and arrow-rs while the defaults do. */ + @Test + public void defaultsMatchTheOtherImplementations() { + assertThat(CdcOptions.DEFAULT.getMinChunkSize()).isEqualTo(256 * 1024L); + assertThat(CdcOptions.DEFAULT.getMaxChunkSize()).isEqualTo(1024 * 1024L); + assertThat(CdcOptions.DEFAULT.getNormLevel()).isZero(); + } + + @Test + public void toStringNamesEverySetting() { + assertThat(options(64 * 1024, 256 * 1024, -1)) + .asString() + .contains("65536") + .contains("262144") + .contains("-1"); + } + + @Test + public void rejectsSizesAtTheSetterThatSaysWhichOneIsWrong() { + assertThatThrownBy(() -> CdcOptions.builder().withMinChunkSize(-1)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessage("Invalid content defined chunking minimum chunk size (negative): -1"); + assertThatThrownBy(() -> CdcOptions.builder().withMaxChunkSize(0)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessage("Invalid content defined chunking maximum chunk size (not positive): 0"); + } + + @Test + public void anUnusableSizeEnvelopeIsRejectedWhenTheOptionsAreBuilt() { + assertThatThrownBy(() -> CdcOptions.builder() + .withMinChunkSize(0) + .withMaxChunkSize(16) + .build()) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("between 1 and 63 bits"); + } + + private static CdcOptions options(long min, long max, int normLevel) { + return CdcOptions.builder() + .withMinChunkSize(min) + .withMaxChunkSize(max) + .withNormLevel(normLevel) + .build(); + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/TestParquetProperties.java b/parquet-column/src/test/java/org/apache/parquet/column/TestParquetProperties.java index 6d51f67cb0..12ee4bce1d 100644 --- a/parquet-column/src/test/java/org/apache/parquet/column/TestParquetProperties.java +++ b/parquet-column/src/test/java/org/apache/parquet/column/TestParquetProperties.java @@ -138,4 +138,49 @@ public void copyBuilder_preservesColumnCodecAndLevel() { assertThat(copy.getColumnCompressionLevel(colB)).isNull(); assertThat(copy.getColumnCodec(colC)).isNull(); } + + // ------------------------------------------- content defined chunking + + private static final CdcOptions CHUNKING = CdcOptions.builder() + .withMinChunkSize(4 * 1024) + .withMaxChunkSize(16 * 1024) + .build(); + + @Test + public void contentDefinedChunking_byDefault_isDisabled() { + assertThat(ParquetProperties.builder().build().isContentDefinedChunkingEnabled()) + .isFalse(); + } + + @Test + public void withContentDefinedChunking_nullOptions_throwsNullPointerException() { + assertThatThrownBy(() -> ParquetProperties.builder().withContentDefinedChunking(null)) + .isInstanceOf(NullPointerException.class) + .hasMessage("CdcOptions cannot be null"); + } + + @Test + public void withContentDefinedChunking_options_enablesChunkingAndSurvivesCopy() { + ParquetProperties original = + ParquetProperties.builder().withContentDefinedChunking(CHUNKING).build(); + ParquetProperties copy = ParquetProperties.copy(original).build(); + + assertThat(copy.isContentDefinedChunkingEnabled()).isTrue(); + assertThat(copy.getCdcOptions()).isSameAs(CHUNKING); + } + + @Test + public void copyBuilder_whileChunkingIsDisabled_stillPreservesTheOptions() { + ParquetProperties disabled = ParquetProperties.builder() + .withContentDefinedChunking(CHUNKING) + .withContentDefinedChunkingEnabled(false) + .build(); + assertThat(disabled.isContentDefinedChunkingEnabled()).isFalse(); + + ParquetProperties reEnabled = ParquetProperties.copy(disabled) + .withContentDefinedChunkingEnabled(true) + .build(); + assertThat(reEnabled.isContentDefinedChunkingEnabled()).isTrue(); + assertThat(reEnabled.getCdcOptions()).isSameAs(CHUNKING); + } } diff --git a/parquet-column/src/test/java/org/apache/parquet/column/impl/ChunkingTestSupport.java b/parquet-column/src/test/java/org/apache/parquet/column/impl/ChunkingTestSupport.java new file mode 100644 index 0000000000..6b22f4ccb9 --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/column/impl/ChunkingTestSupport.java @@ -0,0 +1,158 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import java.util.ArrayList; +import java.util.List; +import java.util.Random; +import java.util.function.IntPredicate; +import java.util.function.ObjIntConsumer; +import java.util.stream.Collectors; +import java.util.stream.IntStream; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.column.ColumnWriteStore; +import org.apache.parquet.column.ColumnWriter; +import org.apache.parquet.column.ParquetProperties; +import org.apache.parquet.column.page.DataPage; +import org.apache.parquet.column.page.mem.MemPageStore; +import org.apache.parquet.column.page.mem.MemPageWriter; +import org.apache.parquet.schema.MessageType; + +final class ChunkingTestSupport { + + private ChunkingTestSupport() {} + + static CdcOptions options(long min, long max, int normLevel) { + return CdcOptions.builder() + .withMinChunkSize(min) + .withMaxChunkSize(max) + .withNormLevel(normLevel) + .build(); + } + + /** A reproducible value stream; the same seed is the same data on any JVM. */ + static long[] newLongs(int count, long seed) { + return new Random(seed).longs(count).toArray(); + } + + /** The generator the golden boundary vectors are built from. */ + static long[] lcg(int count) { + long[] values = new long[count]; + long x = 0x243F6A8885A308D3L; + for (int i = 0; i < count; ++i) { + x = 6364136223846793005L * x + 1442695040888963407L; + values[i] = x; + } + return values; + } + + /** {@code original} with {@code inserted} spliced in at {@code at}, the edit under test. */ + static long[] insert(long[] original, int at, long[] inserted) { + long[] result = new long[original.length + inserted.length]; + System.arraycopy(original, 0, result, 0, at); + System.arraycopy(inserted, 0, result, at, inserted.length); + System.arraycopy(original, at, result, at + inserted.length, original.length - at); + return result; + } + + /** Properties with the position-based page limits lifted, so every page boundary is the chunker's. */ + static ParquetProperties unboundedProps(CdcOptions options) { + return ParquetProperties.builder() + .withPageRowCountLimit(Integer.MAX_VALUE) + .withPageSize(64 * 1024 * 1024) + .withDictionaryEncoding(false) + .withContentDefinedChunking(options) + .build(); + } + + /** + * Writes {@code rows} records through real {@link ColumnWriteStore}s, {@code writeRecord} writing + * record {@code i} to the schema's only column, and returns the pages that come out. A new store, + * made with the same {@code props}, takes over at each of {@code rowGroupStarts}, as for a new row + * group. + */ + static List writePages( + MessageType schema, + ParquetProperties props, + int rows, + ObjIntConsumer writeRecord, + int... rowGroupStarts) { + ColumnDescriptor path = schema.getColumns().get(0); + List pages = new ArrayList<>(); + int record = 0; + for (int end : IntStream.concat(IntStream.of(rowGroupStarts), IntStream.of(rows)) + .toArray()) { + MemPageStore pageStore = new MemPageStore(end - record); + ColumnWriteStore store = props.newColumnWriteStore(schema, pageStore); + ColumnWriter writer = store.getColumnWriter(path); + for (; record < end; ++record) { + writeRecord.accept(writer, record); + store.endRecord(); + } + store.flush(); + pages.addAll(((MemPageWriter) pageStore.getPageWriter(path)).getPages()); + } + return pages; + } + + /** + * Value counts per page, as {@code ChunkingColumnWriter} counts them: a boundary closes the page + * before the triplet that triggered it. + */ + static List chunkSizes(int count, IntPredicate offer) { + List sizes = new ArrayList<>(); + int current = 0; + for (int i = 0; i < count; ++i) { + if (offer.test(i) && current > 0) { + sizes.add(current); + current = 0; + } + current++; + } + if (current > 0) { + sizes.add(current); + } + return sizes; + } + + static List valueCounts(List pages) { + return pages.stream().map(DataPage::getValueCount).collect(Collectors.toList()); + } + + /** How many leading elements the two lists agree on. */ + static int sharedPrefix(List before, List after) { + int i = 0; + while (i < before.size() && i < after.size() && before.get(i).equals(after.get(i))) { + i++; + } + return i; + } + + /** How many trailing elements they agree on, beyond the {@code prefix} already counted. */ + static int sharedSuffix(List before, List after, int prefix) { + int i = 0; + while (i < before.size() - prefix + && i < after.size() - prefix + && before.get(before.size() - 1 - i).equals(after.get(after.size() - 1 - i))) { + i++; + } + return i; + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunker.java b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunker.java new file mode 100644 index 0000000000..d309fe7a21 --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunker.java @@ -0,0 +1,171 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import static org.apache.parquet.column.impl.ChunkingTestSupport.chunkSizes; +import static org.apache.parquet.column.impl.ChunkingTestSupport.insert; +import static org.apache.parquet.column.impl.ChunkingTestSupport.lcg; +import static org.apache.parquet.column.impl.ChunkingTestSupport.newLongs; +import static org.apache.parquet.column.impl.ChunkingTestSupport.options; +import static org.apache.parquet.column.impl.ChunkingTestSupport.sharedPrefix; +import static org.apache.parquet.column.impl.ChunkingTestSupport.sharedSuffix; +import static org.apache.parquet.schema.PrimitiveType.PrimitiveTypeName.INT64; +import static org.assertj.core.api.Assertions.assertThat; + +import java.nio.ByteBuffer; +import java.nio.ByteOrder; +import java.util.ArrayList; +import java.util.List; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.io.api.Binary; +import org.apache.parquet.schema.Types; +import org.junit.jupiter.api.Test; + +/** + * Cut behaviour of the chunker itself; {@code TestRollingHashMask} pins the mask. The expected + * chunk sizes are Arrow C++'s, not this implementation's: pyarrow writes the same values with the + * same 0/64-byte envelope to pages of these sizes. + */ +public class TestCdcChunker { + + /** A tiny envelope, so that few values make many chunks. JUnit gives each test a fresh one. */ + private final CdcChunker chunker = chunkerFor(options(0, 64, 0)); + + @Test + public void producesBoundariesWithinTheSizeEnvelope() { + long min = 512; + long max = 4096; + List sizes = chunkSizesOverLongs(newLongs(20_000, 1), min, max); + + // Every chunk but the last is a completed one, of eight-byte values. + assertThat(sizes).hasSizeGreaterThan(1); + assertThat(sizes.subList(0, sizes.size() - 1)) + .allSatisfy(size -> assertThat((long) size * 8).isBetween(min, max)); + } + + @Test + public void anEditPerturbsOnlyTheChunksAroundIt() { + long[] original = newLongs(50_000, 7); + List before = chunkSizesOverLongs(original, 512, 4096); + List after = chunkSizesOverLongs(insert(original, 20_000, newLongs(500, 99)), 512, 4096); + + int prefix = sharedPrefix(before, after); + assertThat(prefix).as("chunks shared before the edit").isPositive(); + assertThat(sharedSuffix(before, after, prefix)) + .as("nearly every chunk after the edit realigns") + .isGreaterThan((before.size() - prefix) * 9 / 10); + } + + @Test + public void aOneByteBinaryHashesTheSameAsAOneByteFixedWidthValue() { + CdcChunker viaBoolean = chunkerFor(options(0, 1024, 0)); + CdcChunker viaBinary = chunkerFor(options(0, 1024, 0)); + Binary one = Binary.fromConstantByteArray(new byte[] {1}); + Binary zero = Binary.fromConstantByteArray(new byte[] {0}); + List fromBoolean = new ArrayList<>(); + List fromBinary = new ArrayList<>(); + // Varying bytes: a constant byte stream drives the gear hash to a fixed point that never matches. + for (long x : lcg(4000)) { + boolean bit = (x & 0x100000000L) != 0; + fromBoolean.add(viaBoolean.offer(bit, 0, 0)); + fromBinary.add(viaBinary.offer(bit ? one : zero, 0, 0)); + } + assertThat(fromBinary).containsExactlyElementsOf(fromBoolean); + assertThat(fromBinary).contains(true); + } + + /** The minimum is a multiple of eight, so both skip the hash up to the same value. */ + @Test + public void anEightByteBinaryHashesTheSameAsALong() { + long[] values = newLongs(20_000, 19); + CdcChunker viaBinary = chunkerFor(options(512, 4096, 0)); + assertThat(chunkSizes( + values.length, + i -> viaBinary.offer( + Binary.fromConstantByteArray(ByteBuffer.allocate(8) + .order(ByteOrder.LITTLE_ENDIAN) + .putLong(values[i]) + .array()), + 0, + 0))) + .containsExactlyElementsOf(chunkSizesOverLongs(values, 512, 4096)); + } + + /** + * {@code MessageColumnIO} writes {@code writeNull(0, 0)} for a required field a record omits, + * where {@code definitionLevel == maxDef}; there is still no value to hash. The nulls are spread + * out because a gear hash forgets old bytes, so leading ones would perturb almost nothing. + */ + @Test + public void aNullOnARequiredColumnContributesNothingToTheHash() { + CdcChunker withNulls = chunkerFor(options(512, 4096, 0)); + CdcChunker withoutNulls = chunkerFor(options(512, 4096, 0)); + List actual = new ArrayList<>(); + List expected = new ArrayList<>(); + long[] values = newLongs(30_000, 17); + for (int i = 0; i < values.length; ++i) { + if (i % 100 == 0) { + actual.add(withNulls.offerNull(0, 0)); + expected.add(false); + } + actual.add(withNulls.offer(values[i], 0, 0)); + expected.add(withoutNulls.offer(values[i], 0, 0)); + } + assertThat(actual).containsExactlyElementsOf(expected); + assertThat(actual).contains(true); + } + + /** The 4-byte dispatch, which the golden vectors do not reach. */ + @Test + public void chunksIntegers() { + assertThat(chunkSizes(120, i -> chunker.offer(i * 0x9E3779B1, 0, 0))) + .containsExactly(9, 9, 13, 16, 2, 9, 12, 11, 15, 11, 10, 3); + } + + /** + * NaNs differing only in payload must hash differently, which {@code Float.floatToIntBits} would + * prevent. The payloads are quiet NaNs, because {@code Float.intBitsToFloat} may quieten a + * signalling one depending on the platform. + */ + @Test + public void floatsHashTheirRawBitsIncludingNanPayloads() { + assertThat(chunkSizes(120, i -> chunker.offer(Float.intBitsToFloat(0x7FC00001 + i), 0, 0))) + .containsExactly(12, 11, 9, 16, 11, 12, 10, 12, 10, 14, 3); + } + + /** As above, for doubles and {@code Double.doubleToLongBits}. */ + @Test + public void doublesHashTheirRawBitsIncludingNanPayloads() { + assertThat(chunkSizes(120, i -> chunker.offer(Double.longBitsToDouble(0x7FF8000000000001L + i), 0, 0))) + .containsExactly(7, 8, 2, 8, 8, 1, 8, 8, 1, 8, 8, 8, 8, 1, 8, 1, 8, 8, 4, 7); + } + + private static List chunkSizesOverLongs(long[] values, long min, long max) { + CdcChunker chunker = chunkerFor(options(min, max, 0)); + return chunkSizes(values.length, i -> chunker.offer(values[i], 0, 0)); + } + + /** A chunker for a flat required column. */ + private static CdcChunker chunkerFor(CdcOptions options) { + return new CdcChunker( + options, + new ColumnDescriptor(new String[] {"v"}, Types.required(INT64).named("v"), 0, 0)); + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunkerGoldenBoundaries.java b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunkerGoldenBoundaries.java new file mode 100644 index 0000000000..84c9bf8abf --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunkerGoldenBoundaries.java @@ -0,0 +1,233 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import static org.apache.parquet.column.impl.ChunkingTestSupport.lcg; +import static org.apache.parquet.column.impl.ChunkingTestSupport.options; +import static org.apache.parquet.column.impl.ChunkingTestSupport.unboundedProps; +import static org.apache.parquet.column.impl.ChunkingTestSupport.valueCounts; +import static org.apache.parquet.column.impl.ChunkingTestSupport.writePages; +import static org.assertj.core.api.Assertions.assertThat; + +import java.nio.ByteBuffer; +import java.util.Arrays; +import java.util.List; +import java.util.function.IntFunction; +import org.apache.parquet.bytes.BytesUtils; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.io.api.Binary; +import org.apache.parquet.schema.MessageType; +import org.apache.parquet.schema.MessageTypeParser; +import org.apache.parquet.schema.PrimitiveType.PrimitiveTypeName; +import org.junit.jupiter.api.Test; + +/** + * Pins page boundaries to the ones Arrow C++ produces for the same data, which the property tests + * elsewhere cannot: a self-consistent but different chunker passes those. + * + *

The expected values are the page value counts, as {@code parquet-cli pages} lists them, of the + * files this script writes with pyarrow 21 or later. No change to {@link CdcChunker} or + * {@link GearHashTable} may alter them. + * + *

{@code
+ * import pyarrow as pa, pyarrow.parquet as pq
+ *
+ * x, raw = 0x243F6A8885A308D3, []  # ChunkingTestSupport.lcg
+ * for _ in range(200_000):
+ *     x = (6364136223846793005 * x + 1442695040888963407) % 2**64
+ *     raw.append(x)
+ * signed = lambda v: v - 2**64 if v >= 2**63 else v
+ * cases = {
+ *     "required_int64": pa.array([signed(r) for r in raw], pa.int64()),
+ *     "optional_int64": pa.array([None if r % 8 == 0 else signed(r) for r in raw], pa.int64()),
+ *     "optional_utf8": pa.array([None if r % 8 == 0 else "row-%d" % (r % 1_000_000) for r in raw]),
+ *     "optional_list_int64": pa.array(
+ *         [None if r % 11 == 0 else [signed(r >> 16 * k) for k in range(r % 4)] for r in raw],
+ *         pa.list_(pa.int64())),
+ * }
+ * for name, v in cases.items():
+ *     field = pa.field("v", v.type, nullable=name != "required_int64")
+ *     pq.write_table(
+ *         pa.table({"v": v}, schema=pa.schema([field])), name + ".parquet",
+ *         compression="none", use_dictionary=False, use_compliant_nested_type=True,
+ *         data_page_size=1 << 30, max_rows_per_page=1 << 30,
+ *         use_content_defined_chunking={"min_chunk_size": 64 << 10, "max_chunk_size": 256 << 10})
+ * pq.write_table(
+ *     pa.table({"v": cases["required_int64"][:20_000]}, schema=pa.schema([pa.field("v", pa.int64(), False)])),
+ *     "tight_envelope_int64.parquet", compression="none", use_dictionary=False,
+ *     data_page_size=1 << 30, max_rows_per_page=1 << 30,
+ *     use_content_defined_chunking={"min_chunk_size": 0, "max_chunk_size": 4096, "norm_level": -2})
+ * }
+ */ +public class TestCdcChunkerGoldenBoundaries { + + private static final int ROWS = 200_000; + + private static final CdcOptions OPTIONS = options(64 * 1024, 256 * 1024, 0); + private static final CdcOptions TIGHT_OPTIONS = options(0, 4096, -2); + + // A binary value's eight bytes sit at PAD within PADDED, with filler either side. + private static final int PAD = 3; + private static final int PADDED = PAD + 8 + 5; + + /** No levels are hashed. */ + @Test + public void requiredInt64MatchesArrowCpp() { + assertThat(pageValueCounts("message t { required int64 v; }")) + .containsExactly( + 17054, 17725, 12281, 14801, 17490, 15449, 19582, 14469, 22343, 16007, 10731, 20511, 1557); + } + + /** Definition levels are hashed, and the value only where one is present. */ + @Test + public void optionalInt64MatchesArrowCpp() { + assertThat(pageValueCounts("message t { optional int64 v; }")) + .containsExactly( + 13812, 13460, 11759, 11204, 11876, 20212, 10075, 19599, 17988, 10262, 11450, 15229, 11339, + 11377, 10358); + } + + /** The same level path, with the binary rather than the fixed width dispatch. */ + @Test + public void optionalBinaryMatchesArrowCpp() { + assertThat(pageValueCounts("message t { optional binary v (STRING); }")) + .containsExactly( + 11588, 10905, 10978, 13364, 16800, 10173, 12896, 11080, 16760, 12559, 12333, 9652, 10732, 11297, + 12799, 12445, 3639); + } + + /** + * A {@code required binary} column of each value's eight little-endian bytes must reproduce the + * {@code required int64} vector, including bytes above 0x7F that a sign-extended table index would + * hash wrongly. Every case but the first surrounds the value with filler that must not be hashed. + */ + @Test + public void binaryValuesHashOnlyTheirOwnBytes() { + List expected = pageValueCounts("message t { required int64 v; }"); + long[] values = lcg(ROWS); + ByteBuffer direct = ByteBuffer.allocateDirect(ROWS * PADDED); + for (long value : values) { + direct.put(padded(value)); + } + + assertThat(binaryPageValueCounts(i -> Binary.fromConstantByteArray(BytesUtils.longToBytes(values[i])))) + .as("a whole byte array") + .containsExactlyElementsOf(expected); + assertThat(binaryPageValueCounts(i -> Binary.fromConstantByteArray(padded(values[i]), PAD, 8))) + .as("a slice of a byte array") + .containsExactlyElementsOf(expected); + assertThat(binaryPageValueCounts(i -> Binary.fromConstantByteBuffer(direct, i * PADDED + PAD, 8))) + .as("a window onto a direct buffer") + .containsExactlyElementsOf(expected); + } + + /** + * A selective mask and a small maximum, so the maximum size cut decides most boundaries (the + * 512-value chunks are 4096 bytes). The other vectors rarely reach that cut, so this is the one + * that pins it leaving the run counter alone. + */ + @Test + public void aMaximumSizeDominatedEnvelopeMatchesArrowCpp() { + assertThat(pageValueCounts("message t { required int64 v; }", 20_000, TIGHT_OPTIONS)) + .containsExactly( + 511, 512, 345, 512, 220, 512, 361, 512, 74, 512, 512, 199, 512, 382, 512, 512, 509, 512, 512, + 226, 512, 512, 2, 465, 512, 512, 349, 512, 460, 512, 98, 512, 512, 512, 132, 512, 512, 385, 508, + 512, 199, 512, 512, 455, 512, 415, 393); + } + + /** + * Both levels hashed, and cuts only at record starts. The only vector with a repetition level, + * so the only one that pins the order the levels are hashed in. + */ + @Test + public void optionalListOfInt64MatchesArrowCpp() { + assertThat(nestedPageValueCounts()) + .containsExactly( + 13336, 9928, 13972, 11483, 10503, 11533, 15500, 9599, 14956, 11109, 10846, 14956, 13858, 15861, + 13510, 16720, 11116, 10562, 9536, 11052, 12880, 12392, 10121, 14988, 13928, 10761, 11464); + } + + /** + * Writes {@code optional group v (LIST) { repeated group list { optional int64 element } }}, the + * three-level encoding pyarrow produces for {@code list}: a null list is one slot at + * definition level 0, an empty list one slot at level 1, and each element of a present list a + * slot at level 3, the first of them starting the record. + */ + private static List nestedPageValueCounts() { + MessageType schema = MessageTypeParser.parseMessageType( + "message t { optional group v (LIST) { repeated group list { optional int64 element; } } }"); + long[] values = lcg(ROWS); + return valueCounts(writePages(schema, unboundedProps(OPTIONS), ROWS, (writer, i) -> { + long x = values[i]; + if (Long.remainderUnsigned(x, 11) == 0) { + writer.writeNull(0, 0); // the list itself is null + } else { + int n = (int) Long.remainderUnsigned(x, 4); + if (n == 0) { + writer.writeNull(0, 1); // an empty list + } else { + for (int k = 0; k < n; k++) { + writer.write(x >>> (16 * k), k == 0 ? 0 : 1, 3); + } + } + } + })); + } + + /** + * The page value counts of one column of the generated data, with the position-based limits + * lifted here and on the reference side. + */ + private static List pageValueCounts(String schemaText) { + return pageValueCounts(schemaText, ROWS, OPTIONS); + } + + private static List pageValueCounts(String schemaText, int rows, CdcOptions options) { + MessageType schema = MessageTypeParser.parseMessageType(schemaText); + ColumnDescriptor path = schema.getColumns().get(0); + int maxDef = path.getMaxDefinitionLevel(); + boolean binary = path.getPrimitiveType().getPrimitiveTypeName() == PrimitiveTypeName.BINARY; + long[] values = lcg(rows); + return valueCounts(writePages(schema, unboundedProps(options), rows, (writer, i) -> { + long x = values[i]; + if (maxDef > 0 && Long.remainderUnsigned(x, 8) == 0) { + writer.writeNull(0, maxDef - 1); + } else if (binary) { + writer.write(Binary.fromString("row-" + Long.remainderUnsigned(x, 1_000_000L)), 0, maxDef); + } else { + writer.write(x, 0, maxDef); + } + })); + } + + /** The pages of a {@code required binary} column holding {@code value.apply(i)} in row {@code i}. */ + private static List binaryPageValueCounts(IntFunction value) { + MessageType schema = MessageTypeParser.parseMessageType("message t { required binary v; }"); + return valueCounts( + writePages(schema, unboundedProps(OPTIONS), ROWS, (writer, i) -> writer.write(value.apply(i), 0, 0))); + } + + private static byte[] padded(long value) { + byte[] bytes = new byte[PADDED]; + Arrays.fill(bytes, (byte) 0xA5); + System.arraycopy(BytesUtils.longToBytes(value), 0, bytes, PAD, 8); + return bytes; + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcWrite.java b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcWrite.java new file mode 100644 index 0000000000..85894e6b6f --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcWrite.java @@ -0,0 +1,500 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import static org.apache.parquet.column.impl.ChunkingTestSupport.chunkSizes; +import static org.apache.parquet.column.impl.ChunkingTestSupport.insert; +import static org.apache.parquet.column.impl.ChunkingTestSupport.newLongs; +import static org.apache.parquet.column.impl.ChunkingTestSupport.options; +import static org.apache.parquet.column.impl.ChunkingTestSupport.sharedPrefix; +import static org.apache.parquet.column.impl.ChunkingTestSupport.sharedSuffix; +import static org.apache.parquet.column.impl.ChunkingTestSupport.unboundedProps; +import static org.apache.parquet.column.impl.ChunkingTestSupport.valueCounts; +import static org.apache.parquet.column.impl.ChunkingTestSupport.writePages; +import static org.assertj.core.api.Assertions.assertThat; + +import java.io.IOException; +import java.nio.ByteBuffer; +import java.util.ArrayList; +import java.util.Arrays; +import java.util.HashSet; +import java.util.List; +import java.util.Set; +import java.util.TreeSet; +import java.util.function.ObjIntConsumer; +import java.util.function.ObjLongConsumer; +import java.util.function.UnaryOperator; +import java.util.stream.LongStream; +import java.util.stream.Stream; +import org.apache.parquet.bytes.BytesUtils; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.column.ColumnWriteStore; +import org.apache.parquet.column.ColumnWriter; +import org.apache.parquet.column.ParquetProperties; +import org.apache.parquet.column.page.DataPage; +import org.apache.parquet.column.page.DataPageV1; +import org.apache.parquet.column.page.mem.MemPageStore; +import org.apache.parquet.io.api.Binary; +import org.apache.parquet.schema.MessageType; +import org.apache.parquet.schema.MessageTypeParser; +import org.apache.parquet.schema.PrimitiveType.PrimitiveTypeName; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.Arguments; +import org.junit.jupiter.params.provider.EnumSource; +import org.junit.jupiter.params.provider.MethodSource; + +/** Covers {@link ChunkingColumnWriter} through a real {@link ColumnWriteStore} and the pages it writes. */ +public class TestCdcWrite { + + private static final CdcOptions OPTIONS = options(4 * 1024, 16 * 1024, 0); + + private static final MessageType REQUIRED = MessageTypeParser.parseMessageType("message m { required int64 v; }"); + private static final MessageType REQUIRED_BINARY = + MessageTypeParser.parseMessageType("message m { required binary v; }"); + private static final MessageType LIST = MessageTypeParser.parseMessageType("message m { repeated int64 v; }"); + + /** + * A value wider than the maximum chunk size asks for a boundary before itself even as the first + * value of a page, which {@link ColumnWriterBase#writePage()} would reject as empty. Arrow C++ + * filters out the same empty chunk. + */ + @Test + public void aBoundaryOnTheFirstValueOfAPageDoesNotWriteAnEmptyOne() { + List pages = writePages(REQUIRED_BINARY, unboundedProps(options(0, 32, 0)), 4, (writer, i) -> { + byte[] wide = new byte[64]; + Arrays.fill(wide, (byte) i); + writer.write(Binary.fromConstantByteArray(wide), 0, 0); + }); + + assertThat(valueCounts(pages)).containsExactly(1, 1, 1, 1); + } + + /** {@link ChunkingColumnWriter} hooks each {@code write} overload separately. */ + @ParameterizedTest + @MethodSource("primitiveColumns") + public void everyPrimitiveTypeGoesThroughTheChunker(String label, MessageType schema) { + // Pages must fall where a chunker offered the same typed values puts its boundaries: a hook that + // skips the chunker, cuts after the value, or offers another type moves them. + ColumnDescriptor path = schema.getColumns().get(0); + PrimitiveTypeName type = path.getPrimitiveType().getPrimitiveTypeName(); + long[] values = newLongs(60_000, 13); + CdcChunker chunker = new CdcChunker(OPTIONS, path); + List expected = chunkSizes(values.length, i -> offer(chunker, type, values[i])); + + assertThat(expected).hasSizeGreaterThan(1); + assertThat(valueCounts(writePages( + schema, unboundedProps(OPTIONS), values.length, (writer, i) -> write(writer, type, values[i])))) + .containsExactlyElementsOf(expected); + } + + static Stream primitiveColumns() { + return Stream.of( + Arguments.of("int32", MessageTypeParser.parseMessageType("message m { required int32 v; }")), + Arguments.of("int64", REQUIRED), + Arguments.of("float", MessageTypeParser.parseMessageType("message m { required float v; }")), + Arguments.of("double", MessageTypeParser.parseMessageType("message m { required double v; }")), + Arguments.of("boolean", MessageTypeParser.parseMessageType("message m { required boolean v; }")), + Arguments.of("binary", REQUIRED_BINARY), + Arguments.of( + "fixed_len_byte_array", + MessageTypeParser.parseMessageType("message m { required fixed_len_byte_array(8) v; }"))); + } + + /** Writes {@code value} as the physical type {@code type}. */ + private static void write(ColumnWriter writer, PrimitiveTypeName type, long value) { + switch (type) { + case INT32: + writer.write((int) value, 0, 0); + break; + case INT64: + writer.write(value, 0, 0); + break; + case FLOAT: + writer.write(Float.intBitsToFloat((int) value), 0, 0); + break; + case DOUBLE: + writer.write(Double.longBitsToDouble(value), 0, 0); + break; + case BOOLEAN: + writer.write((value & 1) == 0, 0, 0); + break; + default: + writer.write(Binary.fromConstantByteArray(BytesUtils.longToBytes(value)), 0, 0); + } + } + + private static boolean offer(CdcChunker chunker, PrimitiveTypeName type, long value) { + switch (type) { + case INT32: + return chunker.offer((int) value, 0, 0); + case INT64: + return chunker.offer(value, 0, 0); + case FLOAT: + return chunker.offer(Float.intBitsToFloat((int) value), 0, 0); + case DOUBLE: + return chunker.offer(Double.longBitsToDouble(value), 0, 0); + case BOOLEAN: + return chunker.offer((value & 1) == 0, 0, 0); + default: + return chunker.offer(Binary.fromConstantByteArray(BytesUtils.longToBytes(value)), 0, 0); + } + } + + /** The page limits count from the page start in the writer, so each column must keep one. */ + @Test + public void aStoreHandsOutOneChunkingWriterPerColumn() { + ColumnWriteStore store = unboundedProps(OPTIONS).newColumnWriteStore(REQUIRED, new MemPageStore(1)); + ColumnDescriptor path = REQUIRED.getColumns().get(0); + assertThat(store.getColumnWriter(path)).isSameAs(store.getColumnWriter(path)); + } + + /** + * The page size limit still cuts inside a chunk, but counted from the page start as in Arrow C++, + * so it cuts in the same places after an edit. Narrow values, whose default sized chunks run to + * several pages of the default size, keep deduplicating; checked on the store's row-count schedule + * instead, they shared no page at all. + */ + @Test + public void thePageSizeLimitCutsInTheSamePlacesAfterAnEdit() throws IOException { + ParquetProperties props = ParquetProperties.builder() + .withDictionaryEncoding(false) + .withPageRowCountLimit(Integer.MAX_VALUE) + .withContentDefinedChunking(CdcOptions.DEFAULT) + .build(); + long[] original = newLongs(1_500_000, 21); + List before = pageBytes(narrowPages(props, original)); + // Properties of its own, as for another file: they hold the chunking state. + List after = pageBytes( + narrowPages(ParquetProperties.copy(props).build(), insert(original, 10_000, newLongs(100, 22)))); + + Set beforeSet = new HashSet<>(before); + assertThat(before).hasSizeGreaterThan(5); + assertThat(after.stream().filter(beforeSet::contains).count()) + .as("pages of the edited column that are also in the original") + .isGreaterThan(after.size() / 2); + } + + /** The row count limit still applies inside a chunk, at exactly the limit, as in Arrow C++. */ + @Test + public void theRowCountLimitAppliesInsideAChunk() { + ParquetProperties props = ParquetProperties.builder() + .withDictionaryEncoding(false) + .withContentDefinedChunking(CdcOptions.DEFAULT) + .build(); + assertThat(valueCounts(narrowPages(props, newLongs(200_000, 23)))) + .hasSizeGreaterThan(5) + .allSatisfy( + count -> assertThat(count).isLessThanOrEqualTo(ParquetProperties.DEFAULT_PAGE_ROW_COUNT_LIMIT)) + .contains(ParquetProperties.DEFAULT_PAGE_ROW_COUNT_LIMIT); + } + + /** + * A row group's store continues the chunking where the previous one, made with the same properties, + * left it: row groups add page breaks but move no boundary. Here they start a record before and at + * every chunk boundary, where a chunker that restarts its hash, match run, pending match or size + * misses the boundary or ends the next chunk elsewhere. + */ + @Test + public void eachRowGroupContinuesTheChunkingOfThePrevious() { + long[] values = newLongs(20_000, 29); + // Lists of zero to three values, so that matches inside a record carry over to the next. + ObjIntConsumer record = (writer, i) -> { + int length = (int) (values[i] & 3); + if (length == 0) { + writer.writeNull(0, 0); + } + for (int j = 0; j < length; ++j) { + writer.write(values[i] + j, j == 0 ? 0 : 1, 1); + } + }; + int[] recordStarts = new int[values.length + 1]; + for (int i = 0; i < values.length; ++i) { + recordStarts[i + 1] = recordStarts[i] + Math.max(1, (int) (values[i] & 3)); + } + Set pageEnds = pageEnds(writePages(LIST, unboundedProps(OPTIONS), values.length, record)); + Set rowGroupStarts = new TreeSet<>(); + for (int end : pageEnds) { + int chunkStart = Arrays.binarySearch(recordStarts, end); + if (chunkStart < values.length) { + rowGroupStarts.add(chunkStart - 1); + rowGroupStarts.add(chunkStart); + } + } + assertThat(rowGroupStarts).hasSizeGreaterThan(20); + + Set expected = new TreeSet<>(pageEnds); + rowGroupStarts.forEach(start -> expected.add(recordStarts[start])); + assertThat(pageEnds(writePages( + LIST, + unboundedProps(OPTIONS), + values.length, + record, + rowGroupStarts.stream().mapToInt(Integer::intValue).toArray()))) + .containsExactlyElementsOf(expected); + } + + /** Where each page ends, counted in levels from the first. */ + private static Set pageEnds(List pages) { + Set ends = new TreeSet<>(); + int end = 0; + for (DataPage page : pages) { + end += page.getValueCount(); + ends.add(end); + } + return ends; + } + + /** On a list column the row count limit ends pages at record starts only. */ + @Test + public void theRowCountLimitEndsPagesAtRecordStarts() { + ParquetProperties props = ParquetProperties.builder() + .withDictionaryEncoding(false) + .withPageRowCountLimit(100) + .withContentDefinedChunking(options(1 << 30, 1L << 31, 0)) + .build(); + assertThat(valueCounts(writePages(LIST, props, 1000, (writer, i) -> { + writer.write((long) i, 0, 1); + writer.write((long) i, 1, 1); + writer.write((long) i, 1, 1); + }))) + .containsExactly(300, 300, 300, 300, 300, 300, 300, 300, 300, 300); + } + + /** + * The size limits are checked at the first record start after each batch of 1024 levels, as in + * Arrow C++, so a page value count limit of 5000 ends pages at 5120 values. + */ + @Test + public void theSizeLimitsAreCheckedOncePerBatch() { + ParquetProperties props = ParquetProperties.builder() + .withDictionaryEncoding(false) + .withPageRowCountLimit(Integer.MAX_VALUE) + .withPageValueCountThreshold(5000) + .withContentDefinedChunking(options(1 << 30, 1L << 31, 0)) + .build(); + assertThat(valueCounts(writePages(REQUIRED, props, 4 * 5120, (writer, i) -> writer.write((long) i, 0, 0)))) + .containsExactly(5120, 5120, 5120, 5120); + } + + /** + * The size limit applies inside a chunk too, checked once per batch of 1024 values as in Arrow + * C++: 100-byte values in chunks of up to 8 MiB still come out in pages of about the 1 MiB limit, + * except those that end where a chunk does. + */ + @Test + public void thePageSizeLimitAppliesInsideAChunk() { + ParquetProperties props = ParquetProperties.builder() + .withDictionaryEncoding(false) + .withPageRowCountLimit(Integer.MAX_VALUE) + .withContentDefinedChunking(options(4 << 20, 8 << 20, 0)) + .build(); + byte[] value = new byte[100]; + List pages = writePages( + REQUIRED_BINARY, + props, + 100_000, + (writer, i) -> writer.write(Binary.fromConstantByteArray(value), 0, 0)); + + // No page runs more than one batch past the limit, and inside chunks the limit, not the chunker, + // ends most of them. + int threshold = props.getPageSizeThreshold(); + assertThat(pages).allSatisfy(page -> assertThat(page.getUncompressedSize()) + .isLessThanOrEqualTo(threshold + 1024 * (value.length + 4))); + assertThat(pages) + .filteredOn(page -> page.getUncompressedSize() >= threshold) + .hasSizeGreaterThan(5); + } + + /** One- and two-byte values: the narrow case, whose chunks run to the most pages. */ + private static List narrowPages(ParquetProperties props, long[] values) { + return writePages(REQUIRED_BINARY, props, values.length, (writer, i) -> { + byte[] bytes = BytesUtils.longToBytes(values[i]); + writer.write(Binary.fromConstantByteArray(bytes, 0, (values[i] & 1) == 0 ? 1 : 2), 0, 0); + }); + } + + private static List pageBytes(List pages) throws IOException { + List bytes = new ArrayList<>(); + for (DataPage page : pages) { + bytes.add( + Binary.fromConstantByteArray(((DataPageV1) page).getBytes().toByteArray())); + } + return bytes; + } + + /** Records edited. */ + private static final int EDIT = 50; + + /** A column of each shape, its records written from a seed each. */ + private enum Column { + INT32(200_000, "required int32 v;", (writer, seed) -> writer.write((int) seed, 0, 0)), + OPTIONAL_DOUBLE(100_000, "optional double v;", (writer, seed) -> { + if (seed % 5 == 0) { + writer.writeNull(0, 0); + } else { + writer.write(seed / 3.0, 0, 1); + } + }), + BOOLEAN(900_000, "required boolean v;", (writer, seed) -> writer.write((seed & 1) == 0, 0, 0)), + OPTIONAL_BINARY(75_000, "optional binary v;", (writer, seed) -> { + if (seed % 5 == 0) { + writer.writeNull(0, 0); + } else { + writer.write(Binary.fromString(Long.toString(seed, 36)), 0, 1); + } + }), + FIXED_LEN_BYTE_ARRAY( + 50_000, + "required fixed_len_byte_array(16) v;", + (writer, seed) -> writer.write( + Binary.fromConstantByteArray(ByteBuffer.allocate(16) + .putLong(seed) + .putLong(~seed) + .array()), + 0, + 0)), + LIST( + 60_000, + "optional group l (LIST) { repeated group list { optional int32 element; } }", + (writer, seed) -> writeList(writer, seed, 3)), + LIST_OF_GROUPS( + 60_000, + "optional group l (LIST) { repeated group list { optional group element { optional int32 f0; } } }", + (writer, seed) -> writeList(writer, seed, 4)); + + private final int records; + private final MessageType schema; + private final ObjLongConsumer record; + + Column(int records, String fields, ObjLongConsumer record) { + this.records = records; + this.schema = MessageTypeParser.parseMessageType("message m { " + fields + " }"); + this.record = record; + } + + List pageBytes(long[] seeds) throws IOException { + return TestCdcWrite.pageBytes(writePages( + schema, + unboundedProps(options(1024, 4096, 0)), + seeds.length, + (writer, i) -> record.accept(writer, seeds[i]))); + } + + /** A null or empty list, or one of one to six elements, some of them null at every level. */ + private static void writeList(ColumnWriter writer, long seed, int maxDef) { + int kind = (int) (seed >>> 61); + if (kind < 2) { + writer.writeNull(0, kind); + } + for (int j = 0; j < kind - 1; ++j) { + int def = Math.max(2, maxDef - (int) ((seed >>> (4 * j)) & 3)); + if (def == maxDef) { + writer.write((int) (seed >>> (8 * j)), j == 0 ? 0 : 1, def); + } else { + writer.writeNull(j == 0 ? 0 : 1, def); + } + } + } + } + + private enum Edit { + INSERT(seeds -> insert(seeds, seeds.length / 2, newLongs(EDIT, 41))), + DELETE(seeds -> LongStream.concat( + Arrays.stream(seeds, 0, seeds.length / 2), + Arrays.stream(seeds, seeds.length / 2 + EDIT, seeds.length)) + .toArray()), + UPDATE(seeds -> { + long[] updated = seeds.clone(); + System.arraycopy(newLongs(EDIT, 41), 0, updated, seeds.length / 2, EDIT); + return updated; + }), + PREPEND(seeds -> insert(seeds, 0, newLongs(EDIT, 41))), + APPEND(seeds -> insert(seeds, seeds.length, newLongs(EDIT, 41))); + + private final UnaryOperator apply; + + Edit(UnaryOperator apply) { + this.apply = apply; + } + } + + /** + * An edit of a few records changes only the pages around it, in every shape of column: every page + * before it and all but a few after it are byte for byte as before, so they deduplicate. That a few + * change is the algorithm's, as in Arrow C++: the chunking realigns only once a chunk ends in the + * same place again, as the eight matches that end one count from where it began. + */ + @ParameterizedTest + @EnumSource(Column.class) + public void anEditChangesOnlyThePagesAroundIt(Column column) throws IOException { + long[] original = newLongs(column.records, 37); + List before = column.pageBytes(original); + assertThat(before).hasSizeGreaterThan(350); + for (Edit edit : Edit.values()) { + List after = column.pageBytes(edit.apply.apply(original)); + int prefix = sharedPrefix(before, after); + int suffix = sharedSuffix(before, after, prefix); + assertThat(Math.max(before.size(), after.size()) - prefix - suffix) + .as("pages changed of %s by %s", before.size(), edit) + .isLessThanOrEqualTo(20); + } + } + + @Test + public void onlyChunkingRealignsAfterAnInsertion() throws IOException { + // Both writers share the pages before the insertion; only chunking shares any after it. Page + // bytes rather than value counts, because a position-based writer's later pages keep their + // counts but shift their contents. + long[] original = newLongs(60_000, 11); + long[] edited = insert(original, 25_000, newLongs(300, 42)); + + assertThat(sharedBeyondThePrefix(boundedPageBytes(original, false), boundedPageBytes(edited, false))) + .as("a position-based writer realigns nothing after an insertion") + .isZero(); + + List chunkedBefore = boundedPageBytes(original, true); + assertThat(chunkedBefore).hasSizeGreaterThan(10); + assertThat(sharedBeyondThePrefix(chunkedBefore, boundedPageBytes(edited, true))) + .as("chunking shares pages from after the insertion too") + .isPositive(); + } + + /** How many of {@code before}'s pages past the common leading run reappear anywhere in {@code after}. */ + private static long sharedBeyondThePrefix(List before, List after) { + int prefix = sharedPrefix(before, after); + return before.subList(prefix, before.size()).stream() + .filter(after::contains) + .count(); + } + + /** Every page's bytes, with a page size small enough that the position-based limits make many pages. */ + private static List boundedPageBytes(long[] values, boolean chunking) throws IOException { + ParquetProperties props = ParquetProperties.builder() + .withPageSize(16 * 1024) + .withMinRowCountForPageSizeCheck(1) + .withDictionaryEncoding(false) + .withContentDefinedChunking(OPTIONS) + .withContentDefinedChunkingEnabled(chunking) + .build(); + + return pageBytes(writePages(REQUIRED, props, values.length, (writer, i) -> writer.write(values[i], 0, 0))); + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/impl/TestGearHashTable.java b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestGearHashTable.java new file mode 100644 index 0000000000..9a636b815a --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestGearHashTable.java @@ -0,0 +1,54 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import static org.assertj.core.api.Assertions.assertThat; + +import java.nio.ByteBuffer; +import java.security.MessageDigest; +import java.security.NoSuchAlgorithmException; +import java.util.Arrays; +import org.junit.jupiter.api.Test; + +/** + * Pins {@link GearHashTable} to the specification Arrow C++ and arrow-rs generate their identical + * tables from, so a transcription slip fails here instead of silently changing every boundary. + */ +public class TestGearHashTable { + + /** + * One table per match a boundary needs; entry {@code [seed][n]} is the first eight bytes, + * big-endian, of the MD5 of 64 bytes of {@code seed} followed by 64 bytes of {@code n}. + */ + @Test + public void tableMatchesTheMd5Specification() throws NoSuchAlgorithmException { + MessageDigest md5 = MessageDigest.getInstance("MD5"); + long[][] expected = new long[8][256]; + for (int seed = 0; seed < expected.length; ++seed) { + for (int n = 0; n < 256; ++n) { + byte[] input = new byte[128]; + Arrays.fill(input, 0, 64, (byte) seed); + Arrays.fill(input, 64, 128, (byte) n); + expected[seed][n] = ByteBuffer.wrap(md5.digest(input)).getLong(); + } + } + + assertThat(GearHashTable.TABLE).isDeepEqualTo(expected); + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/values/dictionary/TestDictionary.java b/parquet-column/src/test/java/org/apache/parquet/column/values/dictionary/TestDictionary.java index 13033404ce..787511a6ba 100644 --- a/parquet-column/src/test/java/org/apache/parquet/column/values/dictionary/TestDictionary.java +++ b/parquet-column/src/test/java/org/apache/parquet/column/values/dictionary/TestDictionary.java @@ -30,13 +30,17 @@ import java.io.IOException; import java.nio.ByteBuffer; import java.nio.charset.StandardCharsets; +import java.util.ArrayList; +import java.util.List; import org.apache.parquet.bytes.ByteBufferInputStream; import org.apache.parquet.bytes.BytesInput; import org.apache.parquet.bytes.DirectByteBufferAllocator; import org.apache.parquet.bytes.TrackingByteBufferAllocator; +import org.apache.parquet.column.CdcOptions; import org.apache.parquet.column.ColumnDescriptor; import org.apache.parquet.column.Dictionary; import org.apache.parquet.column.Encoding; +import org.apache.parquet.column.ParquetProperties; import org.apache.parquet.column.page.DictionaryPage; import org.apache.parquet.column.values.ValuesReader; import org.apache.parquet.column.values.ValuesWriter; @@ -52,6 +56,7 @@ import org.apache.parquet.column.values.plain.PlainValuesWriter; import org.apache.parquet.io.api.Binary; import org.apache.parquet.schema.PrimitiveType.PrimitiveTypeName; +import org.apache.parquet.schema.Types; import org.junit.jupiter.api.AfterEach; import org.junit.jupiter.api.BeforeEach; import org.junit.jupiter.api.Test; @@ -210,6 +215,68 @@ public void testBinaryDictionaryFallBack() throws IOException { } } + /** + * Without the first-page judgement, as content defined chunking asks for, a dictionary falls back + * only on its size limit, as Arrow C++ decides it, even after a first page of nulls alone. + */ + @Test + public void fallsBackOnTheSizeLimitAloneWithoutAFirstPageJudgement() throws IOException { + // One page of 100 distinct values is less than its dictionary costs. + assertThat(pageEncodings(true, 1 << 20, 100)).containsExactly(PLAIN, PLAIN, PLAIN); + assertThat(pageEncodings(false, 1 << 20, 100)) + .containsExactly(PLAIN_DICTIONARY, PLAIN_DICTIONARY, PLAIN_DICTIONARY); + // 150 entries of 8 bytes fill the limit during the second page. + assertThat(pageEncodings(false, 150 * 8, 100)).containsExactly(PLAIN_DICTIONARY, PLAIN, PLAIN); + assertThat(pageEncodings(false, 1 << 20, 0)) + .containsExactly(PLAIN_DICTIONARY, PLAIN_DICTIONARY, PLAIN_DICTIONARY); + } + + /** The writer factory judges the first page unless content defined chunking is enabled. */ + @Test + public void theFactoryJudgesTheFirstPageUnlessChunking() { + ColumnDescriptor path = + new ColumnDescriptor(new String[] {"v"}, Types.required(BINARY).named("v"), 0, 0); + assertThat(firstPageEncoding(ParquetProperties.builder().build(), path)).isEqualTo(PLAIN); + assertThat(firstPageEncoding( + ParquetProperties.builder() + .withContentDefinedChunking(CdcOptions.DEFAULT) + .build(), + path)) + .isEqualTo(PLAIN_DICTIONARY); + } + + /** The encoding of a first page of 100 distinct values, less than their dictionary costs. */ + private static Encoding firstPageEncoding(ParquetProperties props, ColumnDescriptor path) { + try (ValuesWriter writer = props.newValuesWriter(path)) { + for (int i = 0; i < 100; i++) { + writer.writeBytes(Binary.fromString(String.format("v%03d", i))); + } + writer.getBytes(); + return writer.getEncoding(); + } + } + + /** Three pages of {@code valuesPerPage} distinct four-character values, 8 bytes of raw data each. */ + private List pageEncodings(boolean judgeFirstPage, int maxDictionaryByteSize, int valuesPerPage) + throws IOException { + List encodings = new ArrayList<>(); + try (FallbackValuesWriter cw = new FallbackValuesWriter<>( + new PlainBinaryDictionaryValuesWriter( + maxDictionaryByteSize, PLAIN_DICTIONARY, PLAIN_DICTIONARY, allocator), + new PlainValuesWriter(100, 500, allocator), + judgeFirstPage)) { + for (int page = 0; page < 3; page++) { + for (int i = 0; i < valuesPerPage; i++) { + cw.writeBytes(Binary.fromString(String.format("v%03d", page * valuesPerPage + i))); + } + cw.getBytes(); + encodings.add(cw.getEncoding()); + cw.reset(); + } + } + return encodings; + } + @Test public void testBinaryDictionaryIntegerOverflow() { Binary mock = Mockito.mock(Binary.class); diff --git a/parquet-column/src/test/java/org/apache/parquet/internal/column/chunking/TestRollingHashMask.java b/parquet-column/src/test/java/org/apache/parquet/internal/column/chunking/TestRollingHashMask.java new file mode 100644 index 0000000000..412ff38a2c --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/internal/column/chunking/TestRollingHashMask.java @@ -0,0 +1,112 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.internal.column.chunking; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatCode; +import static org.assertj.core.api.Assertions.assertThatThrownBy; + +import org.junit.jupiter.api.Test; + +/** + * Ported from Arrow C++'s {@code TestCDC.RollingHashMaskCalculation} and + * {@code TestCDC.ChunkSizeParameterValidation} ({@code cpp/src/parquet/chunker_internal_test.cc}), + * with a few edge cases of its own. + */ +public class TestRollingHashMask { + + private static final long MIN_SIZE = 256 * 1024L; + private static final long MAX_SIZE = 1024 * 1024L; + + @Test + public void maskMatchesTheReferenceForEachNormalizationLevel() { + assertThat(RollingHashMask.calculate(MIN_SIZE, MAX_SIZE, 0)).isEqualTo(0xFFFE000000000000L); + assertThat(RollingHashMask.calculate(MIN_SIZE, MAX_SIZE, 1)).isEqualTo(0xFFFC000000000000L); + assertThat(RollingHashMask.calculate(MIN_SIZE, MAX_SIZE, 2)).isEqualTo(0xFFF8000000000000L); + assertThat(RollingHashMask.calculate(MIN_SIZE, MAX_SIZE, 3)).isEqualTo(0xFFF0000000000000L); + assertThat(RollingHashMask.calculate(MIN_SIZE, MAX_SIZE, -1)).isEqualTo(0xFFFF000000000000L); + } + + @Test + public void maskMatchesTheReferenceAtTheEdgesOfItsRange() { + assertThat(RollingHashMask.calculate(0, 32, 0)).isEqualTo(0x8000000000000000L); + assertThat(RollingHashMask.calculate(0, 64, 0)).isEqualTo(0xC000000000000000L); + assertThat(RollingHashMask.calculate(0, 16, -1)).isEqualTo(0x8000000000000000L); + // A zero target clamps to no bits rather than minus one. + assertThat(RollingHashMask.calculate(0, 8, -1)).isEqualTo(0x8000000000000000L); + assertThat(RollingHashMask.calculate(128, 384, -59)).isEqualTo(0xFFFFFFFFFFFFFFFEL); + } + + @Test + public void rejectsANegativeMinimum() { + assertThatThrownBy(() -> RollingHashMask.calculate(-1, 1024, 0)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessage("Invalid content defined chunking minimum chunk size (negative): -1"); + } + + @Test + public void rejectsARangeThatIsNotAscending() { + assertThatThrownBy(() -> RollingHashMask.calculate(1024, 512, 0)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("must be greater than"); + assertThatThrownBy(() -> RollingHashMask.calculate(32, 32, 0)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("must be greater than"); + } + + @Test + public void rejectsASizeRangeTooNarrowForTheNormalizationLevel() { + // With eight tables the mask needs at least one bit, so the min/max gap must be at least 32 at + // normLevel 0, 64 at 1 and 128 at 2. + assertThatThrownBy(() -> RollingHashMask.calculate(0, 16, 0)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("between 1 and 63 bits"); + assertThatCode(() -> RollingHashMask.calculate(0, 32, 0)).doesNotThrowAnyException(); + assertThatThrownBy(() -> RollingHashMask.calculate(32, 48, 0)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("between 1 and 63 bits"); + assertThatCode(() -> RollingHashMask.calculate(32, 64, 0)).doesNotThrowAnyException(); + + assertThatThrownBy(() -> RollingHashMask.calculate(1, 33, 1)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("between 1 and 63 bits"); + assertThatCode(() -> RollingHashMask.calculate(1, 65, 1)).doesNotThrowAnyException(); + + assertThatThrownBy(() -> RollingHashMask.calculate(0, 123, 2)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("between 1 and 63 bits"); + assertThatCode(() -> RollingHashMask.calculate(0, 128, 2)).doesNotThrowAnyException(); + + assertThatThrownBy(() -> RollingHashMask.calculate(128, 384, -60)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("between 1 and 63 bits"); + } + + @Test + public void acceptsALargeEnvelope() { + assertThatCode(() -> RollingHashMask.calculate(1024 * 1024L * 1024L, 2L * 1024 * 1024 * 1024, 0)) + .doesNotThrowAnyException(); + } + + /** A deliberate deviation: Arrow C++ and arrow-rs overflow here. */ + @Test + public void handlesAnEnvelopeThatWouldOverflowTheAverage() { + assertThat(RollingHashMask.calculate(256 * 1024L, Long.MAX_VALUE, 0)).isEqualTo(0xFFFFFFFFFFFFFFC0L); + } +} diff --git a/parquet-hadoop/README.md b/parquet-hadoop/README.md index 13b5132701..0c357edb73 100644 --- a/parquet-hadoop/README.md +++ b/parquet-hadoop/README.md @@ -284,6 +284,42 @@ true if the reader is using a `DirectByteBufferAllocator` --- +**Property:** `parquet.page.content-defined-chunking.enabled` +**Description:** EXPERIMENTAL: Whether data pages also end at boundaries derived from a rolling hash of the +column's values, so that files sharing a run of values share byte-identical pages a content addressable storage +system can deduplicate. The file needs no reader support. +As in Arrow C++, `parquet.page.size` and `parquet.page.row.count.limit` still cut pages inside a chunk, counted from +the page start, so they add pages without moving any after an edit; and a column falls back from dictionary encoding +only when its dictionary outgrows `parquet.dictionary.page.size`, not on its first page. Dictionary ids follow the order +values first appear in, so an edit that adds new values renumbers the later ones within its row group: columns of many +distinct values deduplicate best with `parquet.enable.dictionary` off. +**Default value:** `false` + +--- + +**Property:** `parquet.page.content-defined-chunking.min.size` +**Description:** EXPERIMENTAL: The minimum content defined chunk size in bytes. No chunk is shorter but a file's +last; pages can be, where a row group or a page limit ends one. +**Default value:** `262144` (256 KiB) + +--- + +**Property:** `parquet.page.content-defined-chunking.max.size` +**Description:** EXPERIMENTAL: The maximum content defined chunk size in bytes; a chunk ends here whatever the +rolling hash says. Very small sizes produce very many pages, and encrypted files cannot exceed 32767 pages per column +chunk. +**Default value:** `1048576` (1 MiB) + +--- + +**Property:** `parquet.page.content-defined-chunking.norm.level` +**Description:** EXPERIMENTAL: The normalization level of the rolling hash mask. Raising it makes a boundary more +likely, tightening the chunk size distribution and improving deduplication at the cost of more small pages; lowering +it does the reverse. Values outside `[-3, 3]` are not useful. +**Default value:** `0` + +--- + **Property:** `parquet.page.write-checksum.enabled` **Description:** Whether to write out page level checksums. **Default value:** `true` diff --git a/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetOutputFormat.java b/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetOutputFormat.java index 4db288f455..3ea5d92fad 100644 --- a/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetOutputFormat.java +++ b/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetOutputFormat.java @@ -34,6 +34,7 @@ import org.apache.hadoop.mapreduce.RecordWriter; import org.apache.hadoop.mapreduce.TaskAttemptContext; import org.apache.hadoop.mapreduce.lib.output.FileOutputFormat; +import org.apache.parquet.column.CdcOptions; import org.apache.parquet.column.ParquetProperties; import org.apache.parquet.column.ParquetProperties.WriterVersion; import org.apache.parquet.crypto.FileEncryptionProperties; @@ -164,6 +165,11 @@ public static enum JobSummaryLevel { public static final String STATISTICS_ENABLED = "parquet.column.statistics.enabled"; public static final String SIZE_STATISTICS_ENABLED = "parquet.size.statistics.enabled"; public static final String COLUMN_COMPRESSION_LEVEL_PREFIX = "parquet.compression.level"; + // EXPERIMENTAL: see ParquetProperties.Builder#withContentDefinedChunking + public static final String CONTENT_DEFINED_CHUNKING_ENABLED = "parquet.page.content-defined-chunking.enabled"; + public static final String CONTENT_DEFINED_CHUNKING_MIN_SIZE = "parquet.page.content-defined-chunking.min.size"; + public static final String CONTENT_DEFINED_CHUNKING_MAX_SIZE = "parquet.page.content-defined-chunking.max.size"; + public static final String CONTENT_DEFINED_CHUNKING_NORM_LEVEL = "parquet.page.content-defined-chunking.norm.level"; public static JobSummaryLevel getJobSummaryLevel(Configuration conf) { String level = conf.get(JOB_SUMMARY_LEVEL); @@ -409,6 +415,19 @@ private static int getPageRowCountLimit(Configuration conf) { return conf.getInt(PAGE_ROW_COUNT_LIMIT, ParquetProperties.DEFAULT_PAGE_ROW_COUNT_LIMIT); } + private static boolean getContentDefinedChunkingEnabled(Configuration conf) { + return conf.getBoolean( + CONTENT_DEFINED_CHUNKING_ENABLED, ParquetProperties.DEFAULT_CONTENT_DEFINED_CHUNKING_ENABLED); + } + + private static CdcOptions getCdcOptions(Configuration conf) { + return CdcOptions.builder() + .withMinChunkSize(conf.getLong(CONTENT_DEFINED_CHUNKING_MIN_SIZE, CdcOptions.DEFAULT.getMinChunkSize())) + .withMaxChunkSize(conf.getLong(CONTENT_DEFINED_CHUNKING_MAX_SIZE, CdcOptions.DEFAULT.getMaxChunkSize())) + .withNormLevel(conf.getInt(CONTENT_DEFINED_CHUNKING_NORM_LEVEL, CdcOptions.DEFAULT.getNormLevel())) + .build(); + } + public static void setPageWriteChecksumEnabled(JobContext jobContext, boolean val) { setPageWriteChecksumEnabled(getConfiguration(jobContext), val); } @@ -536,6 +555,10 @@ public RecordWriter getRecordWriter(Configuration conf, Path file, Comp .withPageRowCountLimit(getPageRowCountLimit(conf)) .withPageWriteChecksumEnabled(getPageWriteChecksumEnabled(conf)) .withStatisticsEnabled(getStatisticsEnabled(conf)); + if (getContentDefinedChunkingEnabled(conf)) { + // Only when enabled: building the options validates them, and a disabled job must not fail. + propsBuilder.withContentDefinedChunking(getCdcOptions(conf)); + } new ColumnConfigParser() .withColumnConfig( ENABLE_DICTIONARY, key -> conf.getBoolean(key, false), propsBuilder::withDictionaryEncoding) diff --git a/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetWriter.java b/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetWriter.java index 7dcbb3188c..3fe824fad2 100644 --- a/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetWriter.java +++ b/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetWriter.java @@ -26,6 +26,7 @@ import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.fs.Path; import org.apache.parquet.bytes.ByteBufferAllocator; +import org.apache.parquet.column.CdcOptions; import org.apache.parquet.column.ParquetProperties; import org.apache.parquet.column.ParquetProperties.WriterVersion; import org.apache.parquet.compression.CompressionCodecFactory; @@ -681,6 +682,30 @@ public SELF withPageRowCountLimit(int rowCount) { return self(); } + /** + * EXPERIMENTAL: Enable or disable content defined chunking of data pages; see + * {@link ParquetProperties.Builder#withContentDefinedChunkingEnabled(boolean)}. + * + * @param enabled whether to derive data page boundaries from the content + * @return this builder for method chaining + */ + public SELF withContentDefinedChunkingEnabled(boolean enabled) { + encodingPropsBuilder.withContentDefinedChunkingEnabled(enabled); + return self(); + } + + /** + * EXPERIMENTAL: Enable content defined chunking with the given options; see + * {@link ParquetProperties.Builder#withContentDefinedChunking(CdcOptions)}. + * + * @param options the chunking options + * @return this builder for method chaining + */ + public SELF withContentDefinedChunking(CdcOptions options) { + encodingPropsBuilder.withContentDefinedChunking(options); + return self(); + } + /** * Set the Parquet format dictionary page size used by the constructed * writer. diff --git a/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestCdcWriter.java b/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestCdcWriter.java new file mode 100644 index 0000000000..a88cddc459 --- /dev/null +++ b/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestCdcWriter.java @@ -0,0 +1,478 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.hadoop; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatCode; +import static org.assertj.core.api.Assertions.assertThatThrownBy; + +import java.io.IOException; +import java.util.ArrayList; +import java.util.List; +import java.util.Random; +import java.util.Set; +import java.util.TreeSet; +import java.util.function.UnaryOperator; +import java.util.stream.Collectors; +import java.util.stream.Stream; +import org.apache.hadoop.conf.Configuration; +import org.apache.hadoop.fs.Path; +import org.apache.hadoop.mapreduce.RecordWriter; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.column.Encoding; +import org.apache.parquet.column.ParquetProperties.WriterVersion; +import org.apache.parquet.column.page.DataPage; +import org.apache.parquet.column.page.DataPageV1; +import org.apache.parquet.column.page.DataPageV2; +import org.apache.parquet.column.page.PageReadStore; +import org.apache.parquet.column.page.PageReader; +import org.apache.parquet.example.data.Group; +import org.apache.parquet.example.data.simple.NanoTime; +import org.apache.parquet.example.data.simple.SimpleGroupFactory; +import org.apache.parquet.hadoop.example.ExampleParquetWriter; +import org.apache.parquet.hadoop.example.GroupReadSupport; +import org.apache.parquet.hadoop.example.GroupWriteSupport; +import org.apache.parquet.hadoop.metadata.CompressionCodecName; +import org.apache.parquet.hadoop.util.HadoopInputFile; +import org.apache.parquet.schema.MessageType; +import org.apache.parquet.schema.MessageTypeParser; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.io.TempDir; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.Arguments; +import org.junit.jupiter.params.provider.CsvSource; +import org.junit.jupiter.params.provider.EnumSource; +import org.junit.jupiter.params.provider.MethodSource; + +/** + * Content defined chunking through {@link ParquetWriter.Builder} and {@link ParquetOutputFormat}, into + * a real file that is read back. + */ +public class TestCdcWriter { + + private static final MessageType SCHEMA = + MessageTypeParser.parseMessageType("message t { required int64 id; required binary name (STRING); }"); + + private static final CdcOptions OPTIONS = CdcOptions.builder() + .withMinChunkSize(16 * 1024) + .withMaxChunkSize(64 * 1024) + .build(); + + private static final CdcOptions SMALL_OPTIONS = CdcOptions.builder() + .withMinChunkSize(4 * 1024) + .withMaxChunkSize(16 * 1024) + .build(); + + private static final int ROWS = 40_000; + + @TempDir + private java.nio.file.Path tempDir; + + @ParameterizedTest + @EnumSource(WriterVersion.class) + public void roundTripsEveryValue(WriterVersion version) throws IOException { + Path file = write("roundtrip-" + version, b -> b.withWriterVersion(version)); + assertThat(pageCounts(file)).hasSizeGreaterThan(1); + try (ParquetFileReader reader = ParquetFileReader.open(HadoopInputFile.fromPath(file, new Configuration()))) { + DataPage page = reader.readNextRowGroup() + .getPageReader( + reader.getFileMetaData().getSchema().getColumns().get(1)) + .readPage(); + assertThat(page) + .as("the writer version reaches the pages") + .isInstanceOf(version == WriterVersion.PARQUET_2_0 ? DataPageV2.class : DataPageV1.class); + } + + List read = new ArrayList<>(); + try (ParquetReader reader = ParquetReader.builder(new GroupReadSupport(), file) + .withConf(new Configuration()) + .build()) { + Group g; + while ((g = reader.read()) != null) { + read.add(g.getLong("id", 0) + "|" + g.getString("name", 0)); + } + } + + List expected = new ArrayList<>(ROWS); + for (int i = 0; i < ROWS; i++) { + expected.add(i + "|" + name(i)); + } + assertThat(read).containsExactlyElementsOf(expected); + } + + /** + * Chunking only moves page boundaries: every physical type, null, list and map reads back as it + * does without, across row groups, page limits, dictionary fallback and column chunks of nulls. + */ + @ParameterizedTest + @CsvSource({"PARQUET_1_0, true", "PARQUET_1_0, false", "PARQUET_2_0, true", "PARQUET_2_0, false"}) + public void chunkingLosesNoData(WriterVersion version, boolean dictionary) throws IOException { + MessageType schema = MessageTypeParser.parseMessageType("message t {" + + " required int32 i32; optional int64 i64; optional float f; required double d; optional boolean b;" + + " optional binary s (STRING); optional fixed_len_byte_array(8) x; optional int96 t;" + + " optional group l (LIST) { repeated group list { optional int32 element; } }" + + " optional group m (MAP) { repeated group key_value { required binary key (STRING); optional int64 value; } }" + + " optional int64 sparse;" + + " }"); + SimpleGroupFactory f = new SimpleGroupFactory(schema); + Random random = new Random(31); + List rows = new ArrayList<>(); + for (int i = 0; i < 30_000; i++) { + Group row = f.newGroup().append("i32", random.nextInt(1000)).append("d", random.nextDouble()); + if (random.nextInt(4) > 0) { + row.append("i64", random.nextLong()); + } + if (random.nextInt(8) > 0) { + row.append("f", random.nextFloat()); + } + if (random.nextInt(3) > 0) { + row.append("b", random.nextBoolean()); + } + if (random.nextInt(5) > 0) { + row.append("s", "v" + random.nextInt(random.nextBoolean() ? 300 : Integer.MAX_VALUE)); + } + if (random.nextInt(5) > 0) { + row.append("x", String.format("%08x", random.nextInt())); + } + if (random.nextInt(5) > 0) { + row.append("t", new NanoTime(random.nextInt(3_000_000), random.nextLong())); + } + if (random.nextInt(6) > 0) { + Group list = row.addGroup("l"); + for (int n = random.nextInt(4); n > 0; n--) { + Group element = list.addGroup("list"); + if (random.nextInt(5) > 0) { + element.append("element", random.nextInt()); + } + } + } + if (random.nextInt(6) > 0) { + Group map = row.addGroup("m"); + for (int n = random.nextInt(3); n > 0; n--) { + Group entry = map.addGroup("key_value").append("key", "k" + n); + if (random.nextInt(4) > 0) { + entry.append("value", random.nextLong()); + } + } + } + // Null in the first two row groups, and in the first pages of every later one. + if (i >= 14_000 && i % 7_000 >= 5_000) { + row.append("sparse", random.nextLong()); + } + rows.add(row); + } + List expected = rows.stream().map(Group::toString).collect(Collectors.toList()); + + List files = new ArrayList<>(); + for (boolean chunking : new boolean[] {false, true}) { + Path file = new Path(tempDir.resolve("data-" + version + "-" + dictionary + "-" + chunking + ".parquet") + .toUri()); + try (ParquetWriter writer = ExampleParquetWriter.builder(file) + .withType(schema) + .withWriterVersion(version) + .withDictionaryEncoding(dictionary) + .withRowGroupRowCountLimit(7_000) + .withPageSize(2 * 1024) + .withPageRowCountLimit(2_000) + .withDictionaryPageSize(16 * 1024) + .withContentDefinedChunking(SMALL_OPTIONS) + .withContentDefinedChunkingEnabled(chunking) + .build()) { + for (Group row : rows) { + writer.write(row); + } + } + List read = new ArrayList<>(); + try (ParquetReader reader = + ParquetReader.builder(new GroupReadSupport(), file).build()) { + for (Group row = reader.read(); row != null; row = reader.read()) { + read.add(row.toString()); + } + } + assertThat(read).as(chunking ? "chunked" : "unchunked").containsExactlyElementsOf(expected); + files.add(file); + } + assertThat(pageCounts(files.get(1))) + .as("chunking moves the page boundaries") + .isNotEqualTo(pageCounts(files.get(0))); + } + + /** + * The chunker follows each column across row groups, as arrow-rs's does, so a row group boundary + * adds a page break but moves no chunk boundary. After an edit the chunks realign once, rather than + * again at the start of every later row group, whose position the edit has shifted. The first row + * group ends a row before a chunk boundary, which a chunker restarting its hash, match run or size + * there misses; a match pending across records needs a nested column, as TestCdcWrite has. + */ + @Test + public void chunkingContinuesAcrossRowGroups() throws IOException { + List> whole = pageCountsByRowGroup(write("one-group", UnaryOperator.identity())); + assertThat(whole).hasSize(1); + assertThat(whole.get(0)).hasSizeGreaterThan(4); + int perGroup = whole.get(0).get(0) - 1; + List> groups = pageCountsByRowGroup(write("groups", b -> b.withRowGroupRowCountLimit(perGroup))); + assertThat(groups).hasSizeGreaterThan(4); + + Set expected = new TreeSet<>(pageEnds(whole)); + for (long end = perGroup; end < ROWS; end += perGroup) { + expected.add(end); + } + assertThat(pageEnds(groups)).containsExactlyElementsOf(expected); + } + + /** Where each page ends, counted in values from the start of the file. */ + private static List pageEnds(List> groups) { + List ends = new ArrayList<>(); + long end = 0; + for (List group : groups) { + for (int count : group) { + end += count; + ends.add(end); + } + } + return ends; + } + + /** Through {@link ParquetOutputFormat}, the only reader of the configuration keys. */ + @Test + public void anInvalidEnvelopeIsIgnoredWhileChunkingIsOff() throws Exception { + Configuration conf = chunkingConf(); + conf.setLong(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_MIN_SIZE, 8 * 1024 * 1024); // > the default max + conf.setBoolean(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_ENABLED, false); + + Path off = new Path(tempDir.resolve("stale-config.parquet").toUri()); + assertThatCode(() -> new ParquetOutputFormat() + .getRecordWriter(conf, off, CompressionCodecName.UNCOMPRESSED) + .close(null)) + .as("a stale key must not fail a job that has the feature switched off") + .doesNotThrowAnyException(); + + conf.setBoolean(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_ENABLED, true); + Path on = new Path(tempDir.resolve("stale-config-on.parquet").toUri()); + assertThatThrownBy(() -> + new ParquetOutputFormat().getRecordWriter(conf, on, CompressionCodecName.UNCOMPRESSED)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessage("Invalid content defined chunking size range: maximum chunk size (1048576) must be greater " + + "than minimum chunk size (8388608)"); + } + + /** + * Column chunks of nulls alone, of every dictionary encoded type, keep the dictionary with an empty + * dictionary page, as in Arrow C++, and so does one that starts with pages of them. The file reads + * back whatever the writer version and codec. + */ + @ParameterizedTest + @MethodSource("versionsAndCodecs") + public void nullsReadBackWithTheDictionaryOn(WriterVersion version, CompressionCodecName codec) throws IOException { + MessageType schema = MessageTypeParser.parseMessageType( + "message t { required int64 id; optional int32 late; optional binary b; optional int32 i;" + + " optional int64 l; optional float f; optional double d; optional fixed_len_byte_array(4) x; }"); + Path file = new Path( + tempDir.resolve("nulls-" + version + "-" + codec + ".parquet").toUri()); + try (ParquetWriter writer = ExampleParquetWriter.builder(file) + .withType(schema) + .withWriterVersion(version) + .withCompressionCodec(codec) + .withRowGroupRowCountLimit(ROWS / 4) + .withContentDefinedChunking(SMALL_OPTIONS) + .build()) { + SimpleGroupFactory f = new SimpleGroupFactory(schema); + for (int i = 0; i < ROWS; i++) { + Group row = f.newGroup().append("id", (long) i); + if (i >= ROWS * 5 / 8) { + row.append("late", i % 100); + } + writer.write(row); + } + } + + try (ParquetReader reader = + ParquetReader.builder(new GroupReadSupport(), file).build()) { + for (int i = 0; i < ROWS; i++) { + Group row = reader.read(); + assertThat(row.getLong("id", 0)).isEqualTo(i); + for (String empty : new String[] {"b", "i", "l", "f", "d", "x"}) { + assertThat(row.getFieldRepetitionCount(empty)).as(empty).isZero(); + } + if (i >= ROWS * 5 / 8) { + assertThat(row.getInteger("late", 0)).isEqualTo(i % 100); + } else { + assertThat(row.getFieldRepetitionCount("late")).isZero(); + } + } + assertThat(reader.read()).isNull(); + } + } + + static Stream versionsAndCodecs() { + return Stream.of(WriterVersion.values()).flatMap(version -> Stream.of( + CompressionCodecName.UNCOMPRESSED, + CompressionCodecName.SNAPPY, + CompressionCodecName.GZIP, + CompressionCodecName.ZSTD, + CompressionCodecName.LZ4_RAW) + .map(codec -> Arguments.of(version, codec))); + } + + @Test + public void chunkingCanBeSwitchedOffAfterItsOptionsAreSet() throws IOException { + // With the position-based limits lifted, an unchunked column chunk is one page. + assertThat(pageCounts(write("switched-off", b -> b.withContentDefinedChunkingEnabled(false)))) + .hasSize(1); + } + + @Test + public void theConfigurationKeysActuallyChunk() throws Exception { + Configuration conf = chunkingConf(); + conf.setBoolean(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_ENABLED, true); + conf.setLong(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_MIN_SIZE, 16 * 1024); + conf.setLong(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_MAX_SIZE, 64 * 1024); + conf.setInt(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_NORM_LEVEL, 1); + CdcOptions configured = CdcOptions.builder() + .withMinChunkSize(16 * 1024) + .withMaxChunkSize(64 * 1024) + .withNormLevel(1) + .build(); + List configuredPages = pageCounts(writeThroughOutputFormat(conf, "configured")); + assertThat(configuredPages).hasSizeGreaterThan(1); + assertThat(configuredPages) + .as("every key reaches the writer") + .containsExactlyElementsOf(pageCounts(write("direct", b -> b.withContentDefinedChunking(configured)))); + + conf.unset(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_ENABLED); + assertThat(pageCounts(writeThroughOutputFormat(conf, "unconfigured"))) + .as("and do nothing while the enabled key is unset") + .hasSize(1); + } + + private Configuration chunkingConf() { + Configuration conf = new Configuration(); + GroupWriteSupport.setSchema(SCHEMA, conf); + conf.set(ParquetOutputFormat.WRITE_SUPPORT_CLASS, GroupWriteSupport.class.getName()); + conf.setInt(ParquetOutputFormat.PAGE_ROW_COUNT_LIMIT, Integer.MAX_VALUE); + conf.setInt(ParquetOutputFormat.PAGE_SIZE, 1 << 30); + conf.setBoolean(ParquetOutputFormat.ENABLE_DICTIONARY, false); + return conf; + } + + private Path writeThroughOutputFormat(Configuration conf, String label) throws Exception { + Path file = new Path(tempDir.resolve(label + ".parquet").toUri()); + RecordWriter writer = + new ParquetOutputFormat().getRecordWriter(conf, file, CompressionCodecName.UNCOMPRESSED); + SimpleGroupFactory f = new SimpleGroupFactory(SCHEMA); + for (int i = 0; i < ROWS; i++) { + writer.write(null, f.newGroup().append("id", (long) i).append("name", name(i))); + } + writer.close(null); + return file; + } + + /** + * A chunked first page is cut by content, not size, so it is no sample to judge a dictionary on: + * as in Arrow C++, the dictionary falls back only when it outgrows its size limit. + */ + @Test + public void theDictionaryFallsBackOnItsSizeLimitAlone() throws IOException { + assertThat(encodingsOf(writeHighCardinality("repeating", 12_000))) + .as("a dictionary within its limit is kept") + .contains(Encoding.PLAIN_DICTIONARY) + .doesNotContain(Encoding.PLAIN); + assertThat(encodingsOf(writeHighCardinality("unique", Integer.MAX_VALUE))) + .as("one that outgrows it is dropped") + .contains(Encoding.PLAIN); + } + + private static Set encodingsOf(Path file) throws IOException { + try (ParquetFileReader reader = ParquetFileReader.open(HadoopInputFile.fromPath(file, new Configuration()))) { + return reader.getFooter().getBlocks().get(0).getColumns().get(1).getEncodings(); + } + } + + /** 60 000 chunked rows of ~45-byte values drawn from {@code distinct} different ones. */ + private Path writeHighCardinality(String label, int distinct) throws IOException { + Path file = new Path(tempDir.resolve(label + ".parquet").toUri()); + try (ParquetWriter writer = ExampleParquetWriter.builder(file) + .withType(SCHEMA) + .withCompressionCodec(CompressionCodecName.UNCOMPRESSED) + .withRowGroupSize(1L << 30) + .withContentDefinedChunking(SMALL_OPTIONS) + .build()) { + SimpleGroupFactory f = new SimpleGroupFactory(SCHEMA); + Random random = new Random(7); + String pad = "xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx"; + for (int i = 0; i < 60_000; i++) { + writer.write(f.newGroup().append("id", (long) i).append("name", "v" + random.nextInt(distinct) + pad)); + } + } + return file; + } + + private static String name(int i) { + return "row-" + (i * 2654435761L % 1_000_000L); + } + + /** + * Writes {@code ROWS} rows with chunking on and the position-based page limits lifted, so every + * page boundary is the chunker's. + */ + private Path write(String label, UnaryOperator configure) throws IOException { + Path file = new Path(tempDir.resolve(label + ".parquet").toUri()); + ExampleParquetWriter.Builder builder = ExampleParquetWriter.builder(file) + .withType(SCHEMA) + .withCompressionCodec(CompressionCodecName.UNCOMPRESSED) + .withDictionaryEncoding(false) + .withRowGroupSize(1L << 30) + .withPageSize(1 << 30) + .withPageRowCountLimit(Integer.MAX_VALUE) + .withContentDefinedChunking(OPTIONS); + try (ParquetWriter writer = configure.apply(builder).build()) { + SimpleGroupFactory f = new SimpleGroupFactory(SCHEMA); + for (int i = 0; i < ROWS; i++) { + writer.write(f.newGroup().append("id", (long) i).append("name", name(i))); + } + } + return file; + } + + /** The page value counts of each row group, kept apart rather than concatenated. */ + private static List> pageCountsByRowGroup(Path file) throws IOException { + List> groups = new ArrayList<>(); + try (ParquetFileReader reader = ParquetFileReader.open(HadoopInputFile.fromPath(file, new Configuration()))) { + ColumnDescriptor column = + reader.getFileMetaData().getSchema().getColumns().get(1); + PageReadStore rowGroup; + while ((rowGroup = reader.readNextRowGroup()) != null) { + List counts = new ArrayList<>(); + PageReader pages = rowGroup.getPageReader(column); + for (long read = 0; read < pages.getTotalValueCount(); ) { + DataPage page = pages.readPage(); + counts.add(page.getValueCount()); + read += page.getValueCount(); + } + groups.add(counts); + } + } + return groups; + } + + private static List pageCounts(Path file) throws IOException { + return pageCountsByRowGroup(file).stream().flatMap(List::stream).collect(Collectors.toList()); + } +}