diff --git a/parquet-benchmarks/src/main/java/org/apache/parquet/benchmarks/CdcWriteBenchmarks.java b/parquet-benchmarks/src/main/java/org/apache/parquet/benchmarks/CdcWriteBenchmarks.java new file mode 100644 index 0000000000..51e839ec48 --- /dev/null +++ b/parquet-benchmarks/src/main/java/org/apache/parquet/benchmarks/CdcWriteBenchmarks.java @@ -0,0 +1,159 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.benchmarks; + +import java.io.IOException; +import java.util.ArrayList; +import java.util.List; +import java.util.Random; +import java.util.concurrent.TimeUnit; +import org.apache.parquet.example.data.Group; +import org.apache.parquet.example.data.simple.SimpleGroupFactory; +import org.apache.parquet.hadoop.ParquetFileWriter; +import org.apache.parquet.hadoop.ParquetWriter; +import org.apache.parquet.hadoop.example.ExampleParquetWriter; +import org.apache.parquet.hadoop.metadata.CompressionCodecName; +import org.apache.parquet.io.api.Binary; +import org.apache.parquet.schema.MessageType; +import org.apache.parquet.schema.MessageTypeParser; +import org.openjdk.jmh.annotations.Benchmark; +import org.openjdk.jmh.annotations.BenchmarkMode; +import org.openjdk.jmh.annotations.Fork; +import org.openjdk.jmh.annotations.Level; +import org.openjdk.jmh.annotations.Measurement; +import org.openjdk.jmh.annotations.Mode; +import org.openjdk.jmh.annotations.OutputTimeUnit; +import org.openjdk.jmh.annotations.Param; +import org.openjdk.jmh.annotations.Scope; +import org.openjdk.jmh.annotations.Setup; +import org.openjdk.jmh.annotations.State; +import org.openjdk.jmh.annotations.Warmup; + +/** + * The write cost of content defined chunking: the same rows written with and without it, for rows of + * several shapes, since the cost is the hashing of every value's and level's bytes. + * + *

Both variants lift the page row count limit and turn dictionary encoding off: chunking changes + * where pages end and when a dictionary falls back, either of which would dwarf the hashing. Rows are + * built once and written to {@link BlackHoleOutputFile}, so neither data generation nor I/O is + * measured. The cost while disabled is {@link WriteBenchmarks} compared across revisions. + */ +@BenchmarkMode(Mode.AverageTime) +@Fork(1) +@Warmup(iterations = 3) +@Measurement(iterations = 5) +@OutputTimeUnit(TimeUnit.MILLISECONDS) +@State(Scope.Thread) +public class CdcWriteBenchmarks { + + private static final int ROW_COUNT = 100_000; + + @Param({"false", "true"}) + public boolean chunking; + + /** + * mixed: a long, a 128-byte binary and a list of eight ints; numbers: an int, a long and a nullable + * double; strings: short strings, one of them nullable; lists: nullable lists of zero to eight + * nullable longs. + */ + @Param({"mixed", "numbers", "strings", "lists"}) + public String data; + + private MessageType schema; + private List rows; + + @Setup(Level.Trial) + public void setup() { + Random random = new Random(TestDataFactory.DEFAULT_SEED); + switch (data) { + case "mixed": + schema = MessageTypeParser.parseMessageType( + "message m { required int64 l; required binary b; required group g { repeated int32 i; } }"); + break; + case "numbers": + schema = MessageTypeParser.parseMessageType( + "message m { required int32 i; required int64 l; optional double d; }"); + break; + case "strings": + schema = MessageTypeParser.parseMessageType( + "message m { required binary s (STRING); optional binary t (STRING); }"); + break; + case "lists": + schema = MessageTypeParser.parseMessageType( + "message m { optional group l (LIST) { repeated group list { optional int64 element; } } }"); + break; + default: + throw new IllegalArgumentException("unknown data " + data); + } + Binary[] binaries = TestDataFactory.generateBinaryData(ROW_COUNT, 128, 0, TestDataFactory.DEFAULT_SEED); + SimpleGroupFactory factory = new SimpleGroupFactory(schema); + rows = new ArrayList<>(ROW_COUNT); + for (int i = 0; i < ROW_COUNT; i++) { + Group row = factory.newGroup(); + switch (data) { + case "mixed": + row.append("l", (long) i).append("b", binaries[i]); + Group g = row.addGroup("g"); + for (int j = 0; j < 8; j++) { + g.append("i", random.nextInt()); + } + break; + case "numbers": + row.append("i", random.nextInt()).append("l", random.nextLong()); + if (random.nextInt(10) > 0) { + row.append("d", random.nextDouble()); + } + break; + case "strings": + row.append("s", "s" + random.nextInt(1_000_000)); + if (random.nextInt(10) > 0) { + row.append("t", Long.toString(random.nextLong(), 36)); + } + break; + default: + if (random.nextInt(10) > 0) { + Group list = row.addGroup("l"); + for (int n = random.nextInt(9); n > 0; n--) { + Group element = list.addGroup("list"); + if (random.nextInt(10) > 0) { + element.append("element", random.nextLong()); + } + } + } + } + rows.add(row); + } + } + + @Benchmark + public void write() throws IOException { + try (ParquetWriter writer = ExampleParquetWriter.builder(BlackHoleOutputFile.INSTANCE) + .withWriteMode(ParquetFileWriter.Mode.OVERWRITE) + .withType(schema) + .withCompressionCodec(CompressionCodecName.UNCOMPRESSED) + .withPageRowCountLimit(Integer.MAX_VALUE) + .withDictionaryEncoding(false) + .withContentDefinedChunkingEnabled(chunking) + .build()) { + for (Group row : rows) { + writer.write(row); + } + } + } +} diff --git a/parquet-column/src/main/java/org/apache/parquet/column/CdcOptions.java b/parquet-column/src/main/java/org/apache/parquet/column/CdcOptions.java new file mode 100644 index 0000000000..b074880bae --- /dev/null +++ b/parquet-column/src/main/java/org/apache/parquet/column/CdcOptions.java @@ -0,0 +1,157 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column; + +import org.apache.parquet.Preconditions; +import org.apache.parquet.internal.column.chunking.RollingHashMask; + +/** + * EXPERIMENTAL: The size envelope and normalization level of content defined chunking (CDC), which + * ends data pages at boundaries derived from the column's values, so that files sharing a run of + * values share byte-identical pages. Sizes are measured on values and levels before encoding. + * Dictionary ids follow the order values first appear in, so an edit that adds new values renumbers + * the later ones within its row group: columns of many distinct values deduplicate best without + * dictionary encoding. + * + *

The settings and defaults match {@code CdcOptions} in Arrow C++ and arrow-rs, and so do the + * chunk boundaries for the same physical values. Arrow types converted before writing, such as + * narrow integers, coerced timestamps and decimals, are hashed differently there. Chunk boundaries + * continue across row groups, as in arrow-rs; Arrow C++ restarts them in every row group, so its + * chunks match only in the first. + * + * @see ParquetProperties.Builder#withContentDefinedChunking(CdcOptions) + */ +public final class CdcOptions { + + /** 256 KiB minimum, 1 MiB maximum and normalization level 0. */ + public static final CdcOptions DEFAULT = builder().build(); + + private final long minChunkSize; + private final long maxChunkSize; + private final int normLevel; + + private CdcOptions(Builder builder) { + this.minChunkSize = builder.minChunkSize; + this.maxChunkSize = builder.maxChunkSize; + this.normLevel = builder.normLevel; + // Validate now rather than when the first column writer is built. + RollingHashMask.calculate(minChunkSize, maxChunkSize, normLevel); + } + + /** + * @return the minimum chunk size in bytes + */ + public long getMinChunkSize() { + return minChunkSize; + } + + /** + * @return the maximum chunk size in bytes + */ + public long getMaxChunkSize() { + return maxChunkSize; + } + + /** + * @return the normalization level of the rolling hash mask + */ + public int getNormLevel() { + return normLevel; + } + + @Override + public String toString() { + return "CdcOptions{minChunkSize=" + minChunkSize + ", maxChunkSize=" + maxChunkSize + ", normLevel=" + normLevel + + '}'; + } + + /** + * @return a builder holding the default options + */ + public static Builder builder() { + return new Builder(); + } + + /** EXPERIMENTAL: Builds {@link CdcOptions}. */ + public static class Builder { + private long minChunkSize = 256 * 1024L; + private long maxChunkSize = 1024 * 1024L; + private int normLevel = 0; + + private Builder() {} + + /** + * Set the minimum chunk size in bytes, 256 KiB by default. The rolling hash is not updated + * until a chunk reaches this size, so no chunk is shorter but a file's last; pages can be, where + * a row group or a page limit ends one. + * + * @param minChunkSize the minimum chunk size in bytes + * @return this builder for method chaining + */ + public Builder withMinChunkSize(long minChunkSize) { + Preconditions.checkArgument( + minChunkSize >= 0, + "Invalid content defined chunking minimum chunk size (negative): %s", + minChunkSize); + this.minChunkSize = minChunkSize; + return this; + } + + /** + * Set the maximum chunk size in bytes, 1 MiB by default. A chunk ends when it reaches this size, + * whatever the rolling hash says. {@link ParquetProperties.Builder#withPageSize(int)} separately + * limits the page size; below this it splits chunks into more pages, in the same places after + * an edit, so it does not cost deduplication. + * + * @param maxChunkSize the maximum chunk size in bytes + * @return this builder for method chaining + */ + public Builder withMaxChunkSize(long maxChunkSize) { + Preconditions.checkArgument( + maxChunkSize > 0, + "Invalid content defined chunking maximum chunk size (not positive): %s", + maxChunkSize); + this.maxChunkSize = maxChunkSize; + return this; + } + + /** + * Set the normalization level of the rolling hash mask, 0 by default. Raising it makes a + * boundary more likely, which tightens the chunk size distribution and improves deduplication + * at the cost of more small pages; lowering it does the reverse. Values outside + * {@code [-3, 3]} are not useful. + * + * @param normLevel the normalization level + * @return this builder for method chaining + */ + public Builder withNormLevel(int normLevel) { + this.normLevel = normLevel; + return this; + } + + /** + * @return the options + * @throws IllegalArgumentException if the maximum chunk size is not greater than the minimum, + * or the envelope is too narrow for the normalization level + */ + public CdcOptions build() { + return new CdcOptions(this); + } + } +} diff --git a/parquet-column/src/main/java/org/apache/parquet/column/ParquetProperties.java b/parquet-column/src/main/java/org/apache/parquet/column/ParquetProperties.java index 8fe45e01ef..5556744843 100644 --- a/parquet-column/src/main/java/org/apache/parquet/column/ParquetProperties.java +++ b/parquet-column/src/main/java/org/apache/parquet/column/ParquetProperties.java @@ -29,6 +29,7 @@ import org.apache.parquet.bytes.ByteBufferAllocator; import org.apache.parquet.bytes.CapacityByteArrayOutputStream; import org.apache.parquet.bytes.HeapByteBufferAllocator; +import org.apache.parquet.column.impl.CdcChunkers; import org.apache.parquet.column.impl.ColumnWriteStoreV1; import org.apache.parquet.column.impl.ColumnWriteStoreV2; import org.apache.parquet.column.page.PageWriteStore; @@ -70,6 +71,8 @@ public class ParquetProperties { public static final boolean DEFAULT_PAGE_WRITE_CHECKSUM_ENABLED = true; + public static final boolean DEFAULT_CONTENT_DEFINED_CHUNKING_ENABLED = false; + /** * @deprecated This shared instance can cause thread safety issues when used by multiple builders concurrently. * Use {@code new DefaultValuesWriterFactory()} instead to create individual instances. @@ -138,6 +141,9 @@ public static WriterVersion fromString(String name) { private final ColumnProperty sizeStatistics; private final ColumnProperty columnCodecs; private final ColumnProperty columnCompressionLevels; + private final boolean cdcEnabled; + private final CdcOptions cdcOptions; + private final CdcChunkers cdcChunkers = new CdcChunkers(); private ParquetProperties(Builder builder) { this.pageSizeThreshold = builder.pageSize; @@ -172,6 +178,8 @@ private ParquetProperties(Builder builder) { this.sizeStatistics = builder.sizeStatistics.build(); this.columnCodecs = builder.columnCodecs.build(); this.columnCompressionLevels = builder.columnCompressionLevels.build(); + this.cdcEnabled = builder.cdcEnabled; + this.cdcOptions = builder.cdcOptions; } public static Builder builder() { @@ -319,6 +327,35 @@ public int getRowGroupRowCountLimit() { return rowGroupRowCountLimit; } + /** + * EXPERIMENTAL: Whether data page boundaries are derived from the content of the data. + * + * @return {@code true} if content defined chunking is enabled + */ + public boolean isContentDefinedChunkingEnabled() { + return cdcEnabled; + } + + /** + * EXPERIMENTAL: The content defined chunking options, which only apply while + * {@link #isContentDefinedChunkingEnabled()} is {@code true}. + * + * @return the chunking options, never {@code null} + */ + public CdcOptions getCdcOptions() { + return cdcOptions; + } + + /** + * Internal: the content defined chunking state, which every column write store made with these + * properties continues. + * + * @return the chunkers of the file written with these properties + */ + public CdcChunkers getCdcChunkers() { + return cdcChunkers; + } + public int getPageRowCountLimit() { return pageRowCountLimit; } @@ -413,6 +450,9 @@ public String toString() { + "Bloom filter expected number of distinct values are: " + bloomFilterNDVs + '\n' + "Bloom filter false positive probabilities are: " + bloomFilterFPPs + '\n' + "Page row count limit to " + getPageRowCountLimit() + '\n' + + "Content defined chunking is: " + + (cdcEnabled ? cdcOptions.toString() : "off") + + '\n' + "Writing page checksums is: " + (getPageWriteChecksumEnabled() ? "on" : "off") + '\n' + "Statistics enabled: " + statisticsEnabled + '\n' + "Size statistics enabled: " + sizeStatisticsEnabled; @@ -460,6 +500,8 @@ public static class Builder { private final ColumnProperty.Builder sizeStatistics; private final ColumnProperty.Builder columnCodecs; private final ColumnProperty.Builder columnCompressionLevels; + private boolean cdcEnabled = DEFAULT_CONTENT_DEFINED_CHUNKING_ENABLED; + private CdcOptions cdcOptions = CdcOptions.DEFAULT; private Builder() { enableDict = ColumnProperty.builder().withDefaultValue(DEFAULT_IS_DICTIONARY_ENABLED); @@ -511,6 +553,8 @@ private Builder(ParquetProperties toCopy) { this.sizeStatisticsEnabled = toCopy.sizeStatisticsEnabled; this.columnCodecs = ColumnProperty.builder(toCopy.columnCodecs); this.columnCompressionLevels = ColumnProperty.builder(toCopy.columnCompressionLevels); + this.cdcEnabled = toCopy.cdcEnabled; + this.cdcOptions = toCopy.cdcOptions; } /** @@ -755,6 +799,41 @@ public Builder withPageRowCountLimit(int rowCount) { return this; } + /** + * EXPERIMENTAL: Enable or disable content defined chunking of data pages, using + * {@link CdcOptions#DEFAULT} unless {@link #withContentDefinedChunking(CdcOptions)} sets others. + * Disabled by default. + * + *

As in Arrow C++, {@link #withPageSize(int)} and {@link #withPageRowCountLimit(int)} still + * cut pages inside a chunk, counted from the page start, so they add pages without moving any + * after an edit; and a column falls back from dictionary encoding only when its dictionary + * outgrows {@link #withDictionaryPageSize(int)}, not on its first page. + * + *

The chunking state lives in the built properties and continues across every column write + * store made with them, so chunk boundaries carry over row groups: build them for each file, and + * use them for one file at a time. + * + * @param enabled whether to derive data page boundaries from the content + * @return this builder for method chaining. + */ + public Builder withContentDefinedChunkingEnabled(boolean enabled) { + this.cdcEnabled = enabled; + return this; + } + + /** + * EXPERIMENTAL: Enable content defined chunking with the given options. A later + * {@code withContentDefinedChunkingEnabled(false)} disables it again. + * + * @param options the chunking options + * @return this builder for method chaining. + */ + public Builder withContentDefinedChunking(CdcOptions options) { + this.cdcOptions = Objects.requireNonNull(options, "CdcOptions cannot be null"); + this.cdcEnabled = true; + return this; + } + public Builder withPageWriteChecksumEnabled(boolean val) { this.pageWriteChecksumEnabled = val; return this; diff --git a/parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunker.java b/parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunker.java new file mode 100644 index 0000000000..45cbb28e3b --- /dev/null +++ b/parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunker.java @@ -0,0 +1,183 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import java.nio.ByteBuffer; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.internal.column.chunking.RollingHashMask; +import org.apache.parquet.io.api.Binary; + +/** + * Decides content defined data page boundaries for one column. + * + *

The column writer offers every {@code (definitionLevel, repetitionLevel, value)} triplet, and + * each {@code offer} says whether a page should end before it. + * + *

Ported from Arrow C++ {@code parquet/chunker_internal} and arrow-rs + * {@code column/chunker/cdc.rs}. Pages only deduplicate across implementations while all of them + * place the same boundaries, which {@code TestCdcChunkerGoldenBoundaries} pins against Arrow C++. + * + *

Not thread safe. One instance serves one leaf column for a whole file: it is never reset, at a + * page or a row group boundary, but carried to the next row group's store. + */ +final class CdcChunker { + + private final long minChunkSize; + private final long maxChunkSize; + private final long mask; + private final int maxDef; + private final int maxRep; + + private long rollingHash; + private boolean hasMatched; + private int nthRun; + private long chunkSize; + + CdcChunker(CdcOptions options, ColumnDescriptor path) { + this.minChunkSize = options.getMinChunkSize(); + this.maxChunkSize = options.getMaxChunkSize(); + this.mask = RollingHashMask.calculate(minChunkSize, maxChunkSize, options.getNormLevel()); + this.maxDef = path.getMaxDefinitionLevel(); + this.maxRep = path.getMaxRepetitionLevel(); + } + + boolean offer(int value, int repetitionLevel, int definitionLevel) { + return offerFixedWidth(value, 4, repetitionLevel, definitionLevel); + } + + boolean offer(long value, int repetitionLevel, int definitionLevel) { + return offerFixedWidth(value, 8, repetitionLevel, definitionLevel); + } + + boolean offer(float value, int repetitionLevel, int definitionLevel) { + // Raw bits: floatToIntBits canonicalizes NaN payloads and would diverge from the references. + return offerFixedWidth(Float.floatToRawIntBits(value), 4, repetitionLevel, definitionLevel); + } + + boolean offer(double value, int repetitionLevel, int definitionLevel) { + return offerFixedWidth(Double.doubleToRawLongBits(value), 8, repetitionLevel, definitionLevel); + } + + boolean offer(boolean value, int repetitionLevel, int definitionLevel) { + // One byte per boolean, not one bit, as the references hash it. + return offerFixedWidth(value ? 1 : 0, 1, repetitionLevel, definitionLevel); + } + + private boolean offerFixedWidth(long bits, int width, int repetitionLevel, int definitionLevel) { + rollLevels(repetitionLevel, definitionLevel); + rollFixedWidth(bits, width); + return endsPageHere(repetitionLevel); + } + + boolean offer(Binary value, int repetitionLevel, int definitionLevel) { + rollLevels(repetitionLevel, definitionLevel); + rollBinary(value); + return endsPageHere(repetitionLevel); + } + + /** + * Rolls the levels only, even when {@code definitionLevel == maxDef}: an omitted top-level + * required field arrives here as {@code (0, 0)} and has no value to hash. + */ + boolean offerNull(int repetitionLevel, int definitionLevel) { + rollLevels(repetitionLevel, definitionLevel); + return endsPageHere(repetitionLevel); + } + + /** + * Definition level first, and each level as two bytes: the references hash the levels as int16_t + * in this order. + */ + private void rollLevels(int repetitionLevel, int definitionLevel) { + if (maxDef > 0) { + rollFixedWidth(definitionLevel, 2); + } + if (maxRep > 0) { + rollFixedWidth(repetitionLevel, 2); + } + } + + /** + * Asks {@link #needNewChunk()} only at a record start, because asking consumes a match; a match + * inside a record carries over to the next record start. + */ + private boolean endsPageHere(int repetitionLevel) { + return repetitionLevel == 0 && needNewChunk(); + } + + /** The value's bytes without a length prefix, as the references hash them. */ + private void rollBinary(Binary value) { + chunkSize += value.length(); + if (chunkSize < minChunkSize) { + return; + } + ByteBuffer bytes = value.toByteBuffer(); + long hash = rollingHash; + boolean matched = hasMatched; + long[] table = GearHashTable.TABLE[nthRun]; + for (int i = bytes.position(); i < bytes.limit(); ++i) { + hash = (hash << 1) + table[bytes.get(i) & 0xFF]; + matched |= (hash & mask) == 0; + } + rollingHash = hash; + hasMatched = matched; + } + + /** + * The {@code width} low-order bytes of {@code bits}, little-endian. Like the references, this + * checks the skip window once per value rather than per byte. + */ + private void rollFixedWidth(long bits, int width) { + chunkSize += width; + if (chunkSize < minChunkSize) { + return; + } + long hash = rollingHash; + boolean matched = hasMatched; + long[] table = GearHashTable.TABLE[nthRun]; + for (int i = 0; i < width; ++i) { + hash = (hash << 1) + table[(int) ((bits >>> (8 * i)) & 0xFF)]; + matched |= (hash & mask) == 0; + } + rollingHash = hash; + hasMatched = matched; + } + + /** + * A chunk ends after eight matches, each against the next gear hash table, which approximates a + * normal chunk size distribution; or at {@code maxChunkSize}. As in the references, neither + * resets the rolling hash, and the maximum size cut leaves the run counter alone. + */ + private boolean needNewChunk() { + if (hasMatched) { + hasMatched = false; + if (++nthRun >= GearHashTable.TABLE.length) { + nthRun = 0; + chunkSize = 0; + return true; + } + } + if (chunkSize >= maxChunkSize) { + chunkSize = 0; + return true; + } + return false; + } +} diff --git a/parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunkers.java b/parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunkers.java new file mode 100644 index 0000000000..d721cb4d95 --- /dev/null +++ b/parquet-column/src/main/java/org/apache/parquet/column/impl/CdcChunkers.java @@ -0,0 +1,38 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import java.util.HashMap; +import java.util.Map; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; + +/** + * Internal: the content defined chunkers of one {@link org.apache.parquet.column.ParquetProperties}, + * one per leaf column. Every store made with those properties takes its chunkers from here, so chunk + * boundaries continue across row groups instead of restarting in each, as in arrow-rs. + */ +public final class CdcChunkers { + + private final Map chunkers = new HashMap<>(); + + CdcChunker chunker(ColumnDescriptor path, CdcOptions options) { + return chunkers.computeIfAbsent(path, p -> new CdcChunker(options, p)); + } +} diff --git a/parquet-column/src/main/java/org/apache/parquet/column/impl/ChunkingColumnWriter.java b/parquet-column/src/main/java/org/apache/parquet/column/impl/ChunkingColumnWriter.java new file mode 100644 index 0000000000..496e030607 --- /dev/null +++ b/parquet-column/src/main/java/org/apache/parquet/column/impl/ChunkingColumnWriter.java @@ -0,0 +1,135 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import org.apache.parquet.column.ColumnWriter; +import org.apache.parquet.column.ParquetProperties; +import org.apache.parquet.io.api.Binary; + +/** + * Ends the wrapped writer's page wherever its {@link CdcChunker} places a boundary, and applies the + * page limits the way Arrow C++ does under content defined chunking: measured from the page start + * rather than on the store's schedule, so that they cut a chunk in the same places after an edit. + * + *

A decorator rather than a change to {@link ColumnWriterBase}, so that a writer with content + * defined chunking disabled runs exactly the code it always has. + */ +final class ChunkingColumnWriter implements ColumnWriter { + + // Arrow's default write_batch_size: the size limits are checked once per batch of this many + // levels, counted from the page start, as Arrow C++ and arrow-rs check them. + private static final int BATCH_SIZE = 1024; + + private final ColumnWriterBase writer; + private final CdcChunker chunker; + private final int pageSizeThreshold; + private final int pageValueCountThreshold; + private final int pageRowCountLimit; + + private int levelsInBatch; + + ChunkingColumnWriter(ColumnWriterBase writer, CdcChunker chunker, ParquetProperties props) { + this.writer = writer; + this.chunker = chunker; + this.pageSizeThreshold = props.getPageSizeThreshold(); + this.pageValueCountThreshold = props.getPageValueCountThreshold(); + this.pageRowCountLimit = props.getPageRowCountLimit(); + } + + @Override + public void write(int value, int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offer(value, repetitionLevel, definitionLevel), repetitionLevel); + writer.write(value, repetitionLevel, definitionLevel); + } + + @Override + public void write(long value, int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offer(value, repetitionLevel, definitionLevel), repetitionLevel); + writer.write(value, repetitionLevel, definitionLevel); + } + + @Override + public void write(boolean value, int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offer(value, repetitionLevel, definitionLevel), repetitionLevel); + writer.write(value, repetitionLevel, definitionLevel); + } + + @Override + public void write(Binary value, int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offer(value, repetitionLevel, definitionLevel), repetitionLevel); + writer.write(value, repetitionLevel, definitionLevel); + } + + @Override + public void write(float value, int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offer(value, repetitionLevel, definitionLevel), repetitionLevel); + writer.write(value, repetitionLevel, definitionLevel); + } + + @Override + public void write(double value, int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offer(value, repetitionLevel, definitionLevel), repetitionLevel); + writer.write(value, repetitionLevel, definitionLevel); + } + + @Override + public void writeNull(int repetitionLevel, int definitionLevel) { + beforeTriplet(chunker.offerNull(repetitionLevel, definitionLevel), repetitionLevel); + writer.writeNull(repetitionLevel, definitionLevel); + } + + @Override + public void close() { + writer.close(); + } + + @Override + public long getBufferedSizeInMemory() { + return writer.getBufferedSizeInMemory(); + } + + /** + * Ends the page before this triplet at a chunk boundary or where a page limit is reached. Pages + * only end at a record start: the row count limit at any, the size limits at the first once a + * batch is full. + */ + private void beforeTriplet(boolean chunkBoundary, int repetitionLevel) { + if (repetitionLevel == 0) { + if (chunkBoundary || writer.getPageRowCount() >= pageRowCountLimit) { + endPage(); + } else if (levelsInBatch >= BATCH_SIZE) { + if (writer.getCurrentPageBufferedSize() >= pageSizeThreshold + || writer.getValueCount() >= pageValueCountThreshold) { + endPage(); + } else { + levelsInBatch = 0; + } + } + } + levelsInBatch++; + } + + /** A boundary on the first value of a page has no page to end. */ + private void endPage() { + if (writer.getValueCount() > 0) { + writer.writePage(); + } + levelsInBatch = 0; + } +} diff --git a/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriteStoreBase.java b/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriteStoreBase.java index 9bc7726491..08558d2dc6 100644 --- a/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriteStoreBase.java +++ b/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriteStoreBase.java @@ -23,6 +23,7 @@ import static java.util.Collections.unmodifiableMap; import java.util.Arrays; +import java.util.HashMap; import java.util.Map; import java.util.Map.Entry; import java.util.Set; @@ -44,7 +45,7 @@ abstract class ColumnWriteStoreBase implements ColumnWriteStore { // Used to support the deprecated workflow of ColumnWriteStoreV1 (lazy init of ColumnWriters) private interface ColumnWriterProvider { - ColumnWriter getColumnWriter(ColumnDescriptor path); + ColumnWriterBase getColumnWriter(ColumnDescriptor path); } private final ColumnWriterProvider columnWriterProvider; @@ -53,6 +54,8 @@ private interface ColumnWriterProvider { private static final float THRESHOLD_TOLERANCE_RATIO = 0.1f; // 10 % private final Map columns; + // Content defined chunking wraps each writer once, with a chunker that can outlive the store. + private final Map chunkingColumns = new HashMap<>(); private final ParquetProperties props; private final long thresholdTolerance; private long rowCount; @@ -71,7 +74,7 @@ private interface ColumnWriterProvider { columnWriterProvider = new ColumnWriterProvider() { @Override - public ColumnWriter getColumnWriter(ColumnDescriptor path) { + public ColumnWriterBase getColumnWriter(ColumnDescriptor path) { ColumnWriterBase column = columns.get(path); if (column == null) { column = createColumnWriterBase(path, pageWriteStore.getPageWriter(path), null, props); @@ -96,7 +99,7 @@ public ColumnWriter getColumnWriter(ColumnDescriptor path) { columnWriterProvider = new ColumnWriterProvider() { @Override - public ColumnWriter getColumnWriter(ColumnDescriptor path) { + public ColumnWriterBase getColumnWriter(ColumnDescriptor path) { return columns.get(path); } }; @@ -126,7 +129,7 @@ public ColumnWriter getColumnWriter(ColumnDescriptor path) { columnWriterProvider = new ColumnWriterProvider() { @Override - public ColumnWriter getColumnWriter(ColumnDescriptor path) { + public ColumnWriterBase getColumnWriter(ColumnDescriptor path) { return columns.get(path); } }; @@ -147,7 +150,13 @@ abstract ColumnWriterBase createColumnWriter( @Override public ColumnWriter getColumnWriter(ColumnDescriptor path) { - return columnWriterProvider.getColumnWriter(path); + ColumnWriterBase column = columnWriterProvider.getColumnWriter(path); + if (column == null || !props.isContentDefinedChunkingEnabled()) { + return column; + } + return chunkingColumns.computeIfAbsent( + path, + p -> new ChunkingColumnWriter(column, props.getCdcChunkers().chunker(p, props.getCdcOptions()), props)); } public Set getColumnDescriptors() { @@ -224,7 +233,12 @@ public void close() { public void endRecord() { ++rowCount; if (rowCount >= rowCountForNextSizeCheck) { - sizeCheck(); + if (props.isContentDefinedChunkingEnabled()) { + // ChunkingColumnWriter applies the page limits; the schedule only flushes the cached nulls. + rowCountForNextSizeCheck = rowCount + props.getMinRowCountForPageSizeCheck(); + } else { + sizeCheck(); + } } } diff --git a/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriterBase.java b/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriterBase.java index 408627404d..6ba49fa614 100644 --- a/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriterBase.java +++ b/parquet-column/src/main/java/org/apache/parquet/column/impl/ColumnWriterBase.java @@ -365,6 +365,10 @@ int getValueCount() { return this.valueCount; } + int getPageRowCount() { + return this.pageRowCount; + } + /** * Writes the current data to a new page in the page store */ diff --git a/parquet-column/src/main/java/org/apache/parquet/column/impl/GearHashTable.java b/parquet-column/src/main/java/org/apache/parquet/column/impl/GearHashTable.java new file mode 100644 index 0000000000..c0c961fd42 --- /dev/null +++ b/parquet-column/src/main/java/org/apache/parquet/column/impl/GearHashTable.java @@ -0,0 +1,571 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +/** + * The gear hash tables used by {@link CdcChunker}, copied from Arrow C++ + * ({@code cpp/src/parquet/chunker_internal_generated.h}) -- do not edit. + * + *

Eight tables of 256 64-bit values. Table {@code r}, byte {@code b} is {@code TABLE[r][b]}. + * arrow-rs ({@code parquet/src/column/chunker/cdc_generated.rs}) ships the same tables. + * {@code TestGearHashTable} pins them to the MD5 specification they were generated from. + */ +final class GearHashTable { + + private GearHashTable() {} + + static final long[][] TABLE = { + // seed = 0 + { + 0xf09f35a563783945L, 0x0dcc5b3bc5ae410aL, 0x63f1ea8d22554270L, 0xfbe5ee7bd05a7b61L, + 0x3f692ed5e9934abaL, 0xaab3755952250eb8L, 0xdefb168dc2888fa5L, 0x501b36f7c77a7d47L, + 0xd2fff45d1989642dL, 0x80217c1c600e30a6L, 0xb9469ee2e43df7acL, 0x3654b76a61999706L, + 0x6ea73dfe5de0c6b6L, 0xdfd662e1937a589dL, 0x0dbe0cc74b188a68L, 0xde45f4e6d73ffc6fL, + 0xcdf7a7759e70d87eL, 0x5d6a951b8d38c310L, 0xdc9423c3813fcf2cL, 0x25dc2976e167ffceL, + 0xc2555baa1d031c84L, 0x115bc3f2230a3ab6L, 0xd4b10260f350bedeL, 0xdfd3501ab447d723L, + 0x022e79217edaf167L, 0x1635e2255c5a7526L, 0xa0a750350cc77102L, 0xc027133e05d39f56L, + 0xd949459779cf0387L, 0xb92f1464f5c688c2L, 0xd9ac5f3e8b42f2f3L, 0xdf02bb6f5ecaac21L, + 0x8156f988fac7bfa4L, 0xe4580f97bede2ec8L, 0x44fe7d17a76fca32L, 0x885f59bd54c2014cL, + 0x435e63ec655ffae9L, 0x5ebc51930967b1f1L, 0x5428c2084ac29e47L, 0x9465938fec30e36bL, + 0xc7cb3de4977772cdL, 0x15692d7c201e8c3aL, 0x505ee65cdc4b17f4L, 0x7d9839a0a7aead6bL, + 0xeef5f5b6a0105291L, 0x76c2fb232ce7f5bfL, 0x5c13893c1c3ff3a9L, 0x65b6b547d4442f98L, + 0xb8ad7487c8c96fceL, 0x906bcf51c99974f8L, 0x2f56e48bb943a48cL, 0xbc9ab109f82d3a44L, + 0xcd5160cdc8c7e735L, 0xbe9acb9df3427732L, 0x386b91d477d7fadeL, 0x36be463621dd5af2L, + 0xcbe6a2faffd627a8L, 0x9c8fd528463a2f5aL, 0xb9b88c6bb802b184L, 0xb414b4e665c597c7L, + 0xbedb142568209556L, 0x5360d81c25429dceL, 0x63a69a960a952f37L, 0xc900d63899e1b503L, + 0x1abc63a8b37c7728L, 0xa8b3a8b6409080ebL, 0x495e391f662959f6L, 0xdf1e136f3e12229bL, + 0x33d5fc526b0dd38dL, 0x321221ae2abfac63L, 0x7fde18351fda7395L, 0xed79fe5c3a6aa4c3L, + 0x2dd6965a4867d8d4L, 0x54813ca20fe8799bL, 0x5d59ea6456465c39L, 0x0de0c294d1936b81L, + 0x4aaf0755002c588cL, 0x3530a1857ad04c6dL, 0xb8a64f4ce184442bL, 0xe0def10bceedfa17L, + 0x46e38d0a443757ecL, 0x9795a1c645ee16d7L, 0x7e531def245eac8aL, 0x683b25c43a0716cfL, + 0x884583d372da219dL, 0x5b06b62c910416e5L, 0x54b6902fbebd3dbeL, 0x931198d40a761a75L, + 0xead7d8e830013590L, 0x80b4d5dc99bfacedL, 0xf98272c8108a1ad2L, 0x1adce054289a0ec6L, + 0x7d53a1143c56b465L, 0x497fbe4f00c92b52L, 0x525e4cc2e81ebd69L, 0xc94478e0d5508ff6L, + 0xb8a5da83c196d07cL, 0x7667a921b65b0603L, 0xf236fabbdefe6cd1L, 0x53da978d19a92b98L, + 0xc604f6e97087124dL, 0x2cbd27221924b094L, 0x65cd1102c985b1d2L, 0x08c0755dc1a97eb4L, + 0x5e0419e921c0fef1L, 0x282d2c1196f84a29L, 0xe21117fcfc5793f7L, 0xcf4e985dc38e6c2eL, + 0xd521f4f264d55616L, 0xde69b04c485f2a10L, 0x59410e245305178aL, 0xceab1d477c943601L, + 0xa9805732d71ee5e9L, 0x054cd443896974f6L, 0xf2b517717a423a3eL, 0x09517937fa9fac95L, + 0x4938233e9ca871e3L, 0x9132cbaf56f83ec0L, 0x4703421ed1dd027dL, 0xfd9933f4e6f1ec4eL, + 0xf237c7fded2274a8L, 0xdf4616efe68cd7b4L, 0x5e46de0f39f0a380L, 0x3d41e0c6d8e095b0L, + 0xc5272f8a5bb2df09L, 0x68aa78e8301fb964L, 0xbf5b5b52c8e32ae0L, 0xbf28ed3df74bdcf7L, + 0xd6198f64c833815aL, 0x8cd99d2974267544L, 0xd90560ea4465ff2cL, 0x571d65ad7ad59261L, + 0x309453518baa367aL, 0xa60538377bc79fb2L, 0xace515da1ab4183cL, 0xf56d3c8d891d1c5bL, + 0x5b0d8370b59def49L, 0x775866ce7c83c762L, 0x3d76085695c8e18aL, 0xba064d1a9af1b114L, + 0xc84ef7cd7b98b521L, 0x90b9231681c2bc37L, 0x37e2b13e6f585b6bL, 0x1d0a34e55e0f369fL, + 0x86bb8019cf41447cL, 0x4b95c6ef55b3f71fL, 0x3b6ed1660732b310L, 0x617eee603d137f21L, + 0xf4f6278b464f3bbcL, 0xdfb763b720da205aL, 0x353478899b871cb7L, 0xe45fbbff574cc41eL, + 0x1a94b60847907d72L, 0xb10eef051eff67a5L, 0xf0e012ec6a284d40L, 0xcc1cd1a11b926d7cL, + 0xcf9d9c5453e19cadL, 0x270febcc0fc0e86bL, 0xd6567568778b781eL, 0x7323b98965eeb46bL, + 0xccecd374567086ffL, 0xef7b44bfc497a704L, 0xebc479c051a9f0a5L, 0xc9b7410e3e00a235L, + 0x1d084f7ecdf83dabL, 0xc8a9a97e33ba8ba3L, 0x8c75318f5b2350d6L, 0xaa3cd5d0c684bddaL, + 0xa81125fe0901bedfL, 0xf7bcd76020edfc93L, 0x834ee4c12e75874fL, 0xb2bb8a7beb44fa14L, + 0x32cd26f50a4f4e4dL, 0x0fc5817ca55d959aL, 0xd6e4ae2e3ae10718L, 0x074abdcceb8d6e38L, + 0xc0cc5f4f9b3a9c43L, 0x1115d364363595b2L, 0x69861db2eb19f2e8L, 0x59b8d804cf92bc67L, + 0x9bac9785e5e4b863L, 0x7fa0e17a41869561L, 0x10d3c9633f0c709cL, 0x534a03deee6bc44aL, + 0x73b1f7201257f581L, 0x46fd6a11e2e0706bL, 0x494abb554946e67aL, 0xb5d6da317864dc8eL, + 0x402ded9238f39687L, 0xd8fa37d2cbd6d290L, 0xcc818293fcb06791L, 0x6482ab344806cd4dL, + 0x0956e6ee9d8eb60bL, 0x01fee622d8465ac8L, 0xae7ece370cbd9c35L, 0x7ff09e937a177279L, + 0xa2c29ee7a33ca5f1L, 0x990e8dbee083923bL, 0x4a819b72f610863aL, 0xddecfad79d3f08beL, + 0x627372480fac20a7L, 0x802154d6eca2db4cL, 0x8fcf02e42f805e55L, 0x040a911ff8cea977L, + 0xbb544485bc64d0d4L, 0xaddde1aeb406d0fbL, 0xf6b35fae23dce66fL, 0xc07a9fb3645d2f9bL, + 0xccd113907e9c0fedL, 0xd17af369984fd213L, 0x9223823c59a083e7L, 0xe19d475606b81013L, + 0xe181ac116a90e57aL, 0x71f7b6258c6def4cL, 0x2246f34b45964f7cL, 0xd74aedaea2d31751L, + 0xb1add86e5dd305d1L, 0xeb9ba881f16d6471L, 0xef7600e036f5c6ffL, 0x1d50bc9735b8fb85L, + 0xe63942bd1f3e2969L, 0x9241ba9f8b3f4e72L, 0xee8bb2bca07d35b6L, 0x55cd55dab522654eL, + 0x94d0cfa7c1a6845dL, 0x02f9845d559884c3L, 0x8ce70ea21063b560L, 0xd70998028ef08b74L, + 0xdfdb5bbee310876bL, 0x4e21b2e348256d16L, 0xde007a981c13debcL, 0xe51950cbbddabfddL, + 0xd223301dbe9957c1L, 0x084b8634cc2cce4bL, 0x90e551378aa9d70cL, 0x833b533ac633e448L, + 0x7891e232882da57fL, 0xa1bf26f0163ce2b3L, 0xf33a0171eb9c68d5L, 0x2e7de18ca69b3fa2L, + 0x666fd6f175619199L, 0x1239d37edb5feb9fL, 0xfa9fc9382e61ff5cL, 0x3ca4ad427e3c126fL, + 0x37c6dd4c2c31ae6eL, 0x1f1bacb619d427b2L, 0x7dd09f5d10759afeL, 0xc8d941432327d733L, + 0x2b389ba25e1d43a7L, 0xa4e3030c3740ff21L, 0xcc56dae13fd37463L, 0x2481457c175b560fL, + 0x9deb35bde77c5c41L, 0x847aa6ea5549a0c3L, 0xcde01bb48b6e7f02L, 0x15a28844e64cb211L, + }, + // seed = 1 + { + 0xecfcba92fe5691a3L, 0x71377799fea34699L, 0xb284c9096fa614e5L, 0x54534170f40de6c8L, + 0xbbd804d45884fba3L, 0x44929a896388c8a1L, 0x79b712508e0fa3b1L, 0xeb53ab280af31054L, + 0x351ea23a6319da7aL, 0x2fbe55d9819d85a2L, 0x34f4b6568dcd28b1L, 0x8c94ea5e5d82967aL, + 0x09068d333a46d3c5L, 0x762ad4f64cb73381L, 0xd5c6db5ef0e22640L, 0x36d8ab5a36175680L, + 0xd41fe333cdc3525aL, 0xa1f51dbdf20ce781L, 0x1410a95e786c8be6L, 0x96b7499a670c2b41L, + 0x3912e1037835d893L, 0x272c5bd83e1e9115L, 0x2ea7f91cad82a0d6L, 0xcd10e85662ce9931L, + 0xedad49be8d5e8b74L, 0x7ccd8fe0f37d12bcL, 0xfac0482005eed593L, 0x4513991681f6c8b0L, + 0x2804d612eb0ad37dL, 0x7cca9e8412b81d34L, 0x85ffd6707192b7b8L, 0xea0560aeea954411L, + 0x0122d28226102bbaL, 0xf51c47cdbd22fdd1L, 0x3707d851183ff17cL, 0xaef5a1465f3e902dL, + 0xbcb38c2d8736a04fL, 0x4025317e864bef15L, 0x8d3f66d86e1ea58fL, 0xc16759a3d97ed79aL, + 0x1c62abdc0659f2f5L, 0x23b3eb4e699bd28fL, 0x5083c4fceed3ccafL, 0xa65bf34562cc989cL, + 0xaa5865932fd79064L, 0xf24d08d268c24593L, 0x7fbd00a215196999L, 0x7812cd366d752964L, + 0x62e8dcb27ef3d945L, 0xf08b7984e1b946dcL, 0x547d23ad9a5c1dcfL, 0x496b1fb249b27fb7L, + 0xcd692e1db5f3b3baL, 0x41931e39f1e1bc61L, 0x286c6a7d7edae82bL, 0x17ef6638b6c4ca6eL, + 0x609beb5a2576a934L, 0xcc5e16fe4a69b83cL, 0xbbd14d08b078fc24L, 0x2a617680f481cb94L, + 0x81dbbd5f86e6d039L, 0xeb8205e1fc8ecc3cL, 0xe5e3bb576faa8042L, 0x5d6f1eb9d9df01b5L, + 0x9a47b8739c10fb44L, 0x398a7caad7ea7696L, 0x9c0fc1d7c46adde6L, 0x67cd6de0a51978a6L, + 0x68ccc4b77a21cca4L, 0x1e067066b82f415cL, 0xf7ddade6535e1819L, 0xf2185c884291751bL, + 0xc322b7381fcbe34fL, 0x242f593e88290b9bL, 0x8e11ccc0ea5e84a3L, 0x40e3a2e3346db8a2L, + 0xf18bfc3ad2931a2cL, 0x2468397394b00144L, 0xeae199cce14e6817L, 0x05b462686c75a1aeL, + 0xda096cb859c51673L, 0xd87aeb967a906befL, 0xaabc74493cb02fe6L, 0x74d48fc2e7da143eL, + 0x6ec1c8fed3f2c1fdL, 0xe01e0704b463f18eL, 0xc3d88a4d3a8056e4L, 0xd01ae0ffab6c8f3fL, + 0x881ba052620ae7c7L, 0xcea033aef0a823a5L, 0x8d2cad91d83df1e3L, 0x18746d205e66dbe9L, + 0x3061f8e58d046650L, 0xd819c59f0ce2cf8bL, 0x144e89e93635e870L, 0x3415e88279b21651L, + 0xd6f7ab944b86c3faL, 0x45f1dd15d0f67bdcL, 0xbf0d97c7f4fa24f4L, 0x34a7de520a57fcd2L, + 0x4ba86fda03e9e2bcL, 0xa7995265a025b552L, 0x698f6819d5f51cf7L, 0xd07dbe9d8a156981L, + 0x2683945373857fc1L, 0x116f8a84f96167deL, 0x8bc832bd85595ebfL, 0xb206519d74fdfafaL, + 0xde9519b2e9b5cc5fL, 0x16fdd6f2da1d8163L, 0x7ba32bd48ef56f11L, 0x6f4e4d7ee8b29717L, + 0xd31576dde7468aadL, 0x023bb08848676045L, 0xf6dcc083178160b7L, 0x42035f426250e683L, + 0x343732993cfed89fL, 0x0640a870a22d3d58L, 0x65cff80b53b4ae6aL, 0x27996fa17ab05215L, + 0xfd5db01401b21a04L, 0x894508784bc1673cL, 0x5bfcf43a2380e27dL, 0x4cd6dcc2715583b7L, + 0xa43b3763e7d4c902L, 0x6da83e12ef0c1257L, 0xfe80a602b0335affL, 0x293a7d8f4ff344deL, + 0xb4ae7c2b8956bf5aL, 0x6b45432d38254b4dL, 0xd086acbdf15d9455L, 0xa4d19e43f41ea87bL, + 0xf01f13ba4bb87fbfL, 0xca582cf301a299ffL, 0x0ddad3d45298fa7dL, 0x0646a130459c3999L, + 0xc08e3af3747e2ceeL, 0xfc7db8aa9ed67295L, 0x783b329e7bd79d5fL, 0x732dbc607957af7bL, + 0x8e446ac19fb26555L, 0xff1dfa4d61dc89a5L, 0xb6fbc46bd8d011d8L, 0x185147ec5779f0d7L, + 0x6eb2cf6149a5380fL, 0xb0e773df803a1eaeL, 0xc07706c5519bfce5L, 0xc35abcf54fa95f14L, + 0x40a01d99a38608eaL, 0x776dcd6f603c277fL, 0x6ae12389b1d6d0bbL, 0x8bd981448df92bb9L, + 0x426a6a7ca21a2c16L, 0x87efd5b71c1bad26L, 0x71fb7fc4cd41de48L, 0xdd9033c45619d463L, + 0x40eaab322654cef7L, 0xe077fffed6f3e3a2L, 0x375a4dbef9384447L, 0x2066b009d2c4a100L, + 0xeca4a5794a068447L, 0x2128f64bddf341a1L, 0x738b4bb1be90bd61L, 0x433772cf3813d52eL, + 0x9540c88add8e4474L, 0x0b6d5decd21d3519L, 0x654ead966745642dL, 0xe1bfb03c3b4bdb4cL, + 0x0b977a9937515b1fL, 0x0a4587509ef63870L, 0xe89f0de1d9cfd44aL, 0x23a91390272e7f68L, + 0xd92defbc9096b8d8L, 0x004db87174612539L, 0xc88ecaabdd1a71f1L, 0x050de38393073346L, + 0x8af1426d7964e038L, 0xf352c4fef8ad5c87L, 0x6f26bc7408e26548L, 0x0d41543fd9bf3084L, + 0xfc4e07553a840fc6L, 0x5ef117de86a555a9L, 0x1f11c42dffb5ae1bL, 0x4147648f07490fa5L, + 0x09b35fd7671b21aaL, 0x1453b14f7ccca481L, 0x944f6fcce4c9b2baL, 0x5b08dd2e3583dc06L, + 0xe0220df78dc9c22dL, 0x1c200b9506cbf666L, 0x8a0b7465eadb523bL, 0xfbcb43a91a1e2d80L, + 0xe697f44be3c36a58L, 0x2f8a8e48fb7e350dL, 0x7baba71b8920d55fL, 0x10edc0216105bc96L, + 0x52db07c79d7a7a63L, 0x1916e8cef9452ac3L, 0x5cbbbf21f867b6ccL, 0xadd583365a690a4bL, + 0x4e4ca2c8bffc2fdbL, 0xf5fe3416d2eebcfeL, 0x839af8b85e452476L, 0x8496c0c54ad44e16L, + 0x6c46f1ecad4482bfL, 0xb794cad76ae18715L, 0x67b762eec7c62985L, 0x52dc9e68df5b3a53L, + 0x0cc7e444b422a5f9L, 0xadbfe90841c112b0L, 0xfe37b136f0ca5c34L, 0xcfe9e47948a8d73eL, + 0xee90572b86a30d91L, 0x549e72d8262830aaL, 0x3361564b469f32c6L, 0x1e6eba9e0d2648e2L, + 0x5f8e2b2ac5fcb4ebL, 0xe4224fa5f71f7cc6L, 0x7357a9230c76757bL, 0xcad70f74aaf6b702L, + 0xeef28ced23894cc2L, 0x753fdd3352aefd68L, 0x1fed6ba90bbeb9d2L, 0x05316f4ab4034b4bL, + 0x3396df022b9f63d6L, 0x82d7125a7cfd0935L, 0x3519a71caf1f87f0L, 0xd1dfb7a5cc3974beL, + 0xbfae40ecbdbbcc2aL, 0x152c11778e08dd54L, 0x4a96566a6c848554L, 0x3a84d621c340cdd7L, + 0xfd47aa1887e2fb03L, 0xa63cae94b2f1d099L, 0xed61783f3e5b75e0L, 0xefd44864106019beL, + 0x145ff78b80b081aaL, 0x34670e5fcea9230eL, 0x876ef976328db371L, 0x4221f3a5269942a6L, + 0x95315cbd85c648f4L, 0x3ca344dc7c3b1600L, 0x38421ea39ff28780L, 0x31dbeee967c0435cL, + 0x27437c3e268402e7L, 0xdd0cf8343312a654L, 0x965ab9dad1d8aa29L, 0xf871706dd3e23509L, + 0xce23d06c7a25e699L, 0x1b37d59382b27589L, 0x3407f004723d6324L, 0x56efb69cdb5deaa1L, + 0xf46cdd2b9fd604e0L, 0xcad3ca79fdac69bdL, 0x7252802a574e63cbL, 0xc281fb8acc6ec1d3L, + }, + // seed = 2 + { + 0xdd16cb672ba6979cL, 0x3954eaa9ec41ae41L, 0x52cb802771d2966dL, 0xf57ed8eb0d0294f2L, + 0x768be23c71da2219L, 0x6131e22d95a84ad3L, 0xd849e4e49bb15842L, 0x18e8e5c4978cf00dL, + 0x3af5e5867ce1f9bdL, 0x06c75a9fffe83d63L, 0xe8de75a00b58a065L, 0x0a773251bc0d755aL, + 0x629dc21e54548329L, 0x2a168f5e5a883e70L, 0x33547375f0996c86L, 0xdfcb4c7680451322L, + 0x55c1ecaaaa57e397L, 0x4546c346c24f5a31L, 0x6f8f0401dfabc86cL, 0x7760d2d36ee340b4L, + 0xf6448e48bdeb229dL, 0xba70e1633b4dba65L, 0x069cda561e273054L, 0xa010b6a84aebf340L, + 0x5c23b8229eee34b6L, 0xea63c926d90153afL, 0x7d7de27b3e43ec1bL, 0xea119541eddc3491L, + 0xf1259daeddfc724cL, 0x2873ca9a67730647L, 0xa1e7710dade32607L, 0x758de030b61d43fdL, + 0xd2c9bcbfa475edb4L, 0x18ade47bb8a0aa29L, 0xf7a74af0ff1aea88L, 0x6f8873274a987162L, + 0x6963e8d876f4d282L, 0xd435d4fe448c6c5bL, 0x93ec80ba404cafffL, 0xcf90d24c509e41e7L, + 0x5f0fc8a62923e36eL, 0x9224878fe458f3a4L, 0xd9a039edf1945bcdL, 0x0877d1892c288441L, + 0x75205491f4b4740bL, 0x30f9d2d523a9085bL, 0x4b7f4029fa097c99L, 0x170bb013745709d4L, + 0x7087af537f11ef2eL, 0x28c62b88e08fc464L, 0x84bbcb3e0bb56271L, 0x485a4b099165c681L, + 0x357c63357caa9292L, 0x819eb7d1aee2d27eL, 0xdaa759eb9c0f8c9dL, 0x42cdc36729cc3db5L, + 0x9489aa852eddbb06L, 0x8161e4f85a84e6d4L, 0xa964863fdad3eb29L, 0xcc095ddbce1a6702L, + 0x3ecfadbb8dc2ce58L, 0x971316509b95a231L, 0xc8f484d1dbc38427L, 0xae9c510c463574c0L, + 0xdf2b31179600c21aL, 0x440de87bada4dfa3L, 0xbd8d30f3f6fb7522L, 0x84e6d7f678a0e2d0L, + 0x0ec4d74323e15975L, 0xf6947610dad6d9abL, 0x73a55a95d73fe3a5L, 0x3e5f623024d37edaL, + 0x8d99a728d95d9344L, 0x8b82a7956c4acdc4L, 0x7faeaea4385b27f6L, 0x540625ff4aa2ff21L, + 0x4aa43b3ebd92ce2bL, 0x899646a6df2da807L, 0x49225115780942d7L, 0xe16606636af89525L, + 0xb980bcf893888e33L, 0xf9ed57695291b0d8L, 0x5c6dd14464619afaL, 0x50606d69b733d4f3L, + 0x7fb1af465b990f97L, 0x3fab2634c8bbd936L, 0x556da6168838b902L, 0x0f15975902a30e1fL, + 0xb29d782ae9e1991fL, 0xae00e26ff8f7e739L, 0xd3da86458bb292d5L, 0x4528ee0afb27e4ceL, + 0x49882d5ba49fabadL, 0x7e873b6a7cf875eeL, 0x777edd535113c912L, 0x94ed05e7ff149594L, + 0x0b8f95fc4211df43L, 0x9135c2b42426fef2L, 0x411e6c2b47307073L, 0x503207d1af0c8cf8L, + 0xd76f8619059f9a79L, 0x64d24617855dee45L, 0xf7bc7a877923196aL, 0xd6cc42ed6a65be79L, + 0xe3912ff09d4fc574L, 0x4192d03b2bc2460aL, 0xa0dcc37dad98af85L, 0xfc59049b2a5818a4L, + 0x2128bae90a5b975fL, 0xbe7067ca05ea3294L, 0x5bab7e7753064c4fL, 0x42cbf0949ef88443L, + 0x564df4bbd017492cL, 0xf2c2eb500cf80564L, 0x5b92e67eb00e92afL, 0x8c4103eef59c0341L, + 0x83412122b8284998L, 0x888daf2da0636b6dL, 0x4d54b10303dd07d6L, 0x201190e7c1e7b5edL, + 0x3797510bb53a5771L, 0x03f7bc598b570b79L, 0xdc1e15d67d94f73eL, 0x721e8b499ebe02c1L, + 0x71f954f606d13fa0L, 0x0c7a2e408c168bf0L, 0x07df2ef14f69c89dL, 0xe295096f46b4baafL, + 0x7a2037916438737eL, 0xd1e861aeaf8676eaL, 0xb36ebdce368b8108L, 0xb7e53b090ddb5d25L, + 0x5a606607b390b1aaL, 0x475e52994f4a2471L, 0xbcc2038ba55b2078L, 0x28b8a6b6c80df694L, + 0xb5f0130ec972c9a2L, 0x7a87cd2a93276b54L, 0x4d0eec7ecf92d625L, 0xac1a8ce16269a42eL, + 0xa4ca0237ca9637b8L, 0xd8dc8ff91202b6ffL, 0x75b29846799d7678L, 0x761b11a5edd9c757L, + 0xf2581db294ef3307L, 0xe3173c2b6a48e20fL, 0xe46fd7d486d65b3cL, 0x1352024303580d1fL, + 0x2d665dae485c1d6dL, 0x4e0905c825d74d3bL, 0x14ff470c331c229eL, 0xbdc656b8613d8805L, + 0x36de38e396345721L, 0xaae682c1aa8ff13bL, 0x57eb28d7b85a1052L, 0xf3145290231d443aL, + 0xd0f68095e23cbe39L, 0x67f99b3c2570b33dL, 0x54575285f3017a83L, 0x9b2f7bb03d836a79L, + 0xa57b209d303367a9L, 0x7ccb545dd0939c79L, 0x1392b79a37f4716dL, 0x6e81bb91a3c79bcdL, + 0x2c2cd80307dddf81L, 0xb949e119e2a16cbbL, 0x69625382c4c7596fL, 0xf19c6d97204fb95cL, + 0x1b2ea42a24b6b05eL, 0x8976f83cd43d20acL, 0x7149dd3de44c9872L, 0xc79f1ae2d2623059L, + 0xca17a4f143a414e1L, 0x66d7a1a21b6f0185L, 0xed2c6198fe73f113L, 0x16a5f0295cbe06afL, + 0x5f27162e38d98013L, 0xf54d9f295bdc0f76L, 0x9ba7d562073ef77bL, 0xa4a24daaa2cfc571L, + 0x49884cf486da43cdL, 0x74c641c0e2148a24L, 0xbff9dcbff504c482L, 0xf8fc2d9403c837abL, + 0x6ccc44828af0bb1eL, 0xbcf0d69b4c19dfdbL, 0x8fe0d962d47abf8fL, 0xa65f1d9d5514271dL, + 0x26ff393e62ef6a03L, 0xc7153500f283e8fcL, 0xea5ed99cdd9d15cdL, 0xfc16ac2ba8b48bb7L, + 0xf49694b70041c67aL, 0xbd35dd30f5d15f72L, 0xcf10ad7385f83f98L, 0x709e52e27339cdc2L, + 0xe9505cb3ec893b71L, 0x2ffa610e4a229af7L, 0x12e1bc774d1f0e52L, 0xe301a3bb7eacccc8L, + 0x1fdd3b6dcd877ebfL, 0x56a7e8bda59c05aaL, 0x99acd421035d6ab4L, 0xfd21e401cecd2808L, + 0x9a89d23df8b8d46fL, 0x4e26b1f1eb297b9cL, 0x9df24d973e1eae07L, 0xe6cdc74da62a6318L, + 0xfc360d74df992db0L, 0xf4eca0a739514c98L, 0x481c515ba9bf5215L, 0xce89cce80f5f3022L, + 0xf487a10fc80e4777L, 0x235b379a87e41832L, 0x76f72e028371f194L, 0xd044d4a201325a7dL, + 0x47d8e855e0ffbddeL, 0x268ae196fe7334b0L, 0x123f2b26db46faa8L, 0x11741175b86eb083L, + 0x72ee185a423e6e31L, 0x8da113dfe6f6df89L, 0x286b72e338bbd548L, 0xa922246204973592L, + 0x7237b4f939a6b629L, 0x31babda9bedf039aL, 0xb2e8f18c6aeec258L, 0x0f5f6ce6dd65a45eL, + 0x8f9071a0f23e57d3L, 0x71307115ba598423L, 0xcbe70264c0e1768cL, 0x1c23729f955681a8L, + 0xfbc829099bc2fc24L, 0x9619355cbc37d5d6L, 0xea694d4e59b59a74L, 0xb41cf8d3a7c4f638L, + 0xae1e792df721cd0bL, 0x7cd855d28aac11f6L, 0xca11ba0efec11238L, 0x7c433e554ce261d8L, + 0xe3140366f042b6baL, 0x8a59d68642b3b18cL, 0x094fcdd5d7bccac2L, 0x9517d80356362c37L, + 0x4a20a9949c6c74e8L, 0xc25bcf1699d3b326L, 0xa8893f1d1ed2f340L, 0x9b58986e0e8a886eL, + 0x29d78c647587ce41L, 0x3b210181df471767L, 0xd45e8e807627849dL, 0x1ec56bc3f2b653e3L, + 0x974ff23068558b00L, 0xdb72bdac5d34262cL, 0x23225143bb206b57L, 0xd0a34cfe027cbb7eL, + }, + // seed = 3 + { + 0x39209fb3eb541043L, 0xee0cd3754563088fL, 0x36c05fc545bf8abeL, 0x842cb6381a9d396bL, + 0xd5059dcb443ce3bfL, 0xe92545a8dfa7097eL, 0xb9d47558d8049174L, 0xc6389e426f4c2fc0L, + 0xd8e0a6e4c0b850d3L, 0x7730e54360bd0d0dL, 0x6ecb4d4c50d050d5L, 0x07a16584d4eb229fL, + 0x13305d05f4a92267L, 0xb278ddd75db4baecL, 0x32381b774138608fL, 0x61fe7a7163948057L, + 0x460c58a9092efee6L, 0x553bf895d9b5ff62L, 0x899daf2dabfd0189L, 0xf388ab9c1c4b6f70L, + 0xd600fe47027ea4cdL, 0x16d527ec2b5ef355L, 0x5ac1f58ff6908c81L, 0xa08d79ff8ee9ffe8L, + 0xc1060a80b7a5e117L, 0x14b2c23118c60bdaL, 0x8cc0defbb890df8fL, 0xe29540fd94c6d28bL, + 0xa604f003f82d5b71L, 0xa67583d4eb066d18L, 0xd62cbd796322b3fcL, 0x070cfe244cdcccf3L, + 0x73557c30b3af47e5L, 0x2e544e31153a2163L, 0x996eef7464d5beadL, 0xbc71cb5ab0586cdcL, + 0x0bfcb6c1b517ed69L, 0x62b4f1fcc82e8ca0L, 0x0edbc68f544965c5L, 0x40fa39baa24af412L, + 0xf39aeb2413dab165L, 0x17e6013e7afee738L, 0x8109bff1c8d42a9dL, 0x3cd99863390989b5L, + 0x02021a4cc9c336c8L, 0xa06060778cb60aa4L, 0xd96591db60bc1e06L, 0xd2727175183f4022L, + 0xcdc1f1c5bce3e7ceL, 0xb393ccc447872a37L, 0xdf6efe63257ead3aL, 0x20729d0340dbceb6L, + 0x9f3d2d26fc0ea0d7L, 0xf392e0885189bd79L, 0xdf2ee01eb212b8b6L, 0x6e103a0c0f97e2c3L, + 0x96c604a763bd841bL, 0x9fc590c43bba0169L, 0xf92dcd5ddc248c40L, 0x113a8b54446941dcL, + 0x5943eda146b46bb8L, 0xbf657901a36a39a7L, 0x5a4e0e7ea6568971L, 0xb94c635bae9f9117L, + 0x2626fb65b3a4ef81L, 0xa59bfd5478ce97deL, 0x79112ba9cc1a1c63L, 0xf41f102f002cf39cL, + 0x0a589bcbfb7ff1c8L, 0xa1478c53540c4fa1L, 0x60d55e72c86dfacaL, 0x312e7b6840ea7a39L, + 0x8aae72dcccfe1f75L, 0xff2f51f55bf0247aL, 0x3c2e4b109edb4a90L, 0x5c6d73f6525c7637L, + 0xe49acb04a199f61cL, 0x27860642d966df7fL, 0x541ce75fb1e21c30L, 0xd9fcd6f90806c7ccL, + 0xb87c27bc93a7969bL, 0x92f77a1179b8f8dcL, 0xb1f29379deb89ed4L, 0x7e63ead35808efe7L, + 0x13545183d7fa5420L, 0x575f593e34cf029dL, 0x27f1199fb07344aeL, 0xe67f95f7dc741455L, + 0x49b478b761ab850bL, 0xd7bedf794adfc21eL, 0xdc788dcd2dda40aeL, 0x14673eb9f4d8ad35L, + 0x0cced3c71ecf5eb1L, 0xe62d4e6c84471180L, 0xdfe1b9e2cb4ada7dL, 0x70185a8fce980426L, + 0x0ce2db5e8f9553d6L, 0x1fedc57bb37b7264L, 0xb9310a2e970b3760L, 0x989ff8ab9805e87dL, + 0x0b912d7eb712d9eeL, 0x1fe272830379e67cL, 0x16e6a73aff4738fbL, 0xeed196d98ba43866L, + 0x7088ca12d356cbe2L, 0x23539aa43a71eee0L, 0xed52f0311fa0f7adL, 0xa12b16233f302eeaL, + 0xc477786f0870ecb4L, 0xd603674717a93920L, 0x4abe0ae17fa62a4cL, 0xa18f1ad79e4edc8dL, + 0xc49fe6db967c6981L, 0xcc154d7e3c1271e9L, 0xdd075d640013c0c0L, 0xc026cd797d10922aL, + 0xead7339703f95572L, 0x4342f6f11739eb4bL, 0x9862f4657d15c197L, 0x4f3cb1d4d392f9ffL, + 0xe35bffa018b97d03L, 0x600c755031939ad3L, 0xb8c6557ffea83abfL, 0x14c9e7f2f8a122eaL, + 0x0a2eb9285ee95a7cL, 0x8823fec19840c46fL, 0x2c4c445c736ed1d0L, 0x83181dff233449f1L, + 0x15ed3fca3107bef5L, 0x305e9adb688a4c71L, 0x7dbef196f68a3e2eL, 0x93e47ece3e249187L, + 0x8353c5e890ead93cL, 0xea8a7ae66abafdf7L, 0xf956dbb6becf7f74L, 0x9f37c494fbfdb6e4L, + 0x11c6cbaa2485dd32L, 0x206f336fcca11320L, 0x9befe9a59135d8feL, 0x5f3ef8b8db92c7dbL, + 0xbb305e556ce0ce9aL, 0xf26bdafb1305887fL, 0xcbf28abe23f08c61L, 0x0bc64173b914e00bL, + 0x9168da52e983f54aL, 0x6ea41d09c3574a3eL, 0x78aa44d4a74459aeL, 0x2931422878387bf5L, + 0x018f64a3a92c2d9cL, 0x9be43f6752e66b34L, 0xae378890decd1152L, 0x07325329a1cb7623L, + 0x3b96f4ee3dd9c525L, 0x2d6ebcdbe77d61a3L, 0x10e32b0e975f510cL, 0xffc007b9da959bf9L, + 0x38bf66c6559e5d90L, 0xbe22bdf0bf8899feL, 0x87807d7a991632a8L, 0x149a0d702816766aL, + 0x026f723db057e9abL, 0xeeecb83625ec6798L, 0xcec2ed5984208148L, 0xd985a78e97f03c84L, + 0xf96c279e7927b116L, 0x99d5027b3204f6e2L, 0x13a84878c3d34c55L, 0x5cf5ec96229e9676L, + 0x0bc36b07e4f8e289L, 0xbed33b80a069914dL, 0x2fbfbdd1ff4b9396L, 0xab352bb6982da90fL, + 0x154d219e4fa3f62bL, 0x4d087512bb6b9be7L, 0xc582e31775ee400eL, 0x7dadb002ae8c4a4eL, + 0xaae2957375c1aee2L, 0x5f36ca643356625bL, 0xf87cf8eb76e07fb7L, 0x46f432a755e02cc3L, + 0x36087e07aba09642L, 0xe5642c1e4ebb9939L, 0xb9152d22338eefadL, 0xf7ba44278a22cf7fL, + 0xd3b8013502acd838L, 0x7761511da6482659L, 0xb0857621638e8e50L, 0x552eddb4a8b1d5f5L, + 0xc43d9861e812c3eaL, 0xd765c2aada47910cL, 0x21c935b68f552b19L, 0x6256d5641a2b47dcL, + 0xab711d8e6c94bc79L, 0xa8d0b91a2a01ab81L, 0x5e6d66141e8d632aL, 0x7638285124d5d602L, + 0x794876dbca3e471fL, 0x951937d8682670ceL, 0x0f99cb1f52ed466aL, 0x8c7cd205543b804cL, + 0x2fd24d74a9c33783L, 0xe5dcb7b7762e5af1L, 0x45e6749cca4af77cL, 0x540ac7ee61f2259fL, + 0x89c505c72802ce86L, 0xeab83b9d2d8000d1L, 0x9f01d5e76748d005L, 0xc740aaef3035b6d0L, + 0x49afcd31d582d054L, 0xcba5dc4c1efb5ddcL, 0xc0a4c07434350ca1L, 0xfc8dfaddcc65ee80L, + 0x157c9780f6e4b2d9L, 0x9762a872e1797617L, 0xc4afae2cf3c7e1bdL, 0x71cde14591b595d4L, + 0x8843c3e0e641f3b9L, 0xd92ecd91dce28750L, 0x1474e7a1742cb19fL, 0xec198e22764fa06bL, + 0x39394edb47330c7dL, 0x00ba1d925242533dL, 0xaed8702536c6fb30L, 0x6d3618e531c2967aL, + 0x77f7cedcd7cc0411L, 0xbc1e2ab82be5b752L, 0x07b0cf9223676977L, 0x596c693b099edd53L, + 0xbb7f570f5b9b2811L, 0x96bfdad3c4a6840cL, 0x668015e79b60c534L, 0x3ad38d72123f1366L, + 0x6b994d81d2fcbb09L, 0x70885f022c5052d8L, 0xc891ee79d9306a7bL, 0x2c4df05c0ed02497L, + 0x19ebc13816898be2L, 0xea7c64df11c392a2L, 0xb7663e88dd12e1bdL, 0x79f768cb8e154c21L, + 0x1fb21b12e945933bL, 0xe6a9045643f6906eL, 0x544c47acd7e15371L, 0xb7709b14f727e3d1L, + 0x326ee36a46942971L, 0x477f1cf7b0e2d847L, 0x88b8f6b82b3b0c24L, 0x18bc357b80e3cd5cL, + 0x3333de70e4d66e0bL, 0x4fd4c5e148583cf6L, 0xae1b62f3008c0af3L, 0xc49f419b6ab29cf5L, + 0x2c29fa65afc3fa28L, 0x4b19d93734d03009L, 0x7dd6c09e589276adL, 0x1cece97f30de48adL, + }, + // seed = 4 + { + 0x58bdf4338602e4fbL, 0x71a5620b02c926d5L, 0x3811c960129c2d9fL, 0x29c2fb11fccac567L, + 0x0d6b1ea7780f1352L, 0xcc4d3ddfae3f87b3L, 0xfdd30257362a586bL, 0xabc948fde69f25f1L, + 0x51b3523469d30f7bL, 0xe0f0322724405aceL, 0xd3729266d896da1eL, 0xb10c37e5147915bfL, + 0x8b577039f9fa32a3L, 0xe677c6a9cbfb44b3L, 0x7317a756ebb51a03L, 0xf8e988ef37359485L, + 0x600fc1ef3f469ff3L, 0xbf0b8f8520444e01L, 0x3711168b08b63d73L, 0x34146f2944a6cb36L, + 0x717feb263862cddeL, 0x7185f8347db00412L, 0x900798d82127e693L, 0x84089e976a473268L, + 0x10f8308c0d293719L, 0xf62a618d4e5719b8L, 0x8bdbd257a1a9516fL, 0xf49f666fd7a75110L, + 0xbaf45e2db7864339L, 0xe4efa1ea0c627697L, 0x3e71d4c82a09fe10L, 0x54a2a51cf12127bbL, + 0xa0592c9f54ba14cdL, 0x27dd627a101c7a42L, 0x3d2ceb44b3d20d72L, 0x7ee1f94a68ca8f5dL, + 0x7e8cb8651b006c36L, 0xbd9fa7ca3a475259L, 0x856de173586a7b34L, 0xcedb291b594cb1b5L, + 0xa3d6e462fd21cddcL, 0x74561d10af9118e4L, 0x13a3d389fc2d4b36L, 0xeea8594a4a054856L, + 0xf56d7474d9ba4b13L, 0x25ddce2f6490b2fdL, 0x920653ff3a8d830bL, 0xcd8c0c9cdac740d1L, + 0x2c348a738db9c4a0L, 0x2967ccbe8ea44c22L, 0x47963f69adb049f8L, 0xf9d01eb5b4cf7eb6L, + 0x7a5c26eb63a86bd2L, 0x62ad8b7a71fa0566L, 0xb373213179f250aeL, 0x589d4e9a88245a4dL, + 0x433dafebe2d558a8L, 0x521fbef2c8fe4399L, 0x62a31f9ff9ccd46bL, 0x51602203eba7c1a6L, + 0x9afc8c451b06c99fL, 0xb529085bdbaffceaL, 0xac251825cc75892bL, 0x94976a5bce23d58eL, + 0xdd17925b6c71b515L, 0x568fd07a57bce92eL, 0xefac31200d8bd340L, 0x716c3e466b540ef9L, + 0x3d2c9e380063c69bL, 0x14168f9a3662dd83L, 0xd298c7504dbc412fL, 0x74490a94f016719fL, + 0x0e0da431e1ab80c8L, 0xe321f63dc6b169aeL, 0xf08671544febc95aL, 0x39324450cc394b3bL, + 0xea6e3d35f1aa3a70L, 0x8ef8a886508ce486L, 0xdc1a631ef0a17f06L, 0xfda2b3fbcd79e87bL, + 0xd75bcae936403b10L, 0xf88b5bd9f035f875L, 0xc43efec2e3792dd4L, 0xe9fac21a9d47cd94L, + 0xc2876f0c4b7d47c3L, 0xaba156cf49f368b4L, 0x5ccda2170fa58bf9L, 0xadc92c879ed18df7L, + 0x110c1b227354e6c8L, 0x298ee7a603249200L, 0xde92142ede0e8ee7L, 0x88e4a4610644ba9eL, + 0xbb62d277e7641d3aL, 0xb9be1985b7bf8073L, 0x29024e5426cdb0d1L, 0xf6aefd01f3092ab8L, + 0x2a07087b313133aaL, 0x6d71f445d6dfc839L, 0x1e2412ff12e5526bL, 0xed5cdeba6617b9e1L, + 0x20b1d0d5e5f8760eL, 0x12ff15705c368260L, 0x7bf4338b7c387203L, 0x34ff25f00cd06185L, + 0x1148c706c518cf28L, 0x5c04f0623388f025L, 0xcb9d649275d87d79L, 0x9b5f0c24fabc42ecL, + 0x1a7b5e7964e33858L, 0x2a81bbd8efdc6793L, 0x8d05431ffe42752eL, 0x83915cd511002677L, + 0x580ed4d791837b31L, 0x5982e041d19ff306L, 0xcad0d08fa5d864caL, 0x867bee6efe1afa63L, + 0x26467b0320f23009L, 0xd842414dfda4ec36L, 0x047fcdcbc0a76725L, 0xbddb340a3768aecaL, + 0xef4ce6fa6e99ab45L, 0x88c5b66c7762bf9bL, 0x5679f1c51ffb225dL, 0xdab79048317d77eeL, + 0xf14e9b8a8ba03803L, 0xe77f07f7731184c1L, 0x4c2aab9a108c1ef5L, 0xa137795718e6ad97L, + 0x8d6c7cc73350b88bL, 0x5c34e2ae74131a49L, 0xd4828f579570a056L, 0xb7862594da5336fcL, + 0x6fd590a4a2bed7a5L, 0x138d327de35e0ec1L, 0xe8290eb33d585b0bL, 0xcee01d52cdf88833L, + 0x165c7c76484f160eL, 0x7232653da72fc7f6L, 0x66600f13445ca481L, 0x6bbdf0a01f7b127dL, + 0xd7b71d6a1992c73bL, 0xcf259d37ae3fda4aL, 0xf570c70d05895acfL, 0x1e01e6a3e8f60155L, + 0x2dacbb83c2bd3671L, 0x9c291f5a5bca81afL, 0xd976826c68b4ee90L, 0x95112eec1f6310a2L, + 0x11ebc7f623bc4c9aL, 0x18471781b1122b30L, 0x48f7c65414b00187L, 0x6834b03efa2f5c30L, + 0x0875ef5c2c56b164L, 0x45248d4f2a60ba71L, 0x5a7d466e7f7ba830L, 0x2bebe6a5e42c4a1dL, + 0xd871d8483db51d10L, 0x6ee37decd2fd392fL, 0x7d724392010cede3L, 0x8e96ef11e1c9bcc8L, + 0x804a61d86b89d178L, 0xbb1b83ce956055ecL, 0xcb44e107410ff64fL, 0xc426bb09ee0ba955L, + 0x057c08f42c3dd7f1L, 0x40ea1ec148602bdfL, 0xc24688deeb65d7f1L, 0xd8bcc53c768ba4e4L, + 0x16e0e3af65c1106cL, 0xfc12f7e7d647218bL, 0x70d6e1d3ee93cef4L, 0x01d2a505c4541ef9L, + 0x1ef79e16e764d5c3L, 0x0363d14d13870b98L, 0xb56ef64345d06b11L, 0xe653d557ebb7c346L, + 0x8304a8597c2b2706L, 0x1536e1322ce7e7bbL, 0x525aec08a65af822L, 0x91f66d6e98d28e43L, + 0xe65af12c0b5c0274L, 0xdf6ae56b7d5ea4c2L, 0x5cef621cedf3c81cL, 0x41e8b1ffd4889944L, + 0xb5c0f452c213c3e5L, 0x77af86f3e67e499bL, 0xe20e76ea5b010704L, 0xbdc205ab0c889ec0L, + 0xc76d93eb0469cd83L, 0x17ac27f65cab0034L, 0xd49ec4531fd62133L, 0x07a873ea2f1b9984L, + 0xbff270dfef0032eeL, 0x1764dbe91592f255L, 0xe40363126f79e859L, 0xa06cad3ab46971f6L, + 0x0be596e90dedd875L, 0x3387cce5c1658461L, 0x44246acf88a9585eL, 0xe0ad82b92d5ecb2cL, + 0x2177491c9a1600a6L, 0x16e7c4aac0f02422L, 0x75792eeeec15c4e1L, 0x2309cd359d08ee30L, + 0x7cd9831dd1b83b0aL, 0x374914a7c4ee8cf0L, 0x0dd17765c9ac2e54L, 0xb7847470ba9a7688L, + 0xfba4f4bbe2991173L, 0x422b203fc3de040eL, 0x63bfcaf2ecf2ab0eL, 0x0c5559f3a192946eL, + 0xfdf80675c1847695L, 0xf5f570accab842c9L, 0x65cc5a448767afeaL, 0x1efeb0a7ee234f2fL, + 0x9b05f03d81e7b5d2L, 0xe7c31317a8626cf4L, 0x620f2a53081d0398L, 0x1b6de96cdd9943aeL, + 0x8c226a436777d303L, 0xa08fbbd50fafb10dL, 0x6a64c5ec20104883L, 0x9c9c653502c0f671L, + 0x678a02b2174f52a0L, 0x68e008ba16bbad4bL, 0xa317c16d2efb860fL, 0xeab2075d17ed714cL, + 0x565eeeddf0c4ea15L, 0x8ec8e94d242a6c19L, 0x139e8e27d9000faeL, 0xc977a7ff1b33d2f5L, + 0x1d0accca84420346L, 0xc9e82602cd436e03L, 0x6a2231da53d2ccd3L, 0xb44b12d917826e2aL, + 0x4f4567c6a74cf0b9L, 0xd8e115a42fc6da8fL, 0xb6bbe79d95742a74L, 0x5686c647f1707dabL, + 0xa70d58eb6c008fc5L, 0xaaedc2dbe4418026L, 0x6661e2267bdcfd3dL, 0x4882a6eda7706f9eL, + 0xf6c2d2c912dafdd0L, 0x2f2298c142fd61f9L, 0x31d75afeb17143a8L, 0x1f9b96580a2a982fL, + 0xa6cd3e5604a8ad49L, 0x0dae2a80aad17419L, 0xdb9a9d12868124acL, 0x66b6109f80877facL, + 0x9a81d9c703a94029L, 0xbd3b381b1e03c647L, 0xe88bc07b70f31083L, 0x4e17878356a55822L, + }, + // seed = 5 + { + 0xb3c58c2483ad5eadL, 0x6570847428cdcf6cL, 0x2b38adbf813ac866L, 0x8cb9945d37eb9ad3L, + 0xf5b409ec3d1aed1cL, 0xa35f4bffc9bb5a93L, 0x5db89cde3c9e9340L, 0xff1225231b2afb2bL, + 0x157b0b212b9cc47dL, 0xf03faf97a2b2e04dL, 0x86fdab8544a20f87L, 0xfcb8732744ae5c1cL, + 0xd91744c0787986d5L, 0x5f8db2a76d65ad05L, 0xcff605cbed17a90dL, 0xf80284980a3164e7L, + 0x59cc24e713fccc7dL, 0x268982cada117ce4L, 0xcd020e63896e730eL, 0xe760dc46e9fe9885L, + 0x6aaece8ab49c6b5dL, 0x7451194d597aae3eL, 0x35d4385900332457L, 0xa40fb563a096583dL, + 0xa797b612f7f11b76L, 0x2fed6eb68e6a2b9bL, 0x2f06ee64aeffd943L, 0x9dd0e49d9ca45330L, + 0x97d48f08bd7f1d8fL, 0x1cfa7fe3ebe4d8eeL, 0x2a2ba076bd397d42L, 0x68c4344f7472f333L, + 0xce21ec31987d74b5L, 0xb73dabdc91d84088L, 0x801aadee592222feL, 0xaf41345398ebc3f5L, + 0x8a8f653d7f15ee46L, 0xce2d065ff2ba2965L, 0x4e05da515da2adb7L, 0xa6dbdb8aa25f0fd4L, + 0xca9f9666bbd2d5a9L, 0x6b917ce50bd46408L, 0x1550cc564ba6c84dL, 0xb3063ae043506504L, + 0x84e5f96bb796653dL, 0xe2364798096cf6e3L, 0x3b0dfedf6d3a53d0L, 0xb7e4c7c77bde8d93L, + 0xe99545bac9ab418aL, 0xa0e31f96889507bbL, 0x883c74f80c346885L, 0xf674ae0b039fd341L, + 0x8bb6ce2d5e8d1c75L, 0x0c48737966a7ed7cL, 0x04fcdf897b34c61cL, 0xe96ac181bacbd4d6L, + 0x5a9c55a6106a9c01L, 0x2520f020de4f45d3L, 0x935730955e94d208L, 0xce5ad4d7f3f67d3bL, + 0xa4b6d107fe2d81caL, 0x4f0033f50ae7944eL, 0x32c5d28dd8a645a7L, 0x57ce018223ef1039L, + 0x2cbab15a661ab68eL, 0x6de08798c0b5bec2L, 0xee197fb2c5c007c6L, 0x31b630ac63e7bda2L, + 0xab98785aefe9efe3L, 0xa36006158a606bf7L, 0x7b20376b9f4af635L, 0xa40762fdc3c08680L, + 0x943b5faffd0ebee2L, 0x7f39f41d0b81f06eL, 0x7c4b399b116a90f8L, 0x24e1662ac92bc9f3L, + 0xcf586fc4e8e6c7dbL, 0xe46e0d047eeb12d7L, 0xe8021076e4ea9958L, 0x11fc13492e3ca22aL, + 0xd61eae01410397e3L, 0x7e8c4a58036a8e9fL, 0x068a6de267970745L, 0x64faab129bef1a41L, + 0xb4a6f720943dad01L, 0x631491058d73a9d5L, 0xdad4fe95eab3ec02L, 0x0a8b141c5c3a44f6L, + 0x9fc69d4c2b335b98L, 0x94d5f84a07d6e4cdL, 0x1b73965de143c608L, 0x443932c2dda54bccL, + 0x7397818fb0b04cd2L, 0xef4ab03a1202b277L, 0xf3d2ee459c0c2b92L, 0x182d4daf8b058a87L, + 0x90e63035d7b51368L, 0xba4cd8b9a95d45fdL, 0x12a7392c76731090L, 0x890d264ec5d082d2L, + 0xeeaf5c363da4994eL, 0xd6aad756902123fbL, 0xb531ebebdb28f191L, 0xe71ce659fc59babdL, + 0x37c1b94f63f2dcb5L, 0xe4e3abeb311f9b96L, 0x4a31b72ccb8695d3L, 0x52cae1f0629fdce4L, + 0xe5b0475e2ed71369L, 0x2724e8c3506414fbL, 0xbab0367920672debL, 0x0161a781c305449fL, + 0x37b70f40f5bb60beL, 0xddd1094c50251a01L, 0x3b28283afd17224eL, 0x06dec0cfe889fc6bL, + 0x47608ea95bb4902dL, 0xad883ebc12c00e82L, 0x9e8d7ae0f7a8df29L, 0xa79443e9f7c013a1L, + 0xcfa26f68b7c68b71L, 0x33ae6cc19bda1f23L, 0xd9741e22b407887fL, 0xf2bff78066d46b1cL, + 0x794123191c9d32d4L, 0x56cb6b903764ec76L, 0x98775d0ef91e1a5aL, 0xae7b713bc15c1db9L, + 0x3b4c1a7870ed7a0dL, 0x46666965f305cc34L, 0x0ea0c3b2e9c6b3cdL, 0x4dc387039a143bffL, + 0x5f38bb9229ef9477L, 0xea5d39ba72af7850L, 0x69a5ed0174ce2b6dL, 0x06969a36bfe7594dL, + 0x0adee8e4065ccaa3L, 0x908a581d57113718L, 0x64822d6c5a8190edL, 0x8c5068b56ace4e4cL, + 0x88ba3b4fb4e30befL, 0xa6ec0b8bb5896cfeL, 0x4e23fcc6b47996fdL, 0xe18e75b0dd549c7aL, + 0xcd90f17e106cf939L, 0x1666fdfb2ef7c52fL, 0x4fae325f206dd88cL, 0xe7bc1160e25b062dL, + 0x3cc999cb246db950L, 0xc5930a7326cd5c37L, 0xb008a48a211367bdL, 0xc5559da145a88fd4L, + 0x1e3ad46655fac69cL, 0x7834266b4841bfd7L, 0xa764450fbffc58ccL, 0x54d8cf93a939c667L, + 0x93c51f11b21b2d9dL, 0x0964112082ed65ccL, 0x4c2df21213e7fb03L, 0xf0405bc877468615L, + 0x17b4fc835d116ab4L, 0xa6b112ae5f3cb4efL, 0x23cfc8a7fd38a46eL, 0x8e0a360dc2774808L, + 0x24ca9c8092105ad5L, 0xafd3f75524f2e0d5L, 0x4f39ed7dbaddc24cL, 0xe5e362c7679a7875L, + 0x00914a916b07b389L, 0xdfe1119b7d5ab5daL, 0xabd6ed9940e46161L, 0x630ed2044171e22cL, + 0xdecc244157dd1601L, 0x777e6d5b4b4868d5L, 0x9b3530bee67017d8L, 0xd2faf08b291fdcb9L, + 0x006e99455d6523deL, 0xd559b5817f6955b5L, 0xefcc1063b0088c61L, 0xed73145ae0f00ae7L, + 0xab2af402cf5b7421L, 0x897767f537644926L, 0x26c9c0473ca83695L, 0x192e34e1881b2962L, + 0xf7cf666ec3b3d020L, 0x27f9b79c7404afb7L, 0xe533e8bed3010767L, 0xe5817838e11d05d3L, + 0x65659c531bd36517L, 0xd427c5e0a23836fdL, 0xf3eab7ea58fa3528L, 0x07683adae1289f35L, + 0x201d6af7e896dd32L, 0xd5da938b9a21ad88L, 0x843fb73ad67bc316L, 0x1782ec7d5feef21bL, + 0x943f66f6ec772877L, 0x7e9112e7b26da097L, 0xeac8161f8663c2c7L, 0xe8600db480a9ebf4L, + 0x07807fc90f6eaf5fL, 0xe0e4c9deb41abf83L, 0xbdf533db271f9c15L, 0xb398411b0497afe2L, + 0xdebb45ef25448940L, 0xe7a5decefcd376c4L, 0xaf1ef3c728c83735L, 0xb8b83a99355cb15aL, + 0x6444a0344f1611e4L, 0xe8bb7f5cf3c60179L, 0x77ab5c5177e75ff7L, 0xc38fd6fa849d585dL, + 0x390d57d53029060aL, 0xa66327eb7b8b593cL, 0x6350a14f6fcd5ac9L, 0x2c08125bcd7008b4L, + 0x2d00c299a6a6bf8eL, 0x6b0039c1f68d1445L, 0x0035150c5d06f143L, 0xa34d01628cc927e1L, + 0xdf5b3164d7b2ede1L, 0x8167db1d0583d72eL, 0x4e13b341cd2ae8bcL, 0xa693d9b1f416e306L, + 0xc15ed7ca0bc67609L, 0xdc344313c1c4f0afL, 0x88b6887ccf772bb4L, 0x6326d8f93ca0b20eL, + 0x6964fad667dc2f11L, 0xe9783dd38fc6d515L, 0x359ed258fa022718L, 0x27ac934d1f7fd60aL, + 0xd68130437294dbccL, 0xaf5f869921f8f416L, 0x2b8f149b4ab4bf9fL, 0xc41caca607e421cbL, + 0x7746976904238ef9L, 0x604cb5529b1532f0L, 0x1c94cd17c4c4e4abL, 0xe833274b734d6bbeL, + 0xe9f1d3ef674539ceL, 0x64f56ed68d193c6aL, 0xe34192343d8ecfc1L, 0xcb162f6c3aa71fe8L, + 0x99eaf25f4c0f8fa4L, 0x92f11e7361cb8d02L, 0xb89170cddff37197L, 0x4f86e68a51e071e3L, + 0x31abf6afd911a75bL, 0x6d20cf259c269333L, 0x4150b9f88fcb6513L, 0x705063989ebf7451L, + 0x559231d927c84410L, 0x1ca8ec4b098bc687L, 0xebed22405c9180e0L, 0xaa815b37d052af59L, + }, + // seed = 6 + { + 0x946ac62246e04460L, 0x9cebee264fcbc1aeL, 0x8af54943a415652bL, 0x2b327ed3b17b8682L, + 0x983fde47b3c3847eL, 0x10a3013f99a2ad33L, 0x6e230bb92d2721efL, 0x1cf8b8369e5c5c50L, + 0x7f64017f2b7b3738L, 0xd393248a62417fa1L, 0x9ff01c0b20a372c5L, 0xb0e44abce7e7c220L, + 0xcebb9f88d48a815fL, 0xdb7df6bd09033886L, 0x7844fc82b6fa9091L, 0x72d095449863b8ecL, + 0xc13e678c89da2c7eL, 0x6caf4d5ad231d12fL, 0x2e0ab7b5fcf35c49L, 0xf410720cb932a70fL, + 0xd66ea581f16fce06L, 0x175c9f002f57dc98L, 0xccbcfd0d32988775L, 0xfde4c407d3b0a232L, + 0x5db2931ae7e97223L, 0x6e07e2173085809fL, 0x6e1d1ec0f9cad73cL, 0xb2fc251a7f802619L, + 0xbc1fc17f04f342deL, 0x8de8f21ec658e078L, 0x72c0f40cbee53fd6L, 0x0678244411fc17a1L, + 0x1d5837ca166b9bbdL, 0xc8cada003c554345L, 0x6a2fe2bfb2e58652L, 0xfca9d797a6f7988bL, + 0x6699e24ac737948bL, 0x69623ffcb05789baL, 0x946429c529d95b75L, 0x0d14df0b2a13970fL, + 0x593d8592c440dfecL, 0x2ee176f3d7e74b94L, 0xae003f1da3be9e26L, 0x0c7b02c4c0f6764aL, + 0x3117e2fa1f632462L, 0xf0f23265b6f1eaebL, 0x3111255d9b10c137L, 0xc82745e509a00397L, + 0xbd1d04037005fea7L, 0xe104ab0dd22a9036L, 0x51b27ce50851ac7aL, 0xb2cb9fb21b471b15L, + 0x29d298074c5a3e26L, 0x6ebdf2058b737418L, 0xc4a974041431b96fL, 0x1ec5a30ccb6bdaacL, + 0xe818beede9bf4425L, 0x4b69b1bce67a5555L, 0xf5c35f1eb0d62698L, 0xf4509bbd8e99867cL, + 0xb17206debd52e1bcL, 0x35785668c770b3beL, 0xe9343987ff5863bcL, 0x2ee768499ac73114L, + 0x5132bb3426eeaaf4L, 0x471bce2c6833c5ffL, 0xbb9a2d5428e6f6f9L, 0xd5678943c595792dL, + 0xab2a65e7f81e479cL, 0xa82407bb23990b31L, 0xdae321383984923cL, 0x01823bb22648e6f1L, + 0xda6e8df4214a8b04L, 0x0e172bb88e03d94fL, 0x552da6c22e362777L, 0x7ce67329fb0e90cbL, + 0x7b2d7f287ede7ebfL, 0xd44f8222500651bdL, 0x4acca1ef58fbb8abL, 0x428ecf058df9656bL, + 0xd7e1ec6a8987c185L, 0x365be6a54b253246L, 0x168849be1e271ee8L, 0x6a00f3c4151a8db2L, + 0x37602727ca94b33dL, 0xf6b50f18504fa9ceL, 0x1c10817f6bc872deL, 0x4bfe1fe42b0f3638L, + 0x135fad4b8ef6143bL, 0x1b25ad2bafc25f58L, 0x41e37f85cf321f92L, 0xfc73f75d9d5b9beaL, + 0x9eb3694d1e9cb7e1L, 0x601d51f08fa83b90L, 0x234a2a9b88366f41L, 0x63fe903e16f2c3bfL, + 0x1cdbd34fa751c0b0L, 0x0ce4fc6747c0558cL, 0x51ed72afb8bb49aaL, 0x20313ba13ca12c96L, + 0x271fa38f9ebd54c1L, 0x3696a5ac03a8eddeL, 0x05602be7df625702L, 0x11f1ac73790f7a9fL, + 0xa2836c099f0810bdL, 0xe5ac2e47caa532faL, 0xd9c000a66d39f681L, 0xd93d900e6f3d9d5fL, + 0x792c81c65b7900f2L, 0x5c5dce790ee20da1L, 0x74ff1950edec1aeeL, 0x71fc85fa1e277d8fL, + 0x0e77df17d6546cbcL, 0x07debad44816c3b4L, 0xbafa721581e92a70L, 0x8ab6fbe2ed27bba8L, + 0xe83243a20dea304aL, 0xaa85a63a84c00a07L, 0xde0e79917fc4153aL, 0x21bb445e83537896L, + 0xeedcac49fc0b433aL, 0xffb2926a810ae57aL, 0xf724be1f41d28702L, 0x79cb95746039bb3bL, + 0x5a54fe3742a00900L, 0xda4768d64922c04fL, 0x420396a84a339daeL, 0xa171e26ee5e8724eL, + 0x4c8da7c5d289c20aL, 0x9ebd79a1a8e94742L, 0x39235232b97e9782L, 0xb75df0be9bba7d80L, + 0x0c1d204dd87d48fcL, 0x8f81f3e7177266e8L, 0xe4a460b39e78d72bL, 0x50b98fa151e65351L, + 0xb7cb585c3ee1eddcL, 0x11cdad9a76ee1dc4L, 0xa38054a78595dc1cL, 0x92f09e2ec4978edcL, + 0xa8f0061b5efdabaaL, 0x04bcc4abc224d230L, 0xc58606738e692d46L, 0xdd2b27b565952433L, + 0x19e6ed1b740beec0L, 0xceadd49b2ef9891fL, 0x328178c28fe95cadL, 0xe5ad4c43afe02848L, + 0x03c0cb538cd967c0L, 0xec4352526d19a630L, 0x4c7e99389d39b031L, 0xf65dd05362c2deb6L, + 0xd1e70daf6879d28dL, 0xbe9f57db6309b265L, 0xa4b66f370b872bb7L, 0xe26896fbc6ee1fd5L, + 0xac705e661bfcf7c5L, 0xab4d0d07d7f09940L, 0x976417c06aeb6267L, 0x8161c684a6bd468cL, + 0xf77b6b9976dc4601L, 0xc6489b779a39c12cL, 0xb2aa58d5681cea1aL, 0x043b1b40f8c3e04cL, + 0x681fcbfadc845430L, 0xab8896c921ba8defL, 0x57aaf172606f37b2L, 0xc3735048cd5eb8d7L, + 0xa7078b96955631bdL, 0xdd6b3543aa187f33L, 0xc7103ea4a2a697fdL, 0x8d7b95f6ff1f7407L, + 0xe44f419e84709530L, 0xf340caa9132cbb0aL, 0x2ba407283143c66cL, 0xe1be240ca636c844L, + 0x90d32f2877ac08bcL, 0x5d26e6294b2c8673L, 0x4a6b2f5b27c87a44L, 0x961fb9043f76d34fL, + 0x0afee02d8d3c55d2L, 0x6228e3f48c42e5dcL, 0xc338e69ee6593675L, 0x853f74b16efb7bddL, + 0xd062f40bdd22e687L, 0x647164b9ab4c4190L, 0xf94689f67d598369L, 0x8e4b29d87a5012d7L, + 0xaf02b8b925656fbdL, 0x7a722a767179a630L, 0xb5c8afe937a75aceL, 0xfdb8e8d02d279372L, + 0x887ef700cb25fae1L, 0xcfe9bd912f72cabeL, 0xb1d4dedc24f978deL, 0x517522d38319cc2aL, + 0x7dd87b2b36aab798L, 0x579c4ff3046b5a04L, 0xf5c5975c5028b7a7L, 0x7094579d1000ec84L, + 0xbc8d5b1ea70a5291L, 0x161b2d783be8855cL, 0xd26d0b0d6d18279fL, 0x0be1945f02a78bd5L, + 0xb822a5a9e045415bL, 0x2fe9d68b1ccc3562L, 0xb2e375960033d14fL, 0x26aca04e49b4ff22L, + 0x732a81c862112aeaL, 0x8bd901ed6e4260b8L, 0xe839532c561ad5b0L, 0x8fb6e4d517a79b12L, + 0x0dd37f8c0be9b429L, 0xc8ad87ad12f1b1b0L, 0xc51f3aa62b90318bL, 0x031a7e8b86c1cefcL, + 0xa95547af2b70fc76L, 0x9cb3615c5a98801eL, 0xa387e3c3341d7032L, 0xa087ea52a1debaefL, + 0x16325ec9a2e6e835L, 0x587944a484c585ebL, 0xc8879033bde22eccL, 0xa39dbfce709c464aL, + 0x7acc010f99208774L, 0x98dd2973a096c5adL, 0x26458b51139f198cL, 0x2f5d19575e8c4f02L, + 0x726643f0d38af352L, 0x44d879b6d73e6e94L, 0xa68a03885c980abeL, 0x06048acd161c40c0L, + 0xa4dab8f89d405d28L, 0x7120c880cb04be18L, 0xa062ace22a1cf0cfL, 0x3901a9daf29704f4L, + 0xff08f3ed989db30aL, 0x6d22b13e874c67e9L, 0x80c6f35518d73f4dL, 0xc23c2a521aac6f29L, + 0x2e708fd83aaa42e0L, 0x7fc3780f55f1b0fdL, 0xabb3075c98cf87f2L, 0xb4df3f40f7c61143L, + 0x2a04418098a76d75L, 0x0d9eeee9509b2d37L, 0x6be8ae51f4b59cdcL, 0xe746cc7c00e4a2abL, + 0x785bc6df9cac597cL, 0x33cb6620ce8adc48L, 0xc1ba30739bffcef7L, 0x6d95771f18e503f7L, + 0xf7be3ae2e62652ffL, 0xc8d82ffd2a73c62bL, 0x8725a3ba5b110973L, 0x67ed6b9c724757ecL, + }, + // seed = 7 + { + 0xc0272d42c19ff3aeL, 0x4694228b43ea043bL, 0x5709a6ef8a462841L, 0xc9210a1e538805c9L, + 0x279b171196113ec2L, 0x859b769fc2d9e815L, 0x0d5d3125a2bf14d3L, 0x22bca1cfefa878baL, + 0x481b6bf58037bd83L, 0x4933ba8647728d22L, 0xf08c7b6b56f6e1b6L, 0x374e8af5a15407c7L, + 0xa95c4dc3d2487a5cL, 0x9b832808ff11e751L, 0xf2048507e9da01d5L, 0xa9c576189f544a4aL, + 0xf6c2a45b2e9d2b41L, 0x9b9874c9f10ecc2fL, 0x37d9b5f51f8c149eL, 0x93aead54c9de9467L, + 0x59cf0b4af262da23L, 0xe7e9929af18194b2L, 0x9df2644e33eb0178L, 0xde4122d6f0671938L, + 0xf005786c07f4800bL, 0xb1fc9d254b5d1039L, 0x0bf1088631f6dd7bL, 0x665623f0a4b8f0c7L, + 0x60f0113a9187db7cL, 0xfd7cceda4f0d23a6L, 0x26c01e9d89955940L, 0x33afa1dfc0f5a6a0L, + 0xeb77daf215e9283cL, 0xc7575214bf85edb4L, 0xeb0d804bf297e616L, 0x84bff4ffd564f747L, + 0xc4ac33189246f620L, 0x43ef61213ecc1005L, 0xcbbb0dea6cd96acdL, 0x8ed27abfa8cfcb05L, + 0x543b61529cb996b6L, 0xa5f987ca41ea5e59L, 0x3c50e0ac5254cb7aL, 0x4192b0446c06d1e6L, + 0x3e86592e21b45388L, 0xdb766f06fcc6e51eL, 0x0448ee36efe632dbL, 0x663c9db689253e35L, + 0x72e0bd4985331dd4L, 0xff501b5bf7d94e74L, 0xe911ce758e2113a8L, 0xec3a8d03a75a6ba4L, + 0xaf6b4b72f56edc83L, 0xf284857936c0a391L, 0x5ba6feff407d46f4L, 0x9d689c26de9d6702L, + 0x28c04a9083726b5dL, 0x2ccf4a627a029730L, 0x7b4719500d4f0c71L, 0x76470a9a7da250a8L, + 0xcc48409404a1c890L, 0xccefbdc7ec9a8055L, 0xe0db91bff3cc42d3L, 0x0532436426141254L, + 0xf2ee9325e6f0ff0bL, 0x149c20a5fbb28d9dL, 0xe71624cd8d2d14d4L, 0x8f01d4dc8cc2dd77L, + 0x29cf409b333015b7L, 0xba8bebd211884dd1L, 0xc3396635e8c8db1dL, 0x8ed0f6208d0528b8L, + 0x0d90b43fdd0ee334L, 0xd73c9a3333a044c7L, 0xa2595cd208dbdc38L, 0xae93cb264f940c09L, + 0x8e0538d8afb07a97L, 0x19115ec881385ba2L, 0xa886f9e6a8039c6aL, 0xcd5d62147ce3ecacL, + 0xaecdf9e0bb4969f7L, 0x2ddd631c53dcad10L, 0x73ad1c97b3412054L, 0xb08915fa2722efc6L, + 0x97966047e5067eb0L, 0x337f1675ed91445cL, 0xb3a833d150b96a0dL, 0x5940a98fe35e5e2eL, + 0xfd03cc354ed0d8ffL, 0x4e65b98291a8644aL, 0x14a259f2852a60b2L, 0x7648e3478c1e8e5fL, + 0xbc0fbef6d9a919b4L, 0xbec4302081346cf1L, 0x57d2ce7aa1c7c511L, 0x234c209d8f4e1ac3L, + 0x87cf80cc933ce443L, 0x7c262c616931e94eL, 0xc5e33b049cf9eddfL, 0x1a80790ed03ae51bL, + 0xf2e8b9494f7220cfL, 0x124cb59c14fff3ffL, 0xa8a06cbfdb86ce18L, 0x9068ef1f80b37653L, + 0x0c55417b8d90338fL, 0xcd579a523f6bcd30L, 0xa31bfe2476a8d2a9L, 0x1f8d142208094223L, + 0x332dc40a5203cfadL, 0xf8792fe5b2d33b4cL, 0x443bd9668bf9461eL, 0xc9019db0ace1409eL, + 0x781bea919a113e8bL, 0xb0f11d866abfbeecL, 0xcfe139a60db0c26aL, 0x869ab8721e6aa39eL, + 0xdb48a4977717837aL, 0x588a5ff151065b18L, 0xe4a251ea0028864dL, 0x7f0e43ba408a77c3L, + 0x65f66dd50a536135L, 0x6f49e934d9331c3eL, 0xb8d742e0f0fa6b09L, 0xe4e9b272deca2348L, + 0xaee132ff902f773cL, 0x43f658f7c2a0c90aL, 0x28cb4dbc76cc53eaL, 0x7d92253aa99ac39bL, + 0x4fea3d832370baabL, 0xb29e36936e51d78eL, 0xea10778712321064L, 0xff4f21f8ef274be2L, + 0x84eff18ddfa0933fL, 0xd0ec6a9f86c758a0L, 0xaf82e5973c431ae0L, 0x352023c00c045425L, + 0xad34d7bc4a2f8961L, 0xbdb4a02a24d4dee0L, 0x354a4846d97447cfL, 0x331a8b944d5bc19fL, + 0x5ce04f8e17909035L, 0x6497581bad8f4aabL, 0x07c503bba647111eL, 0x85f412ba78e1f7ffL, + 0x7f3b920fd20f4cffL, 0x424e1a9a4ce34e2fL, 0x3035e2d62e1b9f0aL, 0xef63114bff7b729aL, + 0xe86a05889ab6bb60L, 0xee0830cf095585a1L, 0x4a54f7fa47d9c94bL, 0x17daeece9fcb556aL, + 0xc506d3f391834c6fL, 0xb3f24be362e1af64L, 0xc435e4e23608efddL, 0xeeba9caaa4cc1768L, + 0x5a71f306daddc22dL, 0x18e5205f41eba1a0L, 0x7b29b4d1f6610925L, 0x065cb65a0258d9a9L, + 0x3e5ac8faa9fd1f95L, 0x3b362362c1ea0470L, 0xce0e4f6434db7a2eL, 0xf327341098de52f2L, + 0xcfca3b9e2a1992c3L, 0x7483bf9401233e41L, 0xbafbac531c6f9281L, 0x4b52dd71b2c106f8L, + 0xdf73b66e50b5a1f7L, 0x237aec0202a20283L, 0x23dd5be23dffdf2bL, 0xea9730731ee122efL, + 0x5cb3f846014fbcd3L, 0xc3b21c8ffdce9201L, 0x06a99a02f91a8760L, 0x721a81fa8fd7b7a3L, + 0x6aafcdddc53cbcd8L, 0xd03b464005a93bccL, 0x8212edc1b1669dcbL, 0x71f4c31364c31bc7L, + 0xfeeec0eba8772307L, 0x1948d00a13d88cf1L, 0x19064fd6d943ada8L, 0x4ec8d31722697bfdL, + 0x596d9a953a516609L, 0xc4cb4bff53507da2L, 0x1d59f3c5be36e4caL, 0xe5b4fc5bf6044c9bL, + 0x1bb74e052232f735L, 0x04e8a0db611ddd5dL, 0x8d04eaa009b421bfL, 0xa7878ae0ac0e6d58L, + 0x28c1030217cab2b3L, 0x827943767e56a883L, 0x28fce5fa02d22809L, 0xb30c322fffc8c58eL, + 0x1ca5a6a9f8066c5bL, 0xb24db5f1462b2513L, 0x02f653b89b7e5f6cL, 0xe31f8fb5d5f78eeeL, + 0x266acc514ed93501L, 0x936879d1c6fddcc4L, 0xcd51be3636af1952L, 0x3fdbb6fc332c78c8L, + 0x9eb656379fa73094L, 0x056146cc92fa0f96L, 0xed6c4f1836c027c3L, 0x021e0bb5d2113f2aL, + 0x8983e42ec1c626b3L, 0x73ea9bc6513ad9c9L, 0x0c904903b24f4247L, 0xacbac1e6243e2525L, + 0x0b1069a0c230fb06L, 0x77d709fca3fc1ce5L, 0x87ad0f65020947e6L, 0x555302641c53f4e6L, + 0x65ea87871fa9aaeeL, 0x58aaf4ecc1067bb4L, 0x1a66c48cc4c65b3fL, 0xca96aca48b2ea969L, + 0xa68eb70bad14de2bL, 0x5ccdb3d7e00a6f6eL, 0xe178fbfec73fe72fL, 0x2b63d6a16b83e890L, + 0x32fdb7a5330fbae0L, 0x2ab5803c8d1bf32cL, 0xda838388c1527c94L, 0x16a50bdc4de24acbL, + 0xe561301f134c074aL, 0xd7ae63d2816b4db1L, 0x036aabd4df0dd741L, 0xc5e0db8783435b9dL, + 0x9c4386cf0a07f3b2L, 0x6a72ac1aa56a13a1L, 0x299bbdb04bb20a23L, 0x138c1018fda16b81L, + 0x0e354f0b3bda49dfL, 0x9f4c295b23127437L, 0xd133ceb2bd561341L, 0xd8b4bfd5a526ac29L, + 0xcdd0a70ddc1c7bbdL, 0x81dce595bf572225L, 0x1c6f925c05f6efd7L, 0x8ae5097553856ea0L, + 0x3aabeaeef248f60dL, 0xd9005809d19a69e2L, 0x2a3a1a314311cc27L, 0x89bb2dc76b2b624aL, + 0x50a2a95d0412e289L, 0x9def8df564e68581L, 0xf49010a9b2e2ea5cL, 0x8602ae175d9ff3f0L, + 0xbf037e245369a618L, 0x8038164365f6e2b5L, 0xe2e1f6163b4e8d08L, 0x8df9314914f0857eL, + }, + }; +} diff --git a/parquet-column/src/main/java/org/apache/parquet/column/values/dictionary/DictionaryValuesWriter.java b/parquet-column/src/main/java/org/apache/parquet/column/values/dictionary/DictionaryValuesWriter.java index f4ed350e3a..0dcac0f716 100644 --- a/parquet-column/src/main/java/org/apache/parquet/column/values/dictionary/DictionaryValuesWriter.java +++ b/parquet-column/src/main/java/org/apache/parquet/column/values/dictionary/DictionaryValuesWriter.java @@ -82,6 +82,9 @@ public abstract class DictionaryValuesWriter extends ValuesWriter implements Req /* size in items of the dictionary at the end of last dictionary encoded page (in case the current page falls back to PLAIN) */ protected int lastUsedDictionarySize; + /* whether a page was dictionary encoded since the dictionary was reset, so one is due even if empty */ + protected boolean encodedAPage; + /* dictionary encoded values */ protected IntList encodedValues = new IntList(); @@ -113,6 +116,14 @@ protected DictionaryPage dictPage(ValuesWriter dictPageWriter) { return ret; } + /** + * A dictionary page without entries where pages were dictionary encoded with nulls alone, as Arrow + * C++ writes one, since such a page still needs a dictionary to be read. + */ + protected DictionaryPage emptyDictPage() { + return encodedAPage ? new DictionaryPage(BytesInput.empty(), 0, encodingForDictionaryPage) : null; + } + @Override public boolean shouldFallBack() { // if the dictionary reaches the max byte size or the values can not be encoded on 4 bytes anymore. @@ -173,6 +184,7 @@ public BytesInput getBytes() { // remember size of dictionary when we last wrote a page lastUsedDictionarySize = getDictionarySize(); lastUsedDictionaryByteSize = Math.toIntExact(dictionaryByteSize); + encodedAPage = true; return bytes; } catch (IOException e) { throw new ParquetEncodingException("could not encode the values", e); @@ -201,6 +213,7 @@ public void close() { public void resetDictionary() { lastUsedDictionaryByteSize = 0; lastUsedDictionarySize = 0; + encodedAPage = false; dictionaryTooBig = false; dictionaryByteSize = 0; clearDictionaryContent(); @@ -264,7 +277,7 @@ public DictionaryPage toDictPageAndClose() { } return dictPage(dictionaryEncoder); } - return null; + return emptyDictPage(); } @Override @@ -334,7 +347,7 @@ public DictionaryPage toDictPageAndClose() { } return dictPage(dictionaryEncoder); } - return null; + return emptyDictPage(); } } @@ -376,7 +389,7 @@ public DictionaryPage toDictPageAndClose() { } return dictPage(dictionaryEncoder); } - return null; + return emptyDictPage(); } @Override @@ -448,7 +461,7 @@ public DictionaryPage toDictPageAndClose() { } return dictPage(dictionaryEncoder); } - return null; + return emptyDictPage(); } @Override @@ -523,7 +536,7 @@ public DictionaryPage toDictPageAndClose() { } return dictPage(dictionaryEncoder); } - return null; + return emptyDictPage(); } @Override @@ -596,7 +609,7 @@ public DictionaryPage toDictPageAndClose() { } return dictPage(dictionaryEncoder); } - return null; + return emptyDictPage(); } @Override diff --git a/parquet-column/src/main/java/org/apache/parquet/column/values/factory/DefaultValuesWriterFactory.java b/parquet-column/src/main/java/org/apache/parquet/column/values/factory/DefaultValuesWriterFactory.java index 4c03e6b65e..80566ab47e 100644 --- a/parquet-column/src/main/java/org/apache/parquet/column/values/factory/DefaultValuesWriterFactory.java +++ b/parquet-column/src/main/java/org/apache/parquet/column/values/factory/DefaultValuesWriterFactory.java @@ -110,8 +110,12 @@ static ValuesWriter dictWriterWithFallBack( Encoding dataPageEncoding, ValuesWriter writerToFallBackTo) { if (parquetProperties.isDictionaryEnabled(path)) { - return FallbackValuesWriter.of( - dictionaryWriter(path, parquetProperties, dictPageEncoding, dataPageEncoding), writerToFallBackTo); + // Under content defined chunking the first page is cut by content, not size, so it is no + // sample to judge a dictionary on; like Arrow C++, fall back on the dictionary size alone. + return new FallbackValuesWriter<>( + dictionaryWriter(path, parquetProperties, dictPageEncoding, dataPageEncoding), + writerToFallBackTo, + !parquetProperties.isContentDefinedChunkingEnabled()); } else { return writerToFallBackTo; } diff --git a/parquet-column/src/main/java/org/apache/parquet/column/values/fallback/FallbackValuesWriter.java b/parquet-column/src/main/java/org/apache/parquet/column/values/fallback/FallbackValuesWriter.java index 41fe484f37..2b6a3ed2a5 100644 --- a/parquet-column/src/main/java/org/apache/parquet/column/values/fallback/FallbackValuesWriter.java +++ b/parquet-column/src/main/java/org/apache/parquet/column/values/fallback/FallbackValuesWriter.java @@ -62,11 +62,28 @@ public static Ported from Arrow C++ {@code parquet::internal::CalculateMask} and arrow-rs + * {@code CdcChunker::calculate_mask}; the arithmetic has to stay bit-identical to theirs. + */ +public final class RollingHashMask { + + /** A boundary needs this many rolling hash matches, one per gear hash table. */ + private static final int NUM_GEARHASH_TABLES = 8; + + private RollingHashMask() {} + + /** + * Derives the mask, validating the envelope. + * + *

A mask with the top {@code n} bits set matches a uniform hash with probability 1/2^n. The + * target is the average chunk size minus the skipped {@code minChunkSize}, divided by the + * {@value #NUM_GEARHASH_TABLES} matches a boundary needs. + * + * @param minChunkSize the minimum chunk size in bytes + * @param maxChunkSize the maximum chunk size in bytes + * @param normLevel the normalization level + * @return the rolling hash mask + * @throws IllegalArgumentException if the arguments cannot produce a usable mask + */ + public static long calculate(long minChunkSize, long maxChunkSize, int normLevel) { + Preconditions.checkArgument( + minChunkSize >= 0, "Invalid content defined chunking minimum chunk size (negative): %s", minChunkSize); + Preconditions.checkArgument( + maxChunkSize > minChunkSize, + "Invalid content defined chunking size range: maximum chunk size (%s) must be greater than minimum chunk size (%s)", + maxChunkSize, + minChunkSize); + + // Halve before adding, so that the sum cannot overflow; the references do overflow there. + long avgChunkSize = minChunkSize / 2 + maxChunkSize / 2 + (minChunkSize % 2 + maxChunkSize % 2) / 2; + long targetSize = (avgChunkSize - minChunkSize) / NUM_GEARHASH_TABLES; + int targetBits = Long.SIZE - Long.numberOfLeadingZeros(targetSize); + int maskBits = targetBits == 0 ? 0 : targetBits - 1; + int effectiveBits = maskBits - normLevel; + + // Java takes a long's shift count modulo 64, so an out-of-range width must be rejected here. + Preconditions.checkArgument( + effectiveBits >= 1 && effectiveBits <= 63, + "The content defined chunking mask must be between 1 and 63 bits but was %s" + + " (minimum chunk size %s, maximum chunk size %s, normalization level %s)", + effectiveBits, + minChunkSize, + maxChunkSize, + normLevel); + return -1L << (Long.SIZE - effectiveBits); + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/TestCdcOptions.java b/parquet-column/src/test/java/org/apache/parquet/column/TestCdcOptions.java new file mode 100644 index 0000000000..3a8f7775c5 --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/column/TestCdcOptions.java @@ -0,0 +1,72 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatThrownBy; + +import org.junit.jupiter.api.Test; + +public class TestCdcOptions { + + /** Boundaries only match Arrow C++ and arrow-rs while the defaults do. */ + @Test + public void defaultsMatchTheOtherImplementations() { + assertThat(CdcOptions.DEFAULT.getMinChunkSize()).isEqualTo(256 * 1024L); + assertThat(CdcOptions.DEFAULT.getMaxChunkSize()).isEqualTo(1024 * 1024L); + assertThat(CdcOptions.DEFAULT.getNormLevel()).isZero(); + } + + @Test + public void toStringNamesEverySetting() { + assertThat(options(64 * 1024, 256 * 1024, -1)) + .asString() + .contains("65536") + .contains("262144") + .contains("-1"); + } + + @Test + public void rejectsSizesAtTheSetterThatSaysWhichOneIsWrong() { + assertThatThrownBy(() -> CdcOptions.builder().withMinChunkSize(-1)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessage("Invalid content defined chunking minimum chunk size (negative): -1"); + assertThatThrownBy(() -> CdcOptions.builder().withMaxChunkSize(0)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessage("Invalid content defined chunking maximum chunk size (not positive): 0"); + } + + @Test + public void anUnusableSizeEnvelopeIsRejectedWhenTheOptionsAreBuilt() { + assertThatThrownBy(() -> CdcOptions.builder() + .withMinChunkSize(0) + .withMaxChunkSize(16) + .build()) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("between 1 and 63 bits"); + } + + private static CdcOptions options(long min, long max, int normLevel) { + return CdcOptions.builder() + .withMinChunkSize(min) + .withMaxChunkSize(max) + .withNormLevel(normLevel) + .build(); + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/TestParquetProperties.java b/parquet-column/src/test/java/org/apache/parquet/column/TestParquetProperties.java index 6d51f67cb0..12ee4bce1d 100644 --- a/parquet-column/src/test/java/org/apache/parquet/column/TestParquetProperties.java +++ b/parquet-column/src/test/java/org/apache/parquet/column/TestParquetProperties.java @@ -138,4 +138,49 @@ public void copyBuilder_preservesColumnCodecAndLevel() { assertThat(copy.getColumnCompressionLevel(colB)).isNull(); assertThat(copy.getColumnCodec(colC)).isNull(); } + + // ------------------------------------------- content defined chunking + + private static final CdcOptions CHUNKING = CdcOptions.builder() + .withMinChunkSize(4 * 1024) + .withMaxChunkSize(16 * 1024) + .build(); + + @Test + public void contentDefinedChunking_byDefault_isDisabled() { + assertThat(ParquetProperties.builder().build().isContentDefinedChunkingEnabled()) + .isFalse(); + } + + @Test + public void withContentDefinedChunking_nullOptions_throwsNullPointerException() { + assertThatThrownBy(() -> ParquetProperties.builder().withContentDefinedChunking(null)) + .isInstanceOf(NullPointerException.class) + .hasMessage("CdcOptions cannot be null"); + } + + @Test + public void withContentDefinedChunking_options_enablesChunkingAndSurvivesCopy() { + ParquetProperties original = + ParquetProperties.builder().withContentDefinedChunking(CHUNKING).build(); + ParquetProperties copy = ParquetProperties.copy(original).build(); + + assertThat(copy.isContentDefinedChunkingEnabled()).isTrue(); + assertThat(copy.getCdcOptions()).isSameAs(CHUNKING); + } + + @Test + public void copyBuilder_whileChunkingIsDisabled_stillPreservesTheOptions() { + ParquetProperties disabled = ParquetProperties.builder() + .withContentDefinedChunking(CHUNKING) + .withContentDefinedChunkingEnabled(false) + .build(); + assertThat(disabled.isContentDefinedChunkingEnabled()).isFalse(); + + ParquetProperties reEnabled = ParquetProperties.copy(disabled) + .withContentDefinedChunkingEnabled(true) + .build(); + assertThat(reEnabled.isContentDefinedChunkingEnabled()).isTrue(); + assertThat(reEnabled.getCdcOptions()).isSameAs(CHUNKING); + } } diff --git a/parquet-column/src/test/java/org/apache/parquet/column/impl/ChunkingTestSupport.java b/parquet-column/src/test/java/org/apache/parquet/column/impl/ChunkingTestSupport.java new file mode 100644 index 0000000000..6b22f4ccb9 --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/column/impl/ChunkingTestSupport.java @@ -0,0 +1,158 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import java.util.ArrayList; +import java.util.List; +import java.util.Random; +import java.util.function.IntPredicate; +import java.util.function.ObjIntConsumer; +import java.util.stream.Collectors; +import java.util.stream.IntStream; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.column.ColumnWriteStore; +import org.apache.parquet.column.ColumnWriter; +import org.apache.parquet.column.ParquetProperties; +import org.apache.parquet.column.page.DataPage; +import org.apache.parquet.column.page.mem.MemPageStore; +import org.apache.parquet.column.page.mem.MemPageWriter; +import org.apache.parquet.schema.MessageType; + +final class ChunkingTestSupport { + + private ChunkingTestSupport() {} + + static CdcOptions options(long min, long max, int normLevel) { + return CdcOptions.builder() + .withMinChunkSize(min) + .withMaxChunkSize(max) + .withNormLevel(normLevel) + .build(); + } + + /** A reproducible value stream; the same seed is the same data on any JVM. */ + static long[] newLongs(int count, long seed) { + return new Random(seed).longs(count).toArray(); + } + + /** The generator the golden boundary vectors are built from. */ + static long[] lcg(int count) { + long[] values = new long[count]; + long x = 0x243F6A8885A308D3L; + for (int i = 0; i < count; ++i) { + x = 6364136223846793005L * x + 1442695040888963407L; + values[i] = x; + } + return values; + } + + /** {@code original} with {@code inserted} spliced in at {@code at}, the edit under test. */ + static long[] insert(long[] original, int at, long[] inserted) { + long[] result = new long[original.length + inserted.length]; + System.arraycopy(original, 0, result, 0, at); + System.arraycopy(inserted, 0, result, at, inserted.length); + System.arraycopy(original, at, result, at + inserted.length, original.length - at); + return result; + } + + /** Properties with the position-based page limits lifted, so every page boundary is the chunker's. */ + static ParquetProperties unboundedProps(CdcOptions options) { + return ParquetProperties.builder() + .withPageRowCountLimit(Integer.MAX_VALUE) + .withPageSize(64 * 1024 * 1024) + .withDictionaryEncoding(false) + .withContentDefinedChunking(options) + .build(); + } + + /** + * Writes {@code rows} records through real {@link ColumnWriteStore}s, {@code writeRecord} writing + * record {@code i} to the schema's only column, and returns the pages that come out. A new store, + * made with the same {@code props}, takes over at each of {@code rowGroupStarts}, as for a new row + * group. + */ + static List writePages( + MessageType schema, + ParquetProperties props, + int rows, + ObjIntConsumer writeRecord, + int... rowGroupStarts) { + ColumnDescriptor path = schema.getColumns().get(0); + List pages = new ArrayList<>(); + int record = 0; + for (int end : IntStream.concat(IntStream.of(rowGroupStarts), IntStream.of(rows)) + .toArray()) { + MemPageStore pageStore = new MemPageStore(end - record); + ColumnWriteStore store = props.newColumnWriteStore(schema, pageStore); + ColumnWriter writer = store.getColumnWriter(path); + for (; record < end; ++record) { + writeRecord.accept(writer, record); + store.endRecord(); + } + store.flush(); + pages.addAll(((MemPageWriter) pageStore.getPageWriter(path)).getPages()); + } + return pages; + } + + /** + * Value counts per page, as {@code ChunkingColumnWriter} counts them: a boundary closes the page + * before the triplet that triggered it. + */ + static List chunkSizes(int count, IntPredicate offer) { + List sizes = new ArrayList<>(); + int current = 0; + for (int i = 0; i < count; ++i) { + if (offer.test(i) && current > 0) { + sizes.add(current); + current = 0; + } + current++; + } + if (current > 0) { + sizes.add(current); + } + return sizes; + } + + static List valueCounts(List pages) { + return pages.stream().map(DataPage::getValueCount).collect(Collectors.toList()); + } + + /** How many leading elements the two lists agree on. */ + static int sharedPrefix(List before, List after) { + int i = 0; + while (i < before.size() && i < after.size() && before.get(i).equals(after.get(i))) { + i++; + } + return i; + } + + /** How many trailing elements they agree on, beyond the {@code prefix} already counted. */ + static int sharedSuffix(List before, List after, int prefix) { + int i = 0; + while (i < before.size() - prefix + && i < after.size() - prefix + && before.get(before.size() - 1 - i).equals(after.get(after.size() - 1 - i))) { + i++; + } + return i; + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunker.java b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunker.java new file mode 100644 index 0000000000..d309fe7a21 --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunker.java @@ -0,0 +1,171 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import static org.apache.parquet.column.impl.ChunkingTestSupport.chunkSizes; +import static org.apache.parquet.column.impl.ChunkingTestSupport.insert; +import static org.apache.parquet.column.impl.ChunkingTestSupport.lcg; +import static org.apache.parquet.column.impl.ChunkingTestSupport.newLongs; +import static org.apache.parquet.column.impl.ChunkingTestSupport.options; +import static org.apache.parquet.column.impl.ChunkingTestSupport.sharedPrefix; +import static org.apache.parquet.column.impl.ChunkingTestSupport.sharedSuffix; +import static org.apache.parquet.schema.PrimitiveType.PrimitiveTypeName.INT64; +import static org.assertj.core.api.Assertions.assertThat; + +import java.nio.ByteBuffer; +import java.nio.ByteOrder; +import java.util.ArrayList; +import java.util.List; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.io.api.Binary; +import org.apache.parquet.schema.Types; +import org.junit.jupiter.api.Test; + +/** + * Cut behaviour of the chunker itself; {@code TestRollingHashMask} pins the mask. The expected + * chunk sizes are Arrow C++'s, not this implementation's: pyarrow writes the same values with the + * same 0/64-byte envelope to pages of these sizes. + */ +public class TestCdcChunker { + + /** A tiny envelope, so that few values make many chunks. JUnit gives each test a fresh one. */ + private final CdcChunker chunker = chunkerFor(options(0, 64, 0)); + + @Test + public void producesBoundariesWithinTheSizeEnvelope() { + long min = 512; + long max = 4096; + List sizes = chunkSizesOverLongs(newLongs(20_000, 1), min, max); + + // Every chunk but the last is a completed one, of eight-byte values. + assertThat(sizes).hasSizeGreaterThan(1); + assertThat(sizes.subList(0, sizes.size() - 1)) + .allSatisfy(size -> assertThat((long) size * 8).isBetween(min, max)); + } + + @Test + public void anEditPerturbsOnlyTheChunksAroundIt() { + long[] original = newLongs(50_000, 7); + List before = chunkSizesOverLongs(original, 512, 4096); + List after = chunkSizesOverLongs(insert(original, 20_000, newLongs(500, 99)), 512, 4096); + + int prefix = sharedPrefix(before, after); + assertThat(prefix).as("chunks shared before the edit").isPositive(); + assertThat(sharedSuffix(before, after, prefix)) + .as("nearly every chunk after the edit realigns") + .isGreaterThan((before.size() - prefix) * 9 / 10); + } + + @Test + public void aOneByteBinaryHashesTheSameAsAOneByteFixedWidthValue() { + CdcChunker viaBoolean = chunkerFor(options(0, 1024, 0)); + CdcChunker viaBinary = chunkerFor(options(0, 1024, 0)); + Binary one = Binary.fromConstantByteArray(new byte[] {1}); + Binary zero = Binary.fromConstantByteArray(new byte[] {0}); + List fromBoolean = new ArrayList<>(); + List fromBinary = new ArrayList<>(); + // Varying bytes: a constant byte stream drives the gear hash to a fixed point that never matches. + for (long x : lcg(4000)) { + boolean bit = (x & 0x100000000L) != 0; + fromBoolean.add(viaBoolean.offer(bit, 0, 0)); + fromBinary.add(viaBinary.offer(bit ? one : zero, 0, 0)); + } + assertThat(fromBinary).containsExactlyElementsOf(fromBoolean); + assertThat(fromBinary).contains(true); + } + + /** The minimum is a multiple of eight, so both skip the hash up to the same value. */ + @Test + public void anEightByteBinaryHashesTheSameAsALong() { + long[] values = newLongs(20_000, 19); + CdcChunker viaBinary = chunkerFor(options(512, 4096, 0)); + assertThat(chunkSizes( + values.length, + i -> viaBinary.offer( + Binary.fromConstantByteArray(ByteBuffer.allocate(8) + .order(ByteOrder.LITTLE_ENDIAN) + .putLong(values[i]) + .array()), + 0, + 0))) + .containsExactlyElementsOf(chunkSizesOverLongs(values, 512, 4096)); + } + + /** + * {@code MessageColumnIO} writes {@code writeNull(0, 0)} for a required field a record omits, + * where {@code definitionLevel == maxDef}; there is still no value to hash. The nulls are spread + * out because a gear hash forgets old bytes, so leading ones would perturb almost nothing. + */ + @Test + public void aNullOnARequiredColumnContributesNothingToTheHash() { + CdcChunker withNulls = chunkerFor(options(512, 4096, 0)); + CdcChunker withoutNulls = chunkerFor(options(512, 4096, 0)); + List actual = new ArrayList<>(); + List expected = new ArrayList<>(); + long[] values = newLongs(30_000, 17); + for (int i = 0; i < values.length; ++i) { + if (i % 100 == 0) { + actual.add(withNulls.offerNull(0, 0)); + expected.add(false); + } + actual.add(withNulls.offer(values[i], 0, 0)); + expected.add(withoutNulls.offer(values[i], 0, 0)); + } + assertThat(actual).containsExactlyElementsOf(expected); + assertThat(actual).contains(true); + } + + /** The 4-byte dispatch, which the golden vectors do not reach. */ + @Test + public void chunksIntegers() { + assertThat(chunkSizes(120, i -> chunker.offer(i * 0x9E3779B1, 0, 0))) + .containsExactly(9, 9, 13, 16, 2, 9, 12, 11, 15, 11, 10, 3); + } + + /** + * NaNs differing only in payload must hash differently, which {@code Float.floatToIntBits} would + * prevent. The payloads are quiet NaNs, because {@code Float.intBitsToFloat} may quieten a + * signalling one depending on the platform. + */ + @Test + public void floatsHashTheirRawBitsIncludingNanPayloads() { + assertThat(chunkSizes(120, i -> chunker.offer(Float.intBitsToFloat(0x7FC00001 + i), 0, 0))) + .containsExactly(12, 11, 9, 16, 11, 12, 10, 12, 10, 14, 3); + } + + /** As above, for doubles and {@code Double.doubleToLongBits}. */ + @Test + public void doublesHashTheirRawBitsIncludingNanPayloads() { + assertThat(chunkSizes(120, i -> chunker.offer(Double.longBitsToDouble(0x7FF8000000000001L + i), 0, 0))) + .containsExactly(7, 8, 2, 8, 8, 1, 8, 8, 1, 8, 8, 8, 8, 1, 8, 1, 8, 8, 4, 7); + } + + private static List chunkSizesOverLongs(long[] values, long min, long max) { + CdcChunker chunker = chunkerFor(options(min, max, 0)); + return chunkSizes(values.length, i -> chunker.offer(values[i], 0, 0)); + } + + /** A chunker for a flat required column. */ + private static CdcChunker chunkerFor(CdcOptions options) { + return new CdcChunker( + options, + new ColumnDescriptor(new String[] {"v"}, Types.required(INT64).named("v"), 0, 0)); + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunkerGoldenBoundaries.java b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunkerGoldenBoundaries.java new file mode 100644 index 0000000000..84c9bf8abf --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcChunkerGoldenBoundaries.java @@ -0,0 +1,233 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import static org.apache.parquet.column.impl.ChunkingTestSupport.lcg; +import static org.apache.parquet.column.impl.ChunkingTestSupport.options; +import static org.apache.parquet.column.impl.ChunkingTestSupport.unboundedProps; +import static org.apache.parquet.column.impl.ChunkingTestSupport.valueCounts; +import static org.apache.parquet.column.impl.ChunkingTestSupport.writePages; +import static org.assertj.core.api.Assertions.assertThat; + +import java.nio.ByteBuffer; +import java.util.Arrays; +import java.util.List; +import java.util.function.IntFunction; +import org.apache.parquet.bytes.BytesUtils; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.io.api.Binary; +import org.apache.parquet.schema.MessageType; +import org.apache.parquet.schema.MessageTypeParser; +import org.apache.parquet.schema.PrimitiveType.PrimitiveTypeName; +import org.junit.jupiter.api.Test; + +/** + * Pins page boundaries to the ones Arrow C++ produces for the same data, which the property tests + * elsewhere cannot: a self-consistent but different chunker passes those. + * + *

The expected values are the page value counts, as {@code parquet-cli pages} lists them, of the + * files this script writes with pyarrow 21 or later. No change to {@link CdcChunker} or + * {@link GearHashTable} may alter them. + * + *

{@code
+ * import pyarrow as pa, pyarrow.parquet as pq
+ *
+ * x, raw = 0x243F6A8885A308D3, []  # ChunkingTestSupport.lcg
+ * for _ in range(200_000):
+ *     x = (6364136223846793005 * x + 1442695040888963407) % 2**64
+ *     raw.append(x)
+ * signed = lambda v: v - 2**64 if v >= 2**63 else v
+ * cases = {
+ *     "required_int64": pa.array([signed(r) for r in raw], pa.int64()),
+ *     "optional_int64": pa.array([None if r % 8 == 0 else signed(r) for r in raw], pa.int64()),
+ *     "optional_utf8": pa.array([None if r % 8 == 0 else "row-%d" % (r % 1_000_000) for r in raw]),
+ *     "optional_list_int64": pa.array(
+ *         [None if r % 11 == 0 else [signed(r >> 16 * k) for k in range(r % 4)] for r in raw],
+ *         pa.list_(pa.int64())),
+ * }
+ * for name, v in cases.items():
+ *     field = pa.field("v", v.type, nullable=name != "required_int64")
+ *     pq.write_table(
+ *         pa.table({"v": v}, schema=pa.schema([field])), name + ".parquet",
+ *         compression="none", use_dictionary=False, use_compliant_nested_type=True,
+ *         data_page_size=1 << 30, max_rows_per_page=1 << 30,
+ *         use_content_defined_chunking={"min_chunk_size": 64 << 10, "max_chunk_size": 256 << 10})
+ * pq.write_table(
+ *     pa.table({"v": cases["required_int64"][:20_000]}, schema=pa.schema([pa.field("v", pa.int64(), False)])),
+ *     "tight_envelope_int64.parquet", compression="none", use_dictionary=False,
+ *     data_page_size=1 << 30, max_rows_per_page=1 << 30,
+ *     use_content_defined_chunking={"min_chunk_size": 0, "max_chunk_size": 4096, "norm_level": -2})
+ * }
+ */ +public class TestCdcChunkerGoldenBoundaries { + + private static final int ROWS = 200_000; + + private static final CdcOptions OPTIONS = options(64 * 1024, 256 * 1024, 0); + private static final CdcOptions TIGHT_OPTIONS = options(0, 4096, -2); + + // A binary value's eight bytes sit at PAD within PADDED, with filler either side. + private static final int PAD = 3; + private static final int PADDED = PAD + 8 + 5; + + /** No levels are hashed. */ + @Test + public void requiredInt64MatchesArrowCpp() { + assertThat(pageValueCounts("message t { required int64 v; }")) + .containsExactly( + 17054, 17725, 12281, 14801, 17490, 15449, 19582, 14469, 22343, 16007, 10731, 20511, 1557); + } + + /** Definition levels are hashed, and the value only where one is present. */ + @Test + public void optionalInt64MatchesArrowCpp() { + assertThat(pageValueCounts("message t { optional int64 v; }")) + .containsExactly( + 13812, 13460, 11759, 11204, 11876, 20212, 10075, 19599, 17988, 10262, 11450, 15229, 11339, + 11377, 10358); + } + + /** The same level path, with the binary rather than the fixed width dispatch. */ + @Test + public void optionalBinaryMatchesArrowCpp() { + assertThat(pageValueCounts("message t { optional binary v (STRING); }")) + .containsExactly( + 11588, 10905, 10978, 13364, 16800, 10173, 12896, 11080, 16760, 12559, 12333, 9652, 10732, 11297, + 12799, 12445, 3639); + } + + /** + * A {@code required binary} column of each value's eight little-endian bytes must reproduce the + * {@code required int64} vector, including bytes above 0x7F that a sign-extended table index would + * hash wrongly. Every case but the first surrounds the value with filler that must not be hashed. + */ + @Test + public void binaryValuesHashOnlyTheirOwnBytes() { + List expected = pageValueCounts("message t { required int64 v; }"); + long[] values = lcg(ROWS); + ByteBuffer direct = ByteBuffer.allocateDirect(ROWS * PADDED); + for (long value : values) { + direct.put(padded(value)); + } + + assertThat(binaryPageValueCounts(i -> Binary.fromConstantByteArray(BytesUtils.longToBytes(values[i])))) + .as("a whole byte array") + .containsExactlyElementsOf(expected); + assertThat(binaryPageValueCounts(i -> Binary.fromConstantByteArray(padded(values[i]), PAD, 8))) + .as("a slice of a byte array") + .containsExactlyElementsOf(expected); + assertThat(binaryPageValueCounts(i -> Binary.fromConstantByteBuffer(direct, i * PADDED + PAD, 8))) + .as("a window onto a direct buffer") + .containsExactlyElementsOf(expected); + } + + /** + * A selective mask and a small maximum, so the maximum size cut decides most boundaries (the + * 512-value chunks are 4096 bytes). The other vectors rarely reach that cut, so this is the one + * that pins it leaving the run counter alone. + */ + @Test + public void aMaximumSizeDominatedEnvelopeMatchesArrowCpp() { + assertThat(pageValueCounts("message t { required int64 v; }", 20_000, TIGHT_OPTIONS)) + .containsExactly( + 511, 512, 345, 512, 220, 512, 361, 512, 74, 512, 512, 199, 512, 382, 512, 512, 509, 512, 512, + 226, 512, 512, 2, 465, 512, 512, 349, 512, 460, 512, 98, 512, 512, 512, 132, 512, 512, 385, 508, + 512, 199, 512, 512, 455, 512, 415, 393); + } + + /** + * Both levels hashed, and cuts only at record starts. The only vector with a repetition level, + * so the only one that pins the order the levels are hashed in. + */ + @Test + public void optionalListOfInt64MatchesArrowCpp() { + assertThat(nestedPageValueCounts()) + .containsExactly( + 13336, 9928, 13972, 11483, 10503, 11533, 15500, 9599, 14956, 11109, 10846, 14956, 13858, 15861, + 13510, 16720, 11116, 10562, 9536, 11052, 12880, 12392, 10121, 14988, 13928, 10761, 11464); + } + + /** + * Writes {@code optional group v (LIST) { repeated group list { optional int64 element } }}, the + * three-level encoding pyarrow produces for {@code list}: a null list is one slot at + * definition level 0, an empty list one slot at level 1, and each element of a present list a + * slot at level 3, the first of them starting the record. + */ + private static List nestedPageValueCounts() { + MessageType schema = MessageTypeParser.parseMessageType( + "message t { optional group v (LIST) { repeated group list { optional int64 element; } } }"); + long[] values = lcg(ROWS); + return valueCounts(writePages(schema, unboundedProps(OPTIONS), ROWS, (writer, i) -> { + long x = values[i]; + if (Long.remainderUnsigned(x, 11) == 0) { + writer.writeNull(0, 0); // the list itself is null + } else { + int n = (int) Long.remainderUnsigned(x, 4); + if (n == 0) { + writer.writeNull(0, 1); // an empty list + } else { + for (int k = 0; k < n; k++) { + writer.write(x >>> (16 * k), k == 0 ? 0 : 1, 3); + } + } + } + })); + } + + /** + * The page value counts of one column of the generated data, with the position-based limits + * lifted here and on the reference side. + */ + private static List pageValueCounts(String schemaText) { + return pageValueCounts(schemaText, ROWS, OPTIONS); + } + + private static List pageValueCounts(String schemaText, int rows, CdcOptions options) { + MessageType schema = MessageTypeParser.parseMessageType(schemaText); + ColumnDescriptor path = schema.getColumns().get(0); + int maxDef = path.getMaxDefinitionLevel(); + boolean binary = path.getPrimitiveType().getPrimitiveTypeName() == PrimitiveTypeName.BINARY; + long[] values = lcg(rows); + return valueCounts(writePages(schema, unboundedProps(options), rows, (writer, i) -> { + long x = values[i]; + if (maxDef > 0 && Long.remainderUnsigned(x, 8) == 0) { + writer.writeNull(0, maxDef - 1); + } else if (binary) { + writer.write(Binary.fromString("row-" + Long.remainderUnsigned(x, 1_000_000L)), 0, maxDef); + } else { + writer.write(x, 0, maxDef); + } + })); + } + + /** The pages of a {@code required binary} column holding {@code value.apply(i)} in row {@code i}. */ + private static List binaryPageValueCounts(IntFunction value) { + MessageType schema = MessageTypeParser.parseMessageType("message t { required binary v; }"); + return valueCounts( + writePages(schema, unboundedProps(OPTIONS), ROWS, (writer, i) -> writer.write(value.apply(i), 0, 0))); + } + + private static byte[] padded(long value) { + byte[] bytes = new byte[PADDED]; + Arrays.fill(bytes, (byte) 0xA5); + System.arraycopy(BytesUtils.longToBytes(value), 0, bytes, PAD, 8); + return bytes; + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcWrite.java b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcWrite.java new file mode 100644 index 0000000000..85894e6b6f --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestCdcWrite.java @@ -0,0 +1,500 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import static org.apache.parquet.column.impl.ChunkingTestSupport.chunkSizes; +import static org.apache.parquet.column.impl.ChunkingTestSupport.insert; +import static org.apache.parquet.column.impl.ChunkingTestSupport.newLongs; +import static org.apache.parquet.column.impl.ChunkingTestSupport.options; +import static org.apache.parquet.column.impl.ChunkingTestSupport.sharedPrefix; +import static org.apache.parquet.column.impl.ChunkingTestSupport.sharedSuffix; +import static org.apache.parquet.column.impl.ChunkingTestSupport.unboundedProps; +import static org.apache.parquet.column.impl.ChunkingTestSupport.valueCounts; +import static org.apache.parquet.column.impl.ChunkingTestSupport.writePages; +import static org.assertj.core.api.Assertions.assertThat; + +import java.io.IOException; +import java.nio.ByteBuffer; +import java.util.ArrayList; +import java.util.Arrays; +import java.util.HashSet; +import java.util.List; +import java.util.Set; +import java.util.TreeSet; +import java.util.function.ObjIntConsumer; +import java.util.function.ObjLongConsumer; +import java.util.function.UnaryOperator; +import java.util.stream.LongStream; +import java.util.stream.Stream; +import org.apache.parquet.bytes.BytesUtils; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.column.ColumnWriteStore; +import org.apache.parquet.column.ColumnWriter; +import org.apache.parquet.column.ParquetProperties; +import org.apache.parquet.column.page.DataPage; +import org.apache.parquet.column.page.DataPageV1; +import org.apache.parquet.column.page.mem.MemPageStore; +import org.apache.parquet.io.api.Binary; +import org.apache.parquet.schema.MessageType; +import org.apache.parquet.schema.MessageTypeParser; +import org.apache.parquet.schema.PrimitiveType.PrimitiveTypeName; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.Arguments; +import org.junit.jupiter.params.provider.EnumSource; +import org.junit.jupiter.params.provider.MethodSource; + +/** Covers {@link ChunkingColumnWriter} through a real {@link ColumnWriteStore} and the pages it writes. */ +public class TestCdcWrite { + + private static final CdcOptions OPTIONS = options(4 * 1024, 16 * 1024, 0); + + private static final MessageType REQUIRED = MessageTypeParser.parseMessageType("message m { required int64 v; }"); + private static final MessageType REQUIRED_BINARY = + MessageTypeParser.parseMessageType("message m { required binary v; }"); + private static final MessageType LIST = MessageTypeParser.parseMessageType("message m { repeated int64 v; }"); + + /** + * A value wider than the maximum chunk size asks for a boundary before itself even as the first + * value of a page, which {@link ColumnWriterBase#writePage()} would reject as empty. Arrow C++ + * filters out the same empty chunk. + */ + @Test + public void aBoundaryOnTheFirstValueOfAPageDoesNotWriteAnEmptyOne() { + List pages = writePages(REQUIRED_BINARY, unboundedProps(options(0, 32, 0)), 4, (writer, i) -> { + byte[] wide = new byte[64]; + Arrays.fill(wide, (byte) i); + writer.write(Binary.fromConstantByteArray(wide), 0, 0); + }); + + assertThat(valueCounts(pages)).containsExactly(1, 1, 1, 1); + } + + /** {@link ChunkingColumnWriter} hooks each {@code write} overload separately. */ + @ParameterizedTest + @MethodSource("primitiveColumns") + public void everyPrimitiveTypeGoesThroughTheChunker(String label, MessageType schema) { + // Pages must fall where a chunker offered the same typed values puts its boundaries: a hook that + // skips the chunker, cuts after the value, or offers another type moves them. + ColumnDescriptor path = schema.getColumns().get(0); + PrimitiveTypeName type = path.getPrimitiveType().getPrimitiveTypeName(); + long[] values = newLongs(60_000, 13); + CdcChunker chunker = new CdcChunker(OPTIONS, path); + List expected = chunkSizes(values.length, i -> offer(chunker, type, values[i])); + + assertThat(expected).hasSizeGreaterThan(1); + assertThat(valueCounts(writePages( + schema, unboundedProps(OPTIONS), values.length, (writer, i) -> write(writer, type, values[i])))) + .containsExactlyElementsOf(expected); + } + + static Stream primitiveColumns() { + return Stream.of( + Arguments.of("int32", MessageTypeParser.parseMessageType("message m { required int32 v; }")), + Arguments.of("int64", REQUIRED), + Arguments.of("float", MessageTypeParser.parseMessageType("message m { required float v; }")), + Arguments.of("double", MessageTypeParser.parseMessageType("message m { required double v; }")), + Arguments.of("boolean", MessageTypeParser.parseMessageType("message m { required boolean v; }")), + Arguments.of("binary", REQUIRED_BINARY), + Arguments.of( + "fixed_len_byte_array", + MessageTypeParser.parseMessageType("message m { required fixed_len_byte_array(8) v; }"))); + } + + /** Writes {@code value} as the physical type {@code type}. */ + private static void write(ColumnWriter writer, PrimitiveTypeName type, long value) { + switch (type) { + case INT32: + writer.write((int) value, 0, 0); + break; + case INT64: + writer.write(value, 0, 0); + break; + case FLOAT: + writer.write(Float.intBitsToFloat((int) value), 0, 0); + break; + case DOUBLE: + writer.write(Double.longBitsToDouble(value), 0, 0); + break; + case BOOLEAN: + writer.write((value & 1) == 0, 0, 0); + break; + default: + writer.write(Binary.fromConstantByteArray(BytesUtils.longToBytes(value)), 0, 0); + } + } + + private static boolean offer(CdcChunker chunker, PrimitiveTypeName type, long value) { + switch (type) { + case INT32: + return chunker.offer((int) value, 0, 0); + case INT64: + return chunker.offer(value, 0, 0); + case FLOAT: + return chunker.offer(Float.intBitsToFloat((int) value), 0, 0); + case DOUBLE: + return chunker.offer(Double.longBitsToDouble(value), 0, 0); + case BOOLEAN: + return chunker.offer((value & 1) == 0, 0, 0); + default: + return chunker.offer(Binary.fromConstantByteArray(BytesUtils.longToBytes(value)), 0, 0); + } + } + + /** The page limits count from the page start in the writer, so each column must keep one. */ + @Test + public void aStoreHandsOutOneChunkingWriterPerColumn() { + ColumnWriteStore store = unboundedProps(OPTIONS).newColumnWriteStore(REQUIRED, new MemPageStore(1)); + ColumnDescriptor path = REQUIRED.getColumns().get(0); + assertThat(store.getColumnWriter(path)).isSameAs(store.getColumnWriter(path)); + } + + /** + * The page size limit still cuts inside a chunk, but counted from the page start as in Arrow C++, + * so it cuts in the same places after an edit. Narrow values, whose default sized chunks run to + * several pages of the default size, keep deduplicating; checked on the store's row-count schedule + * instead, they shared no page at all. + */ + @Test + public void thePageSizeLimitCutsInTheSamePlacesAfterAnEdit() throws IOException { + ParquetProperties props = ParquetProperties.builder() + .withDictionaryEncoding(false) + .withPageRowCountLimit(Integer.MAX_VALUE) + .withContentDefinedChunking(CdcOptions.DEFAULT) + .build(); + long[] original = newLongs(1_500_000, 21); + List before = pageBytes(narrowPages(props, original)); + // Properties of its own, as for another file: they hold the chunking state. + List after = pageBytes( + narrowPages(ParquetProperties.copy(props).build(), insert(original, 10_000, newLongs(100, 22)))); + + Set beforeSet = new HashSet<>(before); + assertThat(before).hasSizeGreaterThan(5); + assertThat(after.stream().filter(beforeSet::contains).count()) + .as("pages of the edited column that are also in the original") + .isGreaterThan(after.size() / 2); + } + + /** The row count limit still applies inside a chunk, at exactly the limit, as in Arrow C++. */ + @Test + public void theRowCountLimitAppliesInsideAChunk() { + ParquetProperties props = ParquetProperties.builder() + .withDictionaryEncoding(false) + .withContentDefinedChunking(CdcOptions.DEFAULT) + .build(); + assertThat(valueCounts(narrowPages(props, newLongs(200_000, 23)))) + .hasSizeGreaterThan(5) + .allSatisfy( + count -> assertThat(count).isLessThanOrEqualTo(ParquetProperties.DEFAULT_PAGE_ROW_COUNT_LIMIT)) + .contains(ParquetProperties.DEFAULT_PAGE_ROW_COUNT_LIMIT); + } + + /** + * A row group's store continues the chunking where the previous one, made with the same properties, + * left it: row groups add page breaks but move no boundary. Here they start a record before and at + * every chunk boundary, where a chunker that restarts its hash, match run, pending match or size + * misses the boundary or ends the next chunk elsewhere. + */ + @Test + public void eachRowGroupContinuesTheChunkingOfThePrevious() { + long[] values = newLongs(20_000, 29); + // Lists of zero to three values, so that matches inside a record carry over to the next. + ObjIntConsumer record = (writer, i) -> { + int length = (int) (values[i] & 3); + if (length == 0) { + writer.writeNull(0, 0); + } + for (int j = 0; j < length; ++j) { + writer.write(values[i] + j, j == 0 ? 0 : 1, 1); + } + }; + int[] recordStarts = new int[values.length + 1]; + for (int i = 0; i < values.length; ++i) { + recordStarts[i + 1] = recordStarts[i] + Math.max(1, (int) (values[i] & 3)); + } + Set pageEnds = pageEnds(writePages(LIST, unboundedProps(OPTIONS), values.length, record)); + Set rowGroupStarts = new TreeSet<>(); + for (int end : pageEnds) { + int chunkStart = Arrays.binarySearch(recordStarts, end); + if (chunkStart < values.length) { + rowGroupStarts.add(chunkStart - 1); + rowGroupStarts.add(chunkStart); + } + } + assertThat(rowGroupStarts).hasSizeGreaterThan(20); + + Set expected = new TreeSet<>(pageEnds); + rowGroupStarts.forEach(start -> expected.add(recordStarts[start])); + assertThat(pageEnds(writePages( + LIST, + unboundedProps(OPTIONS), + values.length, + record, + rowGroupStarts.stream().mapToInt(Integer::intValue).toArray()))) + .containsExactlyElementsOf(expected); + } + + /** Where each page ends, counted in levels from the first. */ + private static Set pageEnds(List pages) { + Set ends = new TreeSet<>(); + int end = 0; + for (DataPage page : pages) { + end += page.getValueCount(); + ends.add(end); + } + return ends; + } + + /** On a list column the row count limit ends pages at record starts only. */ + @Test + public void theRowCountLimitEndsPagesAtRecordStarts() { + ParquetProperties props = ParquetProperties.builder() + .withDictionaryEncoding(false) + .withPageRowCountLimit(100) + .withContentDefinedChunking(options(1 << 30, 1L << 31, 0)) + .build(); + assertThat(valueCounts(writePages(LIST, props, 1000, (writer, i) -> { + writer.write((long) i, 0, 1); + writer.write((long) i, 1, 1); + writer.write((long) i, 1, 1); + }))) + .containsExactly(300, 300, 300, 300, 300, 300, 300, 300, 300, 300); + } + + /** + * The size limits are checked at the first record start after each batch of 1024 levels, as in + * Arrow C++, so a page value count limit of 5000 ends pages at 5120 values. + */ + @Test + public void theSizeLimitsAreCheckedOncePerBatch() { + ParquetProperties props = ParquetProperties.builder() + .withDictionaryEncoding(false) + .withPageRowCountLimit(Integer.MAX_VALUE) + .withPageValueCountThreshold(5000) + .withContentDefinedChunking(options(1 << 30, 1L << 31, 0)) + .build(); + assertThat(valueCounts(writePages(REQUIRED, props, 4 * 5120, (writer, i) -> writer.write((long) i, 0, 0)))) + .containsExactly(5120, 5120, 5120, 5120); + } + + /** + * The size limit applies inside a chunk too, checked once per batch of 1024 values as in Arrow + * C++: 100-byte values in chunks of up to 8 MiB still come out in pages of about the 1 MiB limit, + * except those that end where a chunk does. + */ + @Test + public void thePageSizeLimitAppliesInsideAChunk() { + ParquetProperties props = ParquetProperties.builder() + .withDictionaryEncoding(false) + .withPageRowCountLimit(Integer.MAX_VALUE) + .withContentDefinedChunking(options(4 << 20, 8 << 20, 0)) + .build(); + byte[] value = new byte[100]; + List pages = writePages( + REQUIRED_BINARY, + props, + 100_000, + (writer, i) -> writer.write(Binary.fromConstantByteArray(value), 0, 0)); + + // No page runs more than one batch past the limit, and inside chunks the limit, not the chunker, + // ends most of them. + int threshold = props.getPageSizeThreshold(); + assertThat(pages).allSatisfy(page -> assertThat(page.getUncompressedSize()) + .isLessThanOrEqualTo(threshold + 1024 * (value.length + 4))); + assertThat(pages) + .filteredOn(page -> page.getUncompressedSize() >= threshold) + .hasSizeGreaterThan(5); + } + + /** One- and two-byte values: the narrow case, whose chunks run to the most pages. */ + private static List narrowPages(ParquetProperties props, long[] values) { + return writePages(REQUIRED_BINARY, props, values.length, (writer, i) -> { + byte[] bytes = BytesUtils.longToBytes(values[i]); + writer.write(Binary.fromConstantByteArray(bytes, 0, (values[i] & 1) == 0 ? 1 : 2), 0, 0); + }); + } + + private static List pageBytes(List pages) throws IOException { + List bytes = new ArrayList<>(); + for (DataPage page : pages) { + bytes.add( + Binary.fromConstantByteArray(((DataPageV1) page).getBytes().toByteArray())); + } + return bytes; + } + + /** Records edited. */ + private static final int EDIT = 50; + + /** A column of each shape, its records written from a seed each. */ + private enum Column { + INT32(200_000, "required int32 v;", (writer, seed) -> writer.write((int) seed, 0, 0)), + OPTIONAL_DOUBLE(100_000, "optional double v;", (writer, seed) -> { + if (seed % 5 == 0) { + writer.writeNull(0, 0); + } else { + writer.write(seed / 3.0, 0, 1); + } + }), + BOOLEAN(900_000, "required boolean v;", (writer, seed) -> writer.write((seed & 1) == 0, 0, 0)), + OPTIONAL_BINARY(75_000, "optional binary v;", (writer, seed) -> { + if (seed % 5 == 0) { + writer.writeNull(0, 0); + } else { + writer.write(Binary.fromString(Long.toString(seed, 36)), 0, 1); + } + }), + FIXED_LEN_BYTE_ARRAY( + 50_000, + "required fixed_len_byte_array(16) v;", + (writer, seed) -> writer.write( + Binary.fromConstantByteArray(ByteBuffer.allocate(16) + .putLong(seed) + .putLong(~seed) + .array()), + 0, + 0)), + LIST( + 60_000, + "optional group l (LIST) { repeated group list { optional int32 element; } }", + (writer, seed) -> writeList(writer, seed, 3)), + LIST_OF_GROUPS( + 60_000, + "optional group l (LIST) { repeated group list { optional group element { optional int32 f0; } } }", + (writer, seed) -> writeList(writer, seed, 4)); + + private final int records; + private final MessageType schema; + private final ObjLongConsumer record; + + Column(int records, String fields, ObjLongConsumer record) { + this.records = records; + this.schema = MessageTypeParser.parseMessageType("message m { " + fields + " }"); + this.record = record; + } + + List pageBytes(long[] seeds) throws IOException { + return TestCdcWrite.pageBytes(writePages( + schema, + unboundedProps(options(1024, 4096, 0)), + seeds.length, + (writer, i) -> record.accept(writer, seeds[i]))); + } + + /** A null or empty list, or one of one to six elements, some of them null at every level. */ + private static void writeList(ColumnWriter writer, long seed, int maxDef) { + int kind = (int) (seed >>> 61); + if (kind < 2) { + writer.writeNull(0, kind); + } + for (int j = 0; j < kind - 1; ++j) { + int def = Math.max(2, maxDef - (int) ((seed >>> (4 * j)) & 3)); + if (def == maxDef) { + writer.write((int) (seed >>> (8 * j)), j == 0 ? 0 : 1, def); + } else { + writer.writeNull(j == 0 ? 0 : 1, def); + } + } + } + } + + private enum Edit { + INSERT(seeds -> insert(seeds, seeds.length / 2, newLongs(EDIT, 41))), + DELETE(seeds -> LongStream.concat( + Arrays.stream(seeds, 0, seeds.length / 2), + Arrays.stream(seeds, seeds.length / 2 + EDIT, seeds.length)) + .toArray()), + UPDATE(seeds -> { + long[] updated = seeds.clone(); + System.arraycopy(newLongs(EDIT, 41), 0, updated, seeds.length / 2, EDIT); + return updated; + }), + PREPEND(seeds -> insert(seeds, 0, newLongs(EDIT, 41))), + APPEND(seeds -> insert(seeds, seeds.length, newLongs(EDIT, 41))); + + private final UnaryOperator apply; + + Edit(UnaryOperator apply) { + this.apply = apply; + } + } + + /** + * An edit of a few records changes only the pages around it, in every shape of column: every page + * before it and all but a few after it are byte for byte as before, so they deduplicate. That a few + * change is the algorithm's, as in Arrow C++: the chunking realigns only once a chunk ends in the + * same place again, as the eight matches that end one count from where it began. + */ + @ParameterizedTest + @EnumSource(Column.class) + public void anEditChangesOnlyThePagesAroundIt(Column column) throws IOException { + long[] original = newLongs(column.records, 37); + List before = column.pageBytes(original); + assertThat(before).hasSizeGreaterThan(350); + for (Edit edit : Edit.values()) { + List after = column.pageBytes(edit.apply.apply(original)); + int prefix = sharedPrefix(before, after); + int suffix = sharedSuffix(before, after, prefix); + assertThat(Math.max(before.size(), after.size()) - prefix - suffix) + .as("pages changed of %s by %s", before.size(), edit) + .isLessThanOrEqualTo(20); + } + } + + @Test + public void onlyChunkingRealignsAfterAnInsertion() throws IOException { + // Both writers share the pages before the insertion; only chunking shares any after it. Page + // bytes rather than value counts, because a position-based writer's later pages keep their + // counts but shift their contents. + long[] original = newLongs(60_000, 11); + long[] edited = insert(original, 25_000, newLongs(300, 42)); + + assertThat(sharedBeyondThePrefix(boundedPageBytes(original, false), boundedPageBytes(edited, false))) + .as("a position-based writer realigns nothing after an insertion") + .isZero(); + + List chunkedBefore = boundedPageBytes(original, true); + assertThat(chunkedBefore).hasSizeGreaterThan(10); + assertThat(sharedBeyondThePrefix(chunkedBefore, boundedPageBytes(edited, true))) + .as("chunking shares pages from after the insertion too") + .isPositive(); + } + + /** How many of {@code before}'s pages past the common leading run reappear anywhere in {@code after}. */ + private static long sharedBeyondThePrefix(List before, List after) { + int prefix = sharedPrefix(before, after); + return before.subList(prefix, before.size()).stream() + .filter(after::contains) + .count(); + } + + /** Every page's bytes, with a page size small enough that the position-based limits make many pages. */ + private static List boundedPageBytes(long[] values, boolean chunking) throws IOException { + ParquetProperties props = ParquetProperties.builder() + .withPageSize(16 * 1024) + .withMinRowCountForPageSizeCheck(1) + .withDictionaryEncoding(false) + .withContentDefinedChunking(OPTIONS) + .withContentDefinedChunkingEnabled(chunking) + .build(); + + return pageBytes(writePages(REQUIRED, props, values.length, (writer, i) -> writer.write(values[i], 0, 0))); + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/impl/TestGearHashTable.java b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestGearHashTable.java new file mode 100644 index 0000000000..9a636b815a --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/column/impl/TestGearHashTable.java @@ -0,0 +1,54 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.column.impl; + +import static org.assertj.core.api.Assertions.assertThat; + +import java.nio.ByteBuffer; +import java.security.MessageDigest; +import java.security.NoSuchAlgorithmException; +import java.util.Arrays; +import org.junit.jupiter.api.Test; + +/** + * Pins {@link GearHashTable} to the specification Arrow C++ and arrow-rs generate their identical + * tables from, so a transcription slip fails here instead of silently changing every boundary. + */ +public class TestGearHashTable { + + /** + * One table per match a boundary needs; entry {@code [seed][n]} is the first eight bytes, + * big-endian, of the MD5 of 64 bytes of {@code seed} followed by 64 bytes of {@code n}. + */ + @Test + public void tableMatchesTheMd5Specification() throws NoSuchAlgorithmException { + MessageDigest md5 = MessageDigest.getInstance("MD5"); + long[][] expected = new long[8][256]; + for (int seed = 0; seed < expected.length; ++seed) { + for (int n = 0; n < 256; ++n) { + byte[] input = new byte[128]; + Arrays.fill(input, 0, 64, (byte) seed); + Arrays.fill(input, 64, 128, (byte) n); + expected[seed][n] = ByteBuffer.wrap(md5.digest(input)).getLong(); + } + } + + assertThat(GearHashTable.TABLE).isDeepEqualTo(expected); + } +} diff --git a/parquet-column/src/test/java/org/apache/parquet/column/values/dictionary/TestDictionary.java b/parquet-column/src/test/java/org/apache/parquet/column/values/dictionary/TestDictionary.java index 13033404ce..787511a6ba 100644 --- a/parquet-column/src/test/java/org/apache/parquet/column/values/dictionary/TestDictionary.java +++ b/parquet-column/src/test/java/org/apache/parquet/column/values/dictionary/TestDictionary.java @@ -30,13 +30,17 @@ import java.io.IOException; import java.nio.ByteBuffer; import java.nio.charset.StandardCharsets; +import java.util.ArrayList; +import java.util.List; import org.apache.parquet.bytes.ByteBufferInputStream; import org.apache.parquet.bytes.BytesInput; import org.apache.parquet.bytes.DirectByteBufferAllocator; import org.apache.parquet.bytes.TrackingByteBufferAllocator; +import org.apache.parquet.column.CdcOptions; import org.apache.parquet.column.ColumnDescriptor; import org.apache.parquet.column.Dictionary; import org.apache.parquet.column.Encoding; +import org.apache.parquet.column.ParquetProperties; import org.apache.parquet.column.page.DictionaryPage; import org.apache.parquet.column.values.ValuesReader; import org.apache.parquet.column.values.ValuesWriter; @@ -52,6 +56,7 @@ import org.apache.parquet.column.values.plain.PlainValuesWriter; import org.apache.parquet.io.api.Binary; import org.apache.parquet.schema.PrimitiveType.PrimitiveTypeName; +import org.apache.parquet.schema.Types; import org.junit.jupiter.api.AfterEach; import org.junit.jupiter.api.BeforeEach; import org.junit.jupiter.api.Test; @@ -210,6 +215,68 @@ public void testBinaryDictionaryFallBack() throws IOException { } } + /** + * Without the first-page judgement, as content defined chunking asks for, a dictionary falls back + * only on its size limit, as Arrow C++ decides it, even after a first page of nulls alone. + */ + @Test + public void fallsBackOnTheSizeLimitAloneWithoutAFirstPageJudgement() throws IOException { + // One page of 100 distinct values is less than its dictionary costs. + assertThat(pageEncodings(true, 1 << 20, 100)).containsExactly(PLAIN, PLAIN, PLAIN); + assertThat(pageEncodings(false, 1 << 20, 100)) + .containsExactly(PLAIN_DICTIONARY, PLAIN_DICTIONARY, PLAIN_DICTIONARY); + // 150 entries of 8 bytes fill the limit during the second page. + assertThat(pageEncodings(false, 150 * 8, 100)).containsExactly(PLAIN_DICTIONARY, PLAIN, PLAIN); + assertThat(pageEncodings(false, 1 << 20, 0)) + .containsExactly(PLAIN_DICTIONARY, PLAIN_DICTIONARY, PLAIN_DICTIONARY); + } + + /** The writer factory judges the first page unless content defined chunking is enabled. */ + @Test + public void theFactoryJudgesTheFirstPageUnlessChunking() { + ColumnDescriptor path = + new ColumnDescriptor(new String[] {"v"}, Types.required(BINARY).named("v"), 0, 0); + assertThat(firstPageEncoding(ParquetProperties.builder().build(), path)).isEqualTo(PLAIN); + assertThat(firstPageEncoding( + ParquetProperties.builder() + .withContentDefinedChunking(CdcOptions.DEFAULT) + .build(), + path)) + .isEqualTo(PLAIN_DICTIONARY); + } + + /** The encoding of a first page of 100 distinct values, less than their dictionary costs. */ + private static Encoding firstPageEncoding(ParquetProperties props, ColumnDescriptor path) { + try (ValuesWriter writer = props.newValuesWriter(path)) { + for (int i = 0; i < 100; i++) { + writer.writeBytes(Binary.fromString(String.format("v%03d", i))); + } + writer.getBytes(); + return writer.getEncoding(); + } + } + + /** Three pages of {@code valuesPerPage} distinct four-character values, 8 bytes of raw data each. */ + private List pageEncodings(boolean judgeFirstPage, int maxDictionaryByteSize, int valuesPerPage) + throws IOException { + List encodings = new ArrayList<>(); + try (FallbackValuesWriter cw = new FallbackValuesWriter<>( + new PlainBinaryDictionaryValuesWriter( + maxDictionaryByteSize, PLAIN_DICTIONARY, PLAIN_DICTIONARY, allocator), + new PlainValuesWriter(100, 500, allocator), + judgeFirstPage)) { + for (int page = 0; page < 3; page++) { + for (int i = 0; i < valuesPerPage; i++) { + cw.writeBytes(Binary.fromString(String.format("v%03d", page * valuesPerPage + i))); + } + cw.getBytes(); + encodings.add(cw.getEncoding()); + cw.reset(); + } + } + return encodings; + } + @Test public void testBinaryDictionaryIntegerOverflow() { Binary mock = Mockito.mock(Binary.class); diff --git a/parquet-column/src/test/java/org/apache/parquet/internal/column/chunking/TestRollingHashMask.java b/parquet-column/src/test/java/org/apache/parquet/internal/column/chunking/TestRollingHashMask.java new file mode 100644 index 0000000000..412ff38a2c --- /dev/null +++ b/parquet-column/src/test/java/org/apache/parquet/internal/column/chunking/TestRollingHashMask.java @@ -0,0 +1,112 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.internal.column.chunking; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatCode; +import static org.assertj.core.api.Assertions.assertThatThrownBy; + +import org.junit.jupiter.api.Test; + +/** + * Ported from Arrow C++'s {@code TestCDC.RollingHashMaskCalculation} and + * {@code TestCDC.ChunkSizeParameterValidation} ({@code cpp/src/parquet/chunker_internal_test.cc}), + * with a few edge cases of its own. + */ +public class TestRollingHashMask { + + private static final long MIN_SIZE = 256 * 1024L; + private static final long MAX_SIZE = 1024 * 1024L; + + @Test + public void maskMatchesTheReferenceForEachNormalizationLevel() { + assertThat(RollingHashMask.calculate(MIN_SIZE, MAX_SIZE, 0)).isEqualTo(0xFFFE000000000000L); + assertThat(RollingHashMask.calculate(MIN_SIZE, MAX_SIZE, 1)).isEqualTo(0xFFFC000000000000L); + assertThat(RollingHashMask.calculate(MIN_SIZE, MAX_SIZE, 2)).isEqualTo(0xFFF8000000000000L); + assertThat(RollingHashMask.calculate(MIN_SIZE, MAX_SIZE, 3)).isEqualTo(0xFFF0000000000000L); + assertThat(RollingHashMask.calculate(MIN_SIZE, MAX_SIZE, -1)).isEqualTo(0xFFFF000000000000L); + } + + @Test + public void maskMatchesTheReferenceAtTheEdgesOfItsRange() { + assertThat(RollingHashMask.calculate(0, 32, 0)).isEqualTo(0x8000000000000000L); + assertThat(RollingHashMask.calculate(0, 64, 0)).isEqualTo(0xC000000000000000L); + assertThat(RollingHashMask.calculate(0, 16, -1)).isEqualTo(0x8000000000000000L); + // A zero target clamps to no bits rather than minus one. + assertThat(RollingHashMask.calculate(0, 8, -1)).isEqualTo(0x8000000000000000L); + assertThat(RollingHashMask.calculate(128, 384, -59)).isEqualTo(0xFFFFFFFFFFFFFFFEL); + } + + @Test + public void rejectsANegativeMinimum() { + assertThatThrownBy(() -> RollingHashMask.calculate(-1, 1024, 0)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessage("Invalid content defined chunking minimum chunk size (negative): -1"); + } + + @Test + public void rejectsARangeThatIsNotAscending() { + assertThatThrownBy(() -> RollingHashMask.calculate(1024, 512, 0)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("must be greater than"); + assertThatThrownBy(() -> RollingHashMask.calculate(32, 32, 0)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("must be greater than"); + } + + @Test + public void rejectsASizeRangeTooNarrowForTheNormalizationLevel() { + // With eight tables the mask needs at least one bit, so the min/max gap must be at least 32 at + // normLevel 0, 64 at 1 and 128 at 2. + assertThatThrownBy(() -> RollingHashMask.calculate(0, 16, 0)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("between 1 and 63 bits"); + assertThatCode(() -> RollingHashMask.calculate(0, 32, 0)).doesNotThrowAnyException(); + assertThatThrownBy(() -> RollingHashMask.calculate(32, 48, 0)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("between 1 and 63 bits"); + assertThatCode(() -> RollingHashMask.calculate(32, 64, 0)).doesNotThrowAnyException(); + + assertThatThrownBy(() -> RollingHashMask.calculate(1, 33, 1)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("between 1 and 63 bits"); + assertThatCode(() -> RollingHashMask.calculate(1, 65, 1)).doesNotThrowAnyException(); + + assertThatThrownBy(() -> RollingHashMask.calculate(0, 123, 2)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("between 1 and 63 bits"); + assertThatCode(() -> RollingHashMask.calculate(0, 128, 2)).doesNotThrowAnyException(); + + assertThatThrownBy(() -> RollingHashMask.calculate(128, 384, -60)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessageContaining("between 1 and 63 bits"); + } + + @Test + public void acceptsALargeEnvelope() { + assertThatCode(() -> RollingHashMask.calculate(1024 * 1024L * 1024L, 2L * 1024 * 1024 * 1024, 0)) + .doesNotThrowAnyException(); + } + + /** A deliberate deviation: Arrow C++ and arrow-rs overflow here. */ + @Test + public void handlesAnEnvelopeThatWouldOverflowTheAverage() { + assertThat(RollingHashMask.calculate(256 * 1024L, Long.MAX_VALUE, 0)).isEqualTo(0xFFFFFFFFFFFFFFC0L); + } +} diff --git a/parquet-hadoop/README.md b/parquet-hadoop/README.md index 13b5132701..0c357edb73 100644 --- a/parquet-hadoop/README.md +++ b/parquet-hadoop/README.md @@ -284,6 +284,42 @@ true if the reader is using a `DirectByteBufferAllocator` --- +**Property:** `parquet.page.content-defined-chunking.enabled` +**Description:** EXPERIMENTAL: Whether data pages also end at boundaries derived from a rolling hash of the +column's values, so that files sharing a run of values share byte-identical pages a content addressable storage +system can deduplicate. The file needs no reader support. +As in Arrow C++, `parquet.page.size` and `parquet.page.row.count.limit` still cut pages inside a chunk, counted from +the page start, so they add pages without moving any after an edit; and a column falls back from dictionary encoding +only when its dictionary outgrows `parquet.dictionary.page.size`, not on its first page. Dictionary ids follow the order +values first appear in, so an edit that adds new values renumbers the later ones within its row group: columns of many +distinct values deduplicate best with `parquet.enable.dictionary` off. +**Default value:** `false` + +--- + +**Property:** `parquet.page.content-defined-chunking.min.size` +**Description:** EXPERIMENTAL: The minimum content defined chunk size in bytes. No chunk is shorter but a file's +last; pages can be, where a row group or a page limit ends one. +**Default value:** `262144` (256 KiB) + +--- + +**Property:** `parquet.page.content-defined-chunking.max.size` +**Description:** EXPERIMENTAL: The maximum content defined chunk size in bytes; a chunk ends here whatever the +rolling hash says. Very small sizes produce very many pages, and encrypted files cannot exceed 32767 pages per column +chunk. +**Default value:** `1048576` (1 MiB) + +--- + +**Property:** `parquet.page.content-defined-chunking.norm.level` +**Description:** EXPERIMENTAL: The normalization level of the rolling hash mask. Raising it makes a boundary more +likely, tightening the chunk size distribution and improving deduplication at the cost of more small pages; lowering +it does the reverse. Values outside `[-3, 3]` are not useful. +**Default value:** `0` + +--- + **Property:** `parquet.page.write-checksum.enabled` **Description:** Whether to write out page level checksums. **Default value:** `true` diff --git a/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetOutputFormat.java b/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetOutputFormat.java index 4db288f455..3ea5d92fad 100644 --- a/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetOutputFormat.java +++ b/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetOutputFormat.java @@ -34,6 +34,7 @@ import org.apache.hadoop.mapreduce.RecordWriter; import org.apache.hadoop.mapreduce.TaskAttemptContext; import org.apache.hadoop.mapreduce.lib.output.FileOutputFormat; +import org.apache.parquet.column.CdcOptions; import org.apache.parquet.column.ParquetProperties; import org.apache.parquet.column.ParquetProperties.WriterVersion; import org.apache.parquet.crypto.FileEncryptionProperties; @@ -164,6 +165,11 @@ public static enum JobSummaryLevel { public static final String STATISTICS_ENABLED = "parquet.column.statistics.enabled"; public static final String SIZE_STATISTICS_ENABLED = "parquet.size.statistics.enabled"; public static final String COLUMN_COMPRESSION_LEVEL_PREFIX = "parquet.compression.level"; + // EXPERIMENTAL: see ParquetProperties.Builder#withContentDefinedChunking + public static final String CONTENT_DEFINED_CHUNKING_ENABLED = "parquet.page.content-defined-chunking.enabled"; + public static final String CONTENT_DEFINED_CHUNKING_MIN_SIZE = "parquet.page.content-defined-chunking.min.size"; + public static final String CONTENT_DEFINED_CHUNKING_MAX_SIZE = "parquet.page.content-defined-chunking.max.size"; + public static final String CONTENT_DEFINED_CHUNKING_NORM_LEVEL = "parquet.page.content-defined-chunking.norm.level"; public static JobSummaryLevel getJobSummaryLevel(Configuration conf) { String level = conf.get(JOB_SUMMARY_LEVEL); @@ -409,6 +415,19 @@ private static int getPageRowCountLimit(Configuration conf) { return conf.getInt(PAGE_ROW_COUNT_LIMIT, ParquetProperties.DEFAULT_PAGE_ROW_COUNT_LIMIT); } + private static boolean getContentDefinedChunkingEnabled(Configuration conf) { + return conf.getBoolean( + CONTENT_DEFINED_CHUNKING_ENABLED, ParquetProperties.DEFAULT_CONTENT_DEFINED_CHUNKING_ENABLED); + } + + private static CdcOptions getCdcOptions(Configuration conf) { + return CdcOptions.builder() + .withMinChunkSize(conf.getLong(CONTENT_DEFINED_CHUNKING_MIN_SIZE, CdcOptions.DEFAULT.getMinChunkSize())) + .withMaxChunkSize(conf.getLong(CONTENT_DEFINED_CHUNKING_MAX_SIZE, CdcOptions.DEFAULT.getMaxChunkSize())) + .withNormLevel(conf.getInt(CONTENT_DEFINED_CHUNKING_NORM_LEVEL, CdcOptions.DEFAULT.getNormLevel())) + .build(); + } + public static void setPageWriteChecksumEnabled(JobContext jobContext, boolean val) { setPageWriteChecksumEnabled(getConfiguration(jobContext), val); } @@ -536,6 +555,10 @@ public RecordWriter getRecordWriter(Configuration conf, Path file, Comp .withPageRowCountLimit(getPageRowCountLimit(conf)) .withPageWriteChecksumEnabled(getPageWriteChecksumEnabled(conf)) .withStatisticsEnabled(getStatisticsEnabled(conf)); + if (getContentDefinedChunkingEnabled(conf)) { + // Only when enabled: building the options validates them, and a disabled job must not fail. + propsBuilder.withContentDefinedChunking(getCdcOptions(conf)); + } new ColumnConfigParser() .withColumnConfig( ENABLE_DICTIONARY, key -> conf.getBoolean(key, false), propsBuilder::withDictionaryEncoding) diff --git a/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetWriter.java b/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetWriter.java index 7dcbb3188c..3fe824fad2 100644 --- a/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetWriter.java +++ b/parquet-hadoop/src/main/java/org/apache/parquet/hadoop/ParquetWriter.java @@ -26,6 +26,7 @@ import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.fs.Path; import org.apache.parquet.bytes.ByteBufferAllocator; +import org.apache.parquet.column.CdcOptions; import org.apache.parquet.column.ParquetProperties; import org.apache.parquet.column.ParquetProperties.WriterVersion; import org.apache.parquet.compression.CompressionCodecFactory; @@ -681,6 +682,30 @@ public SELF withPageRowCountLimit(int rowCount) { return self(); } + /** + * EXPERIMENTAL: Enable or disable content defined chunking of data pages; see + * {@link ParquetProperties.Builder#withContentDefinedChunkingEnabled(boolean)}. + * + * @param enabled whether to derive data page boundaries from the content + * @return this builder for method chaining + */ + public SELF withContentDefinedChunkingEnabled(boolean enabled) { + encodingPropsBuilder.withContentDefinedChunkingEnabled(enabled); + return self(); + } + + /** + * EXPERIMENTAL: Enable content defined chunking with the given options; see + * {@link ParquetProperties.Builder#withContentDefinedChunking(CdcOptions)}. + * + * @param options the chunking options + * @return this builder for method chaining + */ + public SELF withContentDefinedChunking(CdcOptions options) { + encodingPropsBuilder.withContentDefinedChunking(options); + return self(); + } + /** * Set the Parquet format dictionary page size used by the constructed * writer. diff --git a/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestCdcWriter.java b/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestCdcWriter.java new file mode 100644 index 0000000000..a88cddc459 --- /dev/null +++ b/parquet-hadoop/src/test/java/org/apache/parquet/hadoop/TestCdcWriter.java @@ -0,0 +1,478 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ +package org.apache.parquet.hadoop; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatCode; +import static org.assertj.core.api.Assertions.assertThatThrownBy; + +import java.io.IOException; +import java.util.ArrayList; +import java.util.List; +import java.util.Random; +import java.util.Set; +import java.util.TreeSet; +import java.util.function.UnaryOperator; +import java.util.stream.Collectors; +import java.util.stream.Stream; +import org.apache.hadoop.conf.Configuration; +import org.apache.hadoop.fs.Path; +import org.apache.hadoop.mapreduce.RecordWriter; +import org.apache.parquet.column.CdcOptions; +import org.apache.parquet.column.ColumnDescriptor; +import org.apache.parquet.column.Encoding; +import org.apache.parquet.column.ParquetProperties.WriterVersion; +import org.apache.parquet.column.page.DataPage; +import org.apache.parquet.column.page.DataPageV1; +import org.apache.parquet.column.page.DataPageV2; +import org.apache.parquet.column.page.PageReadStore; +import org.apache.parquet.column.page.PageReader; +import org.apache.parquet.example.data.Group; +import org.apache.parquet.example.data.simple.NanoTime; +import org.apache.parquet.example.data.simple.SimpleGroupFactory; +import org.apache.parquet.hadoop.example.ExampleParquetWriter; +import org.apache.parquet.hadoop.example.GroupReadSupport; +import org.apache.parquet.hadoop.example.GroupWriteSupport; +import org.apache.parquet.hadoop.metadata.CompressionCodecName; +import org.apache.parquet.hadoop.util.HadoopInputFile; +import org.apache.parquet.schema.MessageType; +import org.apache.parquet.schema.MessageTypeParser; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.io.TempDir; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.Arguments; +import org.junit.jupiter.params.provider.CsvSource; +import org.junit.jupiter.params.provider.EnumSource; +import org.junit.jupiter.params.provider.MethodSource; + +/** + * Content defined chunking through {@link ParquetWriter.Builder} and {@link ParquetOutputFormat}, into + * a real file that is read back. + */ +public class TestCdcWriter { + + private static final MessageType SCHEMA = + MessageTypeParser.parseMessageType("message t { required int64 id; required binary name (STRING); }"); + + private static final CdcOptions OPTIONS = CdcOptions.builder() + .withMinChunkSize(16 * 1024) + .withMaxChunkSize(64 * 1024) + .build(); + + private static final CdcOptions SMALL_OPTIONS = CdcOptions.builder() + .withMinChunkSize(4 * 1024) + .withMaxChunkSize(16 * 1024) + .build(); + + private static final int ROWS = 40_000; + + @TempDir + private java.nio.file.Path tempDir; + + @ParameterizedTest + @EnumSource(WriterVersion.class) + public void roundTripsEveryValue(WriterVersion version) throws IOException { + Path file = write("roundtrip-" + version, b -> b.withWriterVersion(version)); + assertThat(pageCounts(file)).hasSizeGreaterThan(1); + try (ParquetFileReader reader = ParquetFileReader.open(HadoopInputFile.fromPath(file, new Configuration()))) { + DataPage page = reader.readNextRowGroup() + .getPageReader( + reader.getFileMetaData().getSchema().getColumns().get(1)) + .readPage(); + assertThat(page) + .as("the writer version reaches the pages") + .isInstanceOf(version == WriterVersion.PARQUET_2_0 ? DataPageV2.class : DataPageV1.class); + } + + List read = new ArrayList<>(); + try (ParquetReader reader = ParquetReader.builder(new GroupReadSupport(), file) + .withConf(new Configuration()) + .build()) { + Group g; + while ((g = reader.read()) != null) { + read.add(g.getLong("id", 0) + "|" + g.getString("name", 0)); + } + } + + List expected = new ArrayList<>(ROWS); + for (int i = 0; i < ROWS; i++) { + expected.add(i + "|" + name(i)); + } + assertThat(read).containsExactlyElementsOf(expected); + } + + /** + * Chunking only moves page boundaries: every physical type, null, list and map reads back as it + * does without, across row groups, page limits, dictionary fallback and column chunks of nulls. + */ + @ParameterizedTest + @CsvSource({"PARQUET_1_0, true", "PARQUET_1_0, false", "PARQUET_2_0, true", "PARQUET_2_0, false"}) + public void chunkingLosesNoData(WriterVersion version, boolean dictionary) throws IOException { + MessageType schema = MessageTypeParser.parseMessageType("message t {" + + " required int32 i32; optional int64 i64; optional float f; required double d; optional boolean b;" + + " optional binary s (STRING); optional fixed_len_byte_array(8) x; optional int96 t;" + + " optional group l (LIST) { repeated group list { optional int32 element; } }" + + " optional group m (MAP) { repeated group key_value { required binary key (STRING); optional int64 value; } }" + + " optional int64 sparse;" + + " }"); + SimpleGroupFactory f = new SimpleGroupFactory(schema); + Random random = new Random(31); + List rows = new ArrayList<>(); + for (int i = 0; i < 30_000; i++) { + Group row = f.newGroup().append("i32", random.nextInt(1000)).append("d", random.nextDouble()); + if (random.nextInt(4) > 0) { + row.append("i64", random.nextLong()); + } + if (random.nextInt(8) > 0) { + row.append("f", random.nextFloat()); + } + if (random.nextInt(3) > 0) { + row.append("b", random.nextBoolean()); + } + if (random.nextInt(5) > 0) { + row.append("s", "v" + random.nextInt(random.nextBoolean() ? 300 : Integer.MAX_VALUE)); + } + if (random.nextInt(5) > 0) { + row.append("x", String.format("%08x", random.nextInt())); + } + if (random.nextInt(5) > 0) { + row.append("t", new NanoTime(random.nextInt(3_000_000), random.nextLong())); + } + if (random.nextInt(6) > 0) { + Group list = row.addGroup("l"); + for (int n = random.nextInt(4); n > 0; n--) { + Group element = list.addGroup("list"); + if (random.nextInt(5) > 0) { + element.append("element", random.nextInt()); + } + } + } + if (random.nextInt(6) > 0) { + Group map = row.addGroup("m"); + for (int n = random.nextInt(3); n > 0; n--) { + Group entry = map.addGroup("key_value").append("key", "k" + n); + if (random.nextInt(4) > 0) { + entry.append("value", random.nextLong()); + } + } + } + // Null in the first two row groups, and in the first pages of every later one. + if (i >= 14_000 && i % 7_000 >= 5_000) { + row.append("sparse", random.nextLong()); + } + rows.add(row); + } + List expected = rows.stream().map(Group::toString).collect(Collectors.toList()); + + List files = new ArrayList<>(); + for (boolean chunking : new boolean[] {false, true}) { + Path file = new Path(tempDir.resolve("data-" + version + "-" + dictionary + "-" + chunking + ".parquet") + .toUri()); + try (ParquetWriter writer = ExampleParquetWriter.builder(file) + .withType(schema) + .withWriterVersion(version) + .withDictionaryEncoding(dictionary) + .withRowGroupRowCountLimit(7_000) + .withPageSize(2 * 1024) + .withPageRowCountLimit(2_000) + .withDictionaryPageSize(16 * 1024) + .withContentDefinedChunking(SMALL_OPTIONS) + .withContentDefinedChunkingEnabled(chunking) + .build()) { + for (Group row : rows) { + writer.write(row); + } + } + List read = new ArrayList<>(); + try (ParquetReader reader = + ParquetReader.builder(new GroupReadSupport(), file).build()) { + for (Group row = reader.read(); row != null; row = reader.read()) { + read.add(row.toString()); + } + } + assertThat(read).as(chunking ? "chunked" : "unchunked").containsExactlyElementsOf(expected); + files.add(file); + } + assertThat(pageCounts(files.get(1))) + .as("chunking moves the page boundaries") + .isNotEqualTo(pageCounts(files.get(0))); + } + + /** + * The chunker follows each column across row groups, as arrow-rs's does, so a row group boundary + * adds a page break but moves no chunk boundary. After an edit the chunks realign once, rather than + * again at the start of every later row group, whose position the edit has shifted. The first row + * group ends a row before a chunk boundary, which a chunker restarting its hash, match run or size + * there misses; a match pending across records needs a nested column, as TestCdcWrite has. + */ + @Test + public void chunkingContinuesAcrossRowGroups() throws IOException { + List> whole = pageCountsByRowGroup(write("one-group", UnaryOperator.identity())); + assertThat(whole).hasSize(1); + assertThat(whole.get(0)).hasSizeGreaterThan(4); + int perGroup = whole.get(0).get(0) - 1; + List> groups = pageCountsByRowGroup(write("groups", b -> b.withRowGroupRowCountLimit(perGroup))); + assertThat(groups).hasSizeGreaterThan(4); + + Set expected = new TreeSet<>(pageEnds(whole)); + for (long end = perGroup; end < ROWS; end += perGroup) { + expected.add(end); + } + assertThat(pageEnds(groups)).containsExactlyElementsOf(expected); + } + + /** Where each page ends, counted in values from the start of the file. */ + private static List pageEnds(List> groups) { + List ends = new ArrayList<>(); + long end = 0; + for (List group : groups) { + for (int count : group) { + end += count; + ends.add(end); + } + } + return ends; + } + + /** Through {@link ParquetOutputFormat}, the only reader of the configuration keys. */ + @Test + public void anInvalidEnvelopeIsIgnoredWhileChunkingIsOff() throws Exception { + Configuration conf = chunkingConf(); + conf.setLong(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_MIN_SIZE, 8 * 1024 * 1024); // > the default max + conf.setBoolean(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_ENABLED, false); + + Path off = new Path(tempDir.resolve("stale-config.parquet").toUri()); + assertThatCode(() -> new ParquetOutputFormat() + .getRecordWriter(conf, off, CompressionCodecName.UNCOMPRESSED) + .close(null)) + .as("a stale key must not fail a job that has the feature switched off") + .doesNotThrowAnyException(); + + conf.setBoolean(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_ENABLED, true); + Path on = new Path(tempDir.resolve("stale-config-on.parquet").toUri()); + assertThatThrownBy(() -> + new ParquetOutputFormat().getRecordWriter(conf, on, CompressionCodecName.UNCOMPRESSED)) + .isInstanceOf(IllegalArgumentException.class) + .hasMessage("Invalid content defined chunking size range: maximum chunk size (1048576) must be greater " + + "than minimum chunk size (8388608)"); + } + + /** + * Column chunks of nulls alone, of every dictionary encoded type, keep the dictionary with an empty + * dictionary page, as in Arrow C++, and so does one that starts with pages of them. The file reads + * back whatever the writer version and codec. + */ + @ParameterizedTest + @MethodSource("versionsAndCodecs") + public void nullsReadBackWithTheDictionaryOn(WriterVersion version, CompressionCodecName codec) throws IOException { + MessageType schema = MessageTypeParser.parseMessageType( + "message t { required int64 id; optional int32 late; optional binary b; optional int32 i;" + + " optional int64 l; optional float f; optional double d; optional fixed_len_byte_array(4) x; }"); + Path file = new Path( + tempDir.resolve("nulls-" + version + "-" + codec + ".parquet").toUri()); + try (ParquetWriter writer = ExampleParquetWriter.builder(file) + .withType(schema) + .withWriterVersion(version) + .withCompressionCodec(codec) + .withRowGroupRowCountLimit(ROWS / 4) + .withContentDefinedChunking(SMALL_OPTIONS) + .build()) { + SimpleGroupFactory f = new SimpleGroupFactory(schema); + for (int i = 0; i < ROWS; i++) { + Group row = f.newGroup().append("id", (long) i); + if (i >= ROWS * 5 / 8) { + row.append("late", i % 100); + } + writer.write(row); + } + } + + try (ParquetReader reader = + ParquetReader.builder(new GroupReadSupport(), file).build()) { + for (int i = 0; i < ROWS; i++) { + Group row = reader.read(); + assertThat(row.getLong("id", 0)).isEqualTo(i); + for (String empty : new String[] {"b", "i", "l", "f", "d", "x"}) { + assertThat(row.getFieldRepetitionCount(empty)).as(empty).isZero(); + } + if (i >= ROWS * 5 / 8) { + assertThat(row.getInteger("late", 0)).isEqualTo(i % 100); + } else { + assertThat(row.getFieldRepetitionCount("late")).isZero(); + } + } + assertThat(reader.read()).isNull(); + } + } + + static Stream versionsAndCodecs() { + return Stream.of(WriterVersion.values()).flatMap(version -> Stream.of( + CompressionCodecName.UNCOMPRESSED, + CompressionCodecName.SNAPPY, + CompressionCodecName.GZIP, + CompressionCodecName.ZSTD, + CompressionCodecName.LZ4_RAW) + .map(codec -> Arguments.of(version, codec))); + } + + @Test + public void chunkingCanBeSwitchedOffAfterItsOptionsAreSet() throws IOException { + // With the position-based limits lifted, an unchunked column chunk is one page. + assertThat(pageCounts(write("switched-off", b -> b.withContentDefinedChunkingEnabled(false)))) + .hasSize(1); + } + + @Test + public void theConfigurationKeysActuallyChunk() throws Exception { + Configuration conf = chunkingConf(); + conf.setBoolean(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_ENABLED, true); + conf.setLong(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_MIN_SIZE, 16 * 1024); + conf.setLong(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_MAX_SIZE, 64 * 1024); + conf.setInt(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_NORM_LEVEL, 1); + CdcOptions configured = CdcOptions.builder() + .withMinChunkSize(16 * 1024) + .withMaxChunkSize(64 * 1024) + .withNormLevel(1) + .build(); + List configuredPages = pageCounts(writeThroughOutputFormat(conf, "configured")); + assertThat(configuredPages).hasSizeGreaterThan(1); + assertThat(configuredPages) + .as("every key reaches the writer") + .containsExactlyElementsOf(pageCounts(write("direct", b -> b.withContentDefinedChunking(configured)))); + + conf.unset(ParquetOutputFormat.CONTENT_DEFINED_CHUNKING_ENABLED); + assertThat(pageCounts(writeThroughOutputFormat(conf, "unconfigured"))) + .as("and do nothing while the enabled key is unset") + .hasSize(1); + } + + private Configuration chunkingConf() { + Configuration conf = new Configuration(); + GroupWriteSupport.setSchema(SCHEMA, conf); + conf.set(ParquetOutputFormat.WRITE_SUPPORT_CLASS, GroupWriteSupport.class.getName()); + conf.setInt(ParquetOutputFormat.PAGE_ROW_COUNT_LIMIT, Integer.MAX_VALUE); + conf.setInt(ParquetOutputFormat.PAGE_SIZE, 1 << 30); + conf.setBoolean(ParquetOutputFormat.ENABLE_DICTIONARY, false); + return conf; + } + + private Path writeThroughOutputFormat(Configuration conf, String label) throws Exception { + Path file = new Path(tempDir.resolve(label + ".parquet").toUri()); + RecordWriter writer = + new ParquetOutputFormat().getRecordWriter(conf, file, CompressionCodecName.UNCOMPRESSED); + SimpleGroupFactory f = new SimpleGroupFactory(SCHEMA); + for (int i = 0; i < ROWS; i++) { + writer.write(null, f.newGroup().append("id", (long) i).append("name", name(i))); + } + writer.close(null); + return file; + } + + /** + * A chunked first page is cut by content, not size, so it is no sample to judge a dictionary on: + * as in Arrow C++, the dictionary falls back only when it outgrows its size limit. + */ + @Test + public void theDictionaryFallsBackOnItsSizeLimitAlone() throws IOException { + assertThat(encodingsOf(writeHighCardinality("repeating", 12_000))) + .as("a dictionary within its limit is kept") + .contains(Encoding.PLAIN_DICTIONARY) + .doesNotContain(Encoding.PLAIN); + assertThat(encodingsOf(writeHighCardinality("unique", Integer.MAX_VALUE))) + .as("one that outgrows it is dropped") + .contains(Encoding.PLAIN); + } + + private static Set encodingsOf(Path file) throws IOException { + try (ParquetFileReader reader = ParquetFileReader.open(HadoopInputFile.fromPath(file, new Configuration()))) { + return reader.getFooter().getBlocks().get(0).getColumns().get(1).getEncodings(); + } + } + + /** 60 000 chunked rows of ~45-byte values drawn from {@code distinct} different ones. */ + private Path writeHighCardinality(String label, int distinct) throws IOException { + Path file = new Path(tempDir.resolve(label + ".parquet").toUri()); + try (ParquetWriter writer = ExampleParquetWriter.builder(file) + .withType(SCHEMA) + .withCompressionCodec(CompressionCodecName.UNCOMPRESSED) + .withRowGroupSize(1L << 30) + .withContentDefinedChunking(SMALL_OPTIONS) + .build()) { + SimpleGroupFactory f = new SimpleGroupFactory(SCHEMA); + Random random = new Random(7); + String pad = "xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx"; + for (int i = 0; i < 60_000; i++) { + writer.write(f.newGroup().append("id", (long) i).append("name", "v" + random.nextInt(distinct) + pad)); + } + } + return file; + } + + private static String name(int i) { + return "row-" + (i * 2654435761L % 1_000_000L); + } + + /** + * Writes {@code ROWS} rows with chunking on and the position-based page limits lifted, so every + * page boundary is the chunker's. + */ + private Path write(String label, UnaryOperator configure) throws IOException { + Path file = new Path(tempDir.resolve(label + ".parquet").toUri()); + ExampleParquetWriter.Builder builder = ExampleParquetWriter.builder(file) + .withType(SCHEMA) + .withCompressionCodec(CompressionCodecName.UNCOMPRESSED) + .withDictionaryEncoding(false) + .withRowGroupSize(1L << 30) + .withPageSize(1 << 30) + .withPageRowCountLimit(Integer.MAX_VALUE) + .withContentDefinedChunking(OPTIONS); + try (ParquetWriter writer = configure.apply(builder).build()) { + SimpleGroupFactory f = new SimpleGroupFactory(SCHEMA); + for (int i = 0; i < ROWS; i++) { + writer.write(f.newGroup().append("id", (long) i).append("name", name(i))); + } + } + return file; + } + + /** The page value counts of each row group, kept apart rather than concatenated. */ + private static List> pageCountsByRowGroup(Path file) throws IOException { + List> groups = new ArrayList<>(); + try (ParquetFileReader reader = ParquetFileReader.open(HadoopInputFile.fromPath(file, new Configuration()))) { + ColumnDescriptor column = + reader.getFileMetaData().getSchema().getColumns().get(1); + PageReadStore rowGroup; + while ((rowGroup = reader.readNextRowGroup()) != null) { + List counts = new ArrayList<>(); + PageReader pages = rowGroup.getPageReader(column); + for (long read = 0; read < pages.getTotalValueCount(); ) { + DataPage page = pages.readPage(); + counts.add(page.getValueCount()); + read += page.getValueCount(); + } + groups.add(counts); + } + } + return groups; + } + + private static List pageCounts(Path file) throws IOException { + return pageCountsByRowGroup(file).stream().flatMap(List::stream).collect(Collectors.toList()); + } +}