diff --git a/CHANGELOG.md b/CHANGELOG.md index 642e196f..f0f537f2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - Read and write `vortex.onpair` (unstable edition `unstable2026.06.0`): `tpch_orders.regular` and `clickbench_hits_5k.regular` from the v0.86.1 fixtures now scan, and `WriteOptions.withEdition(Editions.UNSTABLE_2026_06_0)` lets the cascade pick OnPair for string columns ([#425](https://github.com/dfa1/vortex-java/issues/425)). ### Changed +- The writer run-length encodes float columns too (`fastlanes.rle`, as Rust's float RLE scheme does), losslessly: `-0.0` and NaN payloads round-trip ([#438](https://github.com/dfa1/vortex-java/pull/438)). - Cascading writes (`WriteOptions.cascading(n)`) compress run-end columns further: the run ends and values now go through the cascade (bit-packing, frame-of-reference, …) instead of being stored raw, as in Rust; about 1% smaller on the mixed-column size benchmark ([#435](https://github.com/dfa1/vortex-java/pull/435)). ### Fixed diff --git a/docs/compatibility.md b/docs/compatibility.md index 371ff89b..d2a95c4b 100644 --- a/docs/compatibility.md +++ b/docs/compatibility.md @@ -119,7 +119,7 @@ decimals ([#430](https://github.com/dfa1/vortex-java/pull/430)) and nulls in nul | `fastlanes.bitpacked` | `BitpackedEncodingDecoder` | `BitpackedEncodingEncoder` | ✅ | ✅ | Unsigned integer PTypes | | `fastlanes.delta` | `DeltaEncodingDecoder` | `DeltaEncodingEncoder` | ✅ | ✅ | Integer PTypes | | `fastlanes.for` | `FrameOfReferenceEncodingDecoder`| `FrameOfReferenceEncodingEncoder`| ✅ | ✅ | Integer PTypes | -| `fastlanes.rle` | `RleEncodingDecoder` | `RleEncodingEncoder` | ✅ | ✅ | Chunk-based RLE | +| `fastlanes.rle` | `RleEncodingDecoder` | `RleEncodingEncoder` | ✅ | ✅ | Chunk-based RLE. Integers and floats (Rust's int and float RLE schemes); float runs compare raw bits, so -0.0 and NaN payloads round-trip. Cascades values/indices/offsets | | `vortex.patched` | `PatchedEncodingDecoder` | `PatchedEncodingEncoder` | ✅ | ✅ | Primitive PTypes; base + chunked patches (1024-elem blocks) | | `vortex.variant` | `VariantEncodingDecoder` | `VariantEncodingEncoder` | ✅ | ✅ | Canonical container; constant / chunked-of-constants core + optional shredded child. Typed-scalar values only — nested objects need `parquet.variant` (ADR 0014) | | `vortex.onpair` | `OnPairEncodingDecoder` | `OnPairEncodingEncoder` | ✅ | ✅ | Utf8, Binary; unstable edition, cascade candidate only when `UNSTABLE_2026_06_0` is enabled | diff --git a/integration/src/test/java/io/github/dfa1/vortex/integration/JavaWritesRustReadsIntegrationTest.java b/integration/src/test/java/io/github/dfa1/vortex/integration/JavaWritesRustReadsIntegrationTest.java index 6a1d8a21..a952ec08 100644 --- a/integration/src/test/java/io/github/dfa1/vortex/integration/JavaWritesRustReadsIntegrationTest.java +++ b/integration/src/test/java/io/github/dfa1/vortex/integration/JavaWritesRustReadsIntegrationTest.java @@ -1315,6 +1315,57 @@ void javaWriter_rustReader_runEnd_i64(@TempDir Path tmp) throws IOException { assertThat(decoded).containsExactly(data); } + /// Float RLE (Rust's FloatRLEScheme): the writer used to refuse floats outright. Rust must read + /// the raw-bit runs back losslessly, -0.0 and a NaN payload included. + @Test + void javaWriter_rustReader_rle_f64(@TempDir Path tmp) throws IOException { + // Given + Path file = tmp.resolve("java_rle_f64.vtx"); + double nanPayload = Double.longBitsToDouble(0x7ff8_0000_0000_0042L); + double[] data = new double[3_000]; + for (int i = 0; i < data.length; i++) { + data[i] = switch (i / 500) { + case 0 -> 1.25; + case 1 -> -0.0; + case 2 -> 0.0; + case 3 -> nanPayload; + default -> -7.5; + }; + } + try (var ch = FileChannel.open(file, StandardOpenOption.CREATE, StandardOpenOption.WRITE); + var sut = VortexWriter.create(ch, F64_SCHEMA, WriteOptions.defaults(), + List.of(new RleEncodingEncoder()))) { + // When + sut.writeChunk(Map.of(ColumnName.of("v"), data)); + } + + // Then + double[] decoded = readDoubleColumn(file, "v"); + assertThat(Arrays.stream(decoded).mapToLong(Double::doubleToRawLongBits).toArray()) + .containsExactly(Arrays.stream(data).mapToLong(Double::doubleToRawLongBits).toArray()); + } + + /// Float RLE through the cascade: its values child is a double[] the compressor may hand to + /// ALP; whatever wins, vortex-jni must read the column back exactly. + @Test + void javaWriter_jniReader_rle_f64_cascading(@TempDir Path tmp) throws IOException { + // Given — long runs of a few prices: RLE territory + Path file = tmp.resolve("java_rle_f64_cascade.vtx"); + double[] prices = {19.99, 24.5, 7.25}; + double[] data = new double[20_000]; + for (int i = 0; i < data.length; i++) { + data[i] = prices[(i / 700) % prices.length]; + } + try (var ch = FileChannel.open(file, StandardOpenOption.CREATE, StandardOpenOption.WRITE); + var sut = VortexWriter.create(ch, F64_SCHEMA, WriteOptions.cascading(3))) { + // When + sut.writeChunk(Map.of(ColumnName.of("v"), data)); + } + + // Then + assertThat(readDoubleColumn(file, "v")).containsExactly(data); + } + @Test void javaWriter_rustReader_rle_i32(@TempDir Path tmp) throws IOException { // Given — FastLanes RLE: chunk-based RLE with offset; exercises chunk boundary proto fields diff --git a/writer/src/main/java/io/github/dfa1/vortex/writer/encode/RleEncodingEncoder.java b/writer/src/main/java/io/github/dfa1/vortex/writer/encode/RleEncodingEncoder.java index b6850a33..71623d84 100644 --- a/writer/src/main/java/io/github/dfa1/vortex/writer/encode/RleEncodingEncoder.java +++ b/writer/src/main/java/io/github/dfa1/vortex/writer/encode/RleEncodingEncoder.java @@ -28,7 +28,9 @@ public EncodingId encodingId() { @Override public boolean accepts(DType dtype) { - return dtype instanceof DType.Primitive p && !p.ptype().isFloating(); + // every primitive, floats included (Rust's FloatRLEScheme): runs compare raw bits, so -0.0, + // +0.0 and NaN payloads stay distinct and the encoding is lossless + return dtype instanceof DType.Primitive; } /// Encodes a boolean array as `fastlanes.rle`: the same FastLanes 1024-row chunked @@ -137,14 +139,42 @@ public CascadeStep encodeCascade(DType dtype, Object data, EncodeContext ctx) { new EncodeNode[]{null, null, null}, new int[0]); long[] values = Arrays.copyOf(runs.values(), runs.valuesCount()); List slots = List.of( - new ChildSlot(dtype, PrimitiveArrays.fromLongsArray(values, ptype, EncodingId.FASTLANES_RLE), 0, - VALUES_EXCLUDED), + new ChildSlot(dtype, valuesCarrier(values, ptype), 0, VALUES_EXCLUDED), new ChildSlot(new DType.Primitive(PType.U16, false), runs.indices(), 1, POSITIONS_EXCLUDED), new ChildSlot(new DType.Primitive(PType.U64, false), runs.offsets(), 2, POSITIONS_EXCLUDED)); byte[][] stats = ZoneMapStats.of(dtype, data); return new CascadeStep(partialRoot, List.of(), slots, ZoneMapStats.minOf(stats), ZoneMapStats.maxOf(stats), true); } + /// Run values back in the column's carrier array: floats from their raw bits (undoing + /// [#toLongs]), integers through [PrimitiveArrays#fromLongsArray(long[], PType, EncodingId)]. + private static Object valuesCarrier(long[] values, PType ptype) { + return switch (ptype) { + case F64 -> { + double[] r = new double[values.length]; + for (int i = 0; i < r.length; i++) { + r[i] = Double.longBitsToDouble(values[i]); + } + yield r; + } + case F32 -> { + float[] r = new float[values.length]; + for (int i = 0; i < r.length; i++) { + r[i] = Float.intBitsToFloat((int) values[i]); + } + yield r; + } + case F16 -> { + short[] r = new short[values.length]; + for (int i = 0; i < r.length; i++) { + r[i] = (short) values[i]; + } + yield r; + } + default -> PrimitiveArrays.fromLongsArray(values, ptype, EncodingId.FASTLANES_RLE); + }; + } + /// One column run-length encoded in FastLanes 1024-row chunks. /// /// @param values the run values, chunk after chunk (first `valuesCount` are used) diff --git a/writer/src/test/java/io/github/dfa1/vortex/writer/encode/RleEncodingEncoderTest.java b/writer/src/test/java/io/github/dfa1/vortex/writer/encode/RleEncodingEncoderTest.java index 4810e211..b8b78d59 100644 --- a/writer/src/test/java/io/github/dfa1/vortex/writer/encode/RleEncodingEncoderTest.java +++ b/writer/src/test/java/io/github/dfa1/vortex/writer/encode/RleEncodingEncoderTest.java @@ -437,4 +437,49 @@ void encodeCascade_empty_notApplicable() { assertThat(result.applicable()).isFalse(); } } + + /// Float RLE (Rust's `FloatRLEScheme`): runs compare raw bits, so the encoding is lossless — + /// `-0.0` and `+0.0` are separate runs and a NaN payload survives, which `==` would get wrong. + @Nested + class Floats { + + @Test + void accepts_floatDtypes() { + // When / Then + assertThat(ENCODER.accepts(DType.F64)).isTrue(); + assertThat(ENCODER.accepts(DType.F32)).isTrue(); + } + + @Test + void roundTrip_f64_keepsSignedZeroAndNaNPayload() { + // Given + double nanPayload = Double.longBitsToDouble(0x7ff8_0000_0000_0042L); + double[] data = {1.5, 1.5, -0.0, -0.0, 0.0, nanPayload, nanPayload, 1.5}; + EncodeResult encoded = ENCODER.encode(DType.F64, data, EncodeTestHelper.testCtx()); + DecodeContext ctx = DecodeTestHelper.toDecodeContext(encoded, data.length, DType.F64, REGISTRY); + + // When + Array result = DECODER.decode(ctx); + + // Then + for (int i = 0; i < data.length; i++) { + assertThat(Double.doubleToRawLongBits(((io.github.dfa1.vortex.reader.array.DoubleArray) result).getDouble(i))) + .as("index %d", i).isEqualTo(Double.doubleToRawLongBits(data[i])); + } + } + + @Test + void encodeCascade_f32_valuesChildIsFloatArray() { + // Given + float[] data = {2.5f, 2.5f, -1f, -1f, -1f}; + + // When + CascadeStep result = ENCODER.encodeCascade(DType.F32, data, EncodeTestHelper.testCtx()); + + // Then — a float[] so ALP and friends can compete on the run values + ChildSlot values = result.openChildren().getFirst(); + assertThat(values.childDtype()).isEqualTo(DType.F32); + assertThat((float[]) values.childData()).containsExactly(2.5f, -1f); + } + } }