Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- Read and write `vortex.onpair` (unstable edition `unstable2026.06.0`): `tpch_orders.regular` and `clickbench_hits_5k.regular` from the v0.86.1 fixtures now scan, and `WriteOptions.withEdition(Editions.UNSTABLE_2026_06_0)` lets the cascade pick OnPair for string columns ([#425](https://github.com/dfa1/vortex-java/issues/425)).

### Changed
- The writer run-length encodes float columns too (`fastlanes.rle`, as Rust's float RLE scheme does), losslessly: `-0.0` and NaN payloads round-trip ([#438](https://github.com/dfa1/vortex-java/pull/438)).
- Cascading writes (`WriteOptions.cascading(n)`) compress run-end columns further: the run ends and values now go through the cascade (bit-packing, frame-of-reference, …) instead of being stored raw, as in Rust; about 1% smaller on the mixed-column size benchmark ([#435](https://github.com/dfa1/vortex-java/pull/435)).

### Fixed
Expand Down
2 changes: 1 addition & 1 deletion docs/compatibility.md
Original file line number Diff line number Diff line change
Expand Up @@ -119,7 +119,7 @@ decimals ([#430](https://github.com/dfa1/vortex-java/pull/430)) and nulls in nul
| `fastlanes.bitpacked` | `BitpackedEncodingDecoder` | `BitpackedEncodingEncoder` | ✅ | ✅ | Unsigned integer PTypes |
| `fastlanes.delta` | `DeltaEncodingDecoder` | `DeltaEncodingEncoder` | ✅ | ✅ | Integer PTypes |
| `fastlanes.for` | `FrameOfReferenceEncodingDecoder`| `FrameOfReferenceEncodingEncoder`| ✅ | ✅ | Integer PTypes |
| `fastlanes.rle` | `RleEncodingDecoder` | `RleEncodingEncoder` | ✅ | ✅ | Chunk-based RLE |
| `fastlanes.rle` | `RleEncodingDecoder` | `RleEncodingEncoder` | ✅ | ✅ | Chunk-based RLE. Integers and floats (Rust's int and float RLE schemes); float runs compare raw bits, so -0.0 and NaN payloads round-trip. Cascades values/indices/offsets |
| `vortex.patched` | `PatchedEncodingDecoder` | `PatchedEncodingEncoder` | ✅ | ✅ | Primitive PTypes; base + chunked patches (1024-elem blocks) |
| `vortex.variant` | `VariantEncodingDecoder` | `VariantEncodingEncoder` | ✅ | ✅ | Canonical container; constant / chunked-of-constants core + optional shredded child. Typed-scalar values only — nested objects need `parquet.variant` (ADR 0014) |
| `vortex.onpair` | `OnPairEncodingDecoder` | `OnPairEncodingEncoder` | ✅ | ✅ | Utf8, Binary; unstable edition, cascade candidate only when `UNSTABLE_2026_06_0` is enabled |
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -1315,6 +1315,57 @@ void javaWriter_rustReader_runEnd_i64(@TempDir Path tmp) throws IOException {
assertThat(decoded).containsExactly(data);
}

/// Float RLE (Rust's FloatRLEScheme): the writer used to refuse floats outright. Rust must read
/// the raw-bit runs back losslessly, -0.0 and a NaN payload included.
@Test
void javaWriter_rustReader_rle_f64(@TempDir Path tmp) throws IOException {
// Given
Path file = tmp.resolve("java_rle_f64.vtx");
double nanPayload = Double.longBitsToDouble(0x7ff8_0000_0000_0042L);
double[] data = new double[3_000];
for (int i = 0; i < data.length; i++) {
data[i] = switch (i / 500) {
case 0 -> 1.25;
case 1 -> -0.0;
case 2 -> 0.0;
case 3 -> nanPayload;
default -> -7.5;
};
}
try (var ch = FileChannel.open(file, StandardOpenOption.CREATE, StandardOpenOption.WRITE);
var sut = VortexWriter.create(ch, F64_SCHEMA, WriteOptions.defaults(),
List.of(new RleEncodingEncoder()))) {
// When
sut.writeChunk(Map.of(ColumnName.of("v"), data));
}

// Then
double[] decoded = readDoubleColumn(file, "v");
assertThat(Arrays.stream(decoded).mapToLong(Double::doubleToRawLongBits).toArray())
.containsExactly(Arrays.stream(data).mapToLong(Double::doubleToRawLongBits).toArray());
}

/// Float RLE through the cascade: its values child is a double[] the compressor may hand to
/// ALP; whatever wins, vortex-jni must read the column back exactly.
@Test
void javaWriter_jniReader_rle_f64_cascading(@TempDir Path tmp) throws IOException {
// Given — long runs of a few prices: RLE territory
Path file = tmp.resolve("java_rle_f64_cascade.vtx");
double[] prices = {19.99, 24.5, 7.25};
double[] data = new double[20_000];
for (int i = 0; i < data.length; i++) {
data[i] = prices[(i / 700) % prices.length];
}
try (var ch = FileChannel.open(file, StandardOpenOption.CREATE, StandardOpenOption.WRITE);
var sut = VortexWriter.create(ch, F64_SCHEMA, WriteOptions.cascading(3))) {
// When
sut.writeChunk(Map.of(ColumnName.of("v"), data));
}

// Then
assertThat(readDoubleColumn(file, "v")).containsExactly(data);
}

@Test
void javaWriter_rustReader_rle_i32(@TempDir Path tmp) throws IOException {
// Given — FastLanes RLE: chunk-based RLE with offset; exercises chunk boundary proto fields
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -28,7 +28,9 @@ public EncodingId encodingId() {

@Override
public boolean accepts(DType dtype) {
return dtype instanceof DType.Primitive p && !p.ptype().isFloating();
// every primitive, floats included (Rust's FloatRLEScheme): runs compare raw bits, so -0.0,
// +0.0 and NaN payloads stay distinct and the encoding is lossless
return dtype instanceof DType.Primitive;
}

/// Encodes a boolean array as `fastlanes.rle`: the same FastLanes 1024-row chunked
Expand Down Expand Up @@ -137,14 +139,42 @@ public CascadeStep encodeCascade(DType dtype, Object data, EncodeContext ctx) {
new EncodeNode[]{null, null, null}, new int[0]);
long[] values = Arrays.copyOf(runs.values(), runs.valuesCount());
List<ChildSlot> slots = List.of(
new ChildSlot(dtype, PrimitiveArrays.fromLongsArray(values, ptype, EncodingId.FASTLANES_RLE), 0,
VALUES_EXCLUDED),
new ChildSlot(dtype, valuesCarrier(values, ptype), 0, VALUES_EXCLUDED),
new ChildSlot(new DType.Primitive(PType.U16, false), runs.indices(), 1, POSITIONS_EXCLUDED),
new ChildSlot(new DType.Primitive(PType.U64, false), runs.offsets(), 2, POSITIONS_EXCLUDED));
byte[][] stats = ZoneMapStats.of(dtype, data);
return new CascadeStep(partialRoot, List.of(), slots, ZoneMapStats.minOf(stats), ZoneMapStats.maxOf(stats), true);
}

/// Run values back in the column's carrier array: floats from their raw bits (undoing
/// [#toLongs]), integers through [PrimitiveArrays#fromLongsArray(long[], PType, EncodingId)].
private static Object valuesCarrier(long[] values, PType ptype) {
return switch (ptype) {
case F64 -> {
double[] r = new double[values.length];
for (int i = 0; i < r.length; i++) {
r[i] = Double.longBitsToDouble(values[i]);
}
yield r;
}
case F32 -> {
float[] r = new float[values.length];
for (int i = 0; i < r.length; i++) {
r[i] = Float.intBitsToFloat((int) values[i]);
}
yield r;
}
case F16 -> {
short[] r = new short[values.length];
for (int i = 0; i < r.length; i++) {
r[i] = (short) values[i];
}
yield r;
}
default -> PrimitiveArrays.fromLongsArray(values, ptype, EncodingId.FASTLANES_RLE);
};
}

/// One column run-length encoded in FastLanes 1024-row chunks.
///
/// @param values the run values, chunk after chunk (first `valuesCount` are used)
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -437,4 +437,49 @@ void encodeCascade_empty_notApplicable() {
assertThat(result.applicable()).isFalse();
}
}

/// Float RLE (Rust's `FloatRLEScheme`): runs compare raw bits, so the encoding is lossless —
/// `-0.0` and `+0.0` are separate runs and a NaN payload survives, which `==` would get wrong.
@Nested
class Floats {

@Test
void accepts_floatDtypes() {
// When / Then
assertThat(ENCODER.accepts(DType.F64)).isTrue();
assertThat(ENCODER.accepts(DType.F32)).isTrue();
}

@Test
void roundTrip_f64_keepsSignedZeroAndNaNPayload() {
// Given
double nanPayload = Double.longBitsToDouble(0x7ff8_0000_0000_0042L);
double[] data = {1.5, 1.5, -0.0, -0.0, 0.0, nanPayload, nanPayload, 1.5};
EncodeResult encoded = ENCODER.encode(DType.F64, data, EncodeTestHelper.testCtx());
DecodeContext ctx = DecodeTestHelper.toDecodeContext(encoded, data.length, DType.F64, REGISTRY);

// When
Array result = DECODER.decode(ctx);

// Then
for (int i = 0; i < data.length; i++) {
assertThat(Double.doubleToRawLongBits(((io.github.dfa1.vortex.reader.array.DoubleArray) result).getDouble(i)))
.as("index %d", i).isEqualTo(Double.doubleToRawLongBits(data[i]));
}
}

@Test
void encodeCascade_f32_valuesChildIsFloatArray() {
// Given
float[] data = {2.5f, 2.5f, -1f, -1f, -1f};

// When
CascadeStep result = ENCODER.encodeCascade(DType.F32, data, EncodeTestHelper.testCtx());

// Then — a float[] so ALP and friends can compete on the run values
ChildSlot values = result.openChildren().getFirst();
assertThat(values.childDtype()).isEqualTo(DType.F32);
assertThat((float[]) values.childData()).containsExactly(2.5f, -1f);
}
}
}
Loading