Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -37,6 +37,7 @@
import io.github.dfa1.vortex.writer.encode.DecimalBytePartsEncodingEncoder;
import io.github.dfa1.vortex.writer.encode.DecimalEncodingEncoder;
import io.github.dfa1.vortex.writer.encode.DictEncodingEncoder;
import io.github.dfa1.vortex.writer.encode.ZigZagEncodingEncoder;
import io.github.dfa1.vortex.writer.encode.ExtEncodingEncoder;
import io.github.dfa1.vortex.writer.encode.FixedSizeListEncodingEncoder;
import io.github.dfa1.vortex.writer.encode.FrameOfReferenceEncodingEncoder;
Expand Down Expand Up @@ -267,6 +268,10 @@ private static List<EncodingEncoder> buildCascadeCodecs(WriteOptions options) {
// its size (#304). Registered on WriteRegistry already; this adds it as a top-level candidate.
codecs.add(new AlpRdEncodingEncoder());
codecs.add(new FrameOfReferenceEncodingEncoder());
// ZigZag folds signed values to small unsigned magnitudes and cascades them into
// bit-packing, as Rust's ZigZagScheme does (issue #410). It competes with FoR, which
// always fits in as few bits, so it is here for parity rather than for size.
codecs.add(new ZigZagEncodingEncoder());
codecs.add(new RunEndEncodingEncoder());
codecs.add(new RleEncodingEncoder());
codecs.add(new SparseEncodingEncoder());
Expand Down
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
package io.github.dfa1.vortex.writer.encode;

import io.github.dfa1.vortex.core.compute.PrimitiveArrays;
import io.github.dfa1.vortex.core.model.DType;
import io.github.dfa1.vortex.core.model.PType;
import io.github.dfa1.vortex.core.error.VortexException;
Expand All @@ -10,6 +11,7 @@
import java.lang.foreign.MemorySegment;
import java.lang.foreign.ValueLayout;
import java.util.List;
import java.util.Set;

/// Write-only encoder for `vortex.zigzag` — signed integers as zigzag-encoded unsigned values.
public final class ZigZagEncodingEncoder implements EncodingEncoder {
Expand Down Expand Up @@ -122,6 +124,62 @@ public EncodeResult encode(DType dtype, Object data, EncodeContext ctx) {
return new EncodeResult(root, List.of(EncodedBuffer.of(seg, signed)), statsMin, statsMax);
}

/// Barred from the encoded child (issue #410, Rust's `ZigZagScheme` descendant exclusions):
/// zigzag is a bijection that keeps cardinality, runs and value dominance, so if Dict, RunEnd
/// or Sparse lost on the original column they lose on its output too. ZigZag itself would find
/// no negatives in the unsigned output.
private static final Set<EncodingId> ENCODED_EXCLUDED = Set.of(
EncodingId.VORTEX_DICT, EncodingId.VORTEX_RUNEND, EncodingId.VORTEX_SPARSE, EncodingId.VORTEX_ZIGZAG);

/// Cascading zigzag, mirroring Rust's `ZigZagScheme`: the encoded unsigned values become an
/// open child instead of a raw buffer, so the compressor can bit-pack them — zigzag alone keeps
/// the byte width. Not applicable without a negative value: zigzag then only doubles every
/// magnitude.
///
/// On size it never beats frame-of-reference, which competes for the same columns: FoR needs
/// `bits(max - min)`, zigzag `bits(2 * max|v|)`, never fewer (they tie when the data is
/// symmetric around zero). It is kept for parity with the reference compressor.
@Override
public CascadeStep encodeCascade(DType dtype, Object data, EncodeContext ctx) {
PType signed = ((DType.Primitive) dtype).ptype();
long[] values = PrimitiveArrays.toLongs(data, signed, EncodingId.VORTEX_ZIGZAG);
int n = values.length;
// Rust skips on stats before sampling; ours carry no minimum, so find it with a read-only
// scan before allocating anything — the common all-non-negative column stops here
long min = 0L;
for (long v : values) {
min = Math.min(min, v);
}
if (min >= 0L) {
return CascadeStep.notApplicable();
}
long[] encoded = new long[n];
for (int i = 0; i < n; i++) {
long v = values[i];
// computed at 64 bits: for a value sign-extended from a narrower width, the low bits
// of the result equal that width's zigzag, and fromLongsArray truncates to them
encoded[i] = (v << 1) ^ (v >> 63);
}
PType unsigned = toUnsigned(signed);
EncodeNode partialRoot = new EncodeNode(EncodingId.VORTEX_ZIGZAG, null, new EncodeNode[]{null}, new int[0]);
ChildSlot slot = new ChildSlot(new DType.Primitive(unsigned, false),
PrimitiveArrays.fromLongsArray(encoded, unsigned, EncodingId.VORTEX_ZIGZAG), 0, ENCODED_EXCLUDED);
// zigzag is not order-preserving, so the bounds come from the original signed values
byte[][] stats = ZoneMapStats.of(dtype, data);
return new CascadeStep(partialRoot, List.of(), List.of(slot),
ZoneMapStats.minOf(stats), ZoneMapStats.maxOf(stats), true);
}

private static PType toUnsigned(PType signed) {
return switch (signed) {
case I8 -> PType.U8;
case I16 -> PType.U16;
case I32 -> PType.U32;
case I64 -> PType.U64;
default -> throw new VortexException(EncodingId.VORTEX_ZIGZAG, "unsupported ptype: " + signed);
};
}

private static int arrayLength(Object data, PType ptype) {
return switch (ptype) {
case I8 -> ((byte[]) data).length;
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,8 @@
package io.github.dfa1.vortex.writer.encode;

import io.github.dfa1.vortex.core.model.DType;
import io.github.dfa1.vortex.core.model.PType;
import io.github.dfa1.vortex.core.proto.ProtoScalarValue;
import io.github.dfa1.vortex.reader.array.Array;
import io.github.dfa1.vortex.reader.array.ByteArray;
import io.github.dfa1.vortex.reader.array.IntArray;
Expand Down Expand Up @@ -279,4 +281,56 @@ private static io.github.dfa1.vortex.core.proto.ProtoScalarValue scalar(byte[] b
return io.github.dfa1.vortex.core.proto.ProtoScalarValue.decode(seg, 0, seg.byteSize());
}
}

/// Cascading zigzag (issue #410) hands its unsigned output to the compressor as an open child
/// instead of a raw buffer, with Rust's `ZigZagScheme` exclusions on that child.
@Nested
class Cascade {

@Test
void encodeCascade_emitsUnsignedZigZagChild() {
// Given — zigzag maps 0,-1,1,-2,2 to 0,1,2,3,4; I8 extremes map to 254/255
byte[] data = {0, -1, 1, -2, 2, Byte.MAX_VALUE, Byte.MIN_VALUE};
var sut = new ZigZagEncodingEncoder();

// When
CascadeStep result = sut.encodeCascade(new DType.Primitive(PType.I8, false), data,
EncodeTestHelper.testCtx());

// Then
assertThat(result.applicable()).isTrue();
assertThat(result.ownedBuffers()).isEmpty();
ChildSlot child = result.openChildren().getFirst();
assertThat(child.childDtype()).isEqualTo(new DType.Primitive(PType.U8, false));
assertThat((byte[]) child.childData()).containsExactly(0, 1, 2, 3, 4, (byte) 254, (byte) 255);
assertThat(child.excluded()).containsExactlyInAnyOrder(EncodingId.VORTEX_DICT, EncodingId.VORTEX_RUNEND,
EncodingId.VORTEX_SPARSE, EncodingId.VORTEX_ZIGZAG);
}

@Test
void encodeCascade_boundsComeFromSignedValues() {
// Given — zigzag is not order-preserving: -100 encodes larger than 50
long[] data = {50, -100, 7};
var sut = new ZigZagEncodingEncoder();

// When
CascadeStep result = sut.encodeCascade(DTypes.I64, data, EncodeTestHelper.testCtx());

// Then
assertThat(result.statsMin()).isEqualTo(ProtoScalarValue.ofInt64Value(-100L).encode());
assertThat(result.statsMax()).isEqualTo(ProtoScalarValue.ofInt64Value(50L).encode());
}

@Test
void encodeCascade_noNegatives_notApplicable() {
// Given — without a negative value zigzag only doubles every magnitude
long[] data = {0, 3, 9};

// When
CascadeStep result = new ZigZagEncodingEncoder().encodeCascade(DTypes.I64, data, EncodeTestHelper.testCtx());

// Then
assertThat(result.applicable()).isFalse();
}
}
}
Loading