diff --git a/src/java/org/apache/cassandra/config/CassandraRelevantProperties.java b/src/java/org/apache/cassandra/config/CassandraRelevantProperties.java
index 7961ec95fb2f..722f84042b28 100644
--- a/src/java/org/apache/cassandra/config/CassandraRelevantProperties.java
+++ b/src/java/org/apache/cassandra/config/CassandraRelevantProperties.java
@@ -637,6 +637,8 @@ public enum CassandraRelevantProperties
TEST_CASSANDRA_SKIP_SYNC("cassandra.skip_sync"),
TEST_CASSANDRA_SUITENAME("suitename", "suitename_IS_UNDEFINED"),
TEST_CASSANDRA_TESTTAG("cassandra.testtag", "cassandra.testtag_IS_UNDEFINED"),
+ /** Generated examples for the ClusteringDescriptorPrefixView property test; the cost is flat in this range. */
+ TEST_CLUSTERING_PREFIX_VIEW_EXAMPLES("cassandra.test.clustering_prefix_view.examples", "1000"),
TEST_COMPRESSION("cassandra.test.compression"),
TEST_COMPRESSION_ALGO("cassandra.test.compression.algo", "lz4"),
TEST_DEBUG_REF_COUNT("cassandra.debugrefcount"),
@@ -650,8 +652,19 @@ public enum CassandraRelevantProperties
TEST_DIFFERENTIAL_BIGVOLUME_ROUNDS("cassandra.test.differential.bigvolume.rounds", "20"),
TEST_DIFFERENTIAL_BIGVOLUME_ROWS_PER_ROUND("cassandra.test.differential.bigvolume.rows_per_round", "100"),
TEST_DIFFERENTIAL_BIGVOLUME_VALUE_PADDING("cassandra.test.differential.bigvolume.value_padding", "200"),
+ /**
+ * Padding-byte width of the block-boundary sweeps in EdgeCaseDifferentialCompactionTest. It must
+ * exceed the per-row serialization overhead, plus one range tombstone marker for the marker sweep.
+ * Both sweeps fail naming this property if it stops being wide enough to bracket the cut.
+ */
+ TEST_DIFFERENTIAL_BLOCK_BOUNDARY_SWEEP("cassandra.test.differential.block_boundary.sweep", "160"),
/** Number of generated examples the randomized differential soak runs; must be > 0. */
TEST_DIFFERENTIAL_EXAMPLES("cassandra.test.differential.examples"),
+ /**
+ * Upper bound of the randomized soak's per-example hub-partition row count; the floor is a quarter
+ * of it. Zero disables hub partitions, which then fails the soak's own promoted-index assertion.
+ */
+ TEST_DIFFERENTIAL_HUB_ROWS_PER_ROUND("cassandra.test.differential.hub_rows_per_round", "120"),
/**
* Preserves a failed differential comparison's captured sstables for post-mortem instead of deleting
* them. Off by default: the burn scenarios' captures are multi-GB and would fill a CI disk.
@@ -663,6 +676,13 @@ public enum CassandraRelevantProperties
TEST_DIFFERENTIAL_LARGEPARTITION_VALUE_PADDING("cassandra.test.differential.largepartition.value_padding", "240"),
/** Seed for the randomized differential soak; defaults to the wall clock, logged per example. */
TEST_DIFFERENTIAL_SEED("cassandra.test.differential.seed"),
+ /**
+ * Reads every row of a captured output back through a routed slice, so the BTI row trie is
+ * exercised as an index rather than only compared as bytes. On by default; skipped in scale mode.
+ */
+ TEST_DIFFERENTIAL_SLICE_READBACK("cassandra.test.differential.slice_readback", "true"),
+ /** Per-partition slice cap for the read-back; the default clears the widest current scenario. */
+ TEST_DIFFERENTIAL_SLICE_READBACK_MAX_ROWS("cassandra.test.differential.slice_readback.max_rows", "5000"),
/** Column counts for the pathological wide-table differential test. */
TEST_DIFFERENTIAL_WIDE_REGULARS("cassandra.test.differential.wide.regulars", "1800"),
TEST_DIFFERENTIAL_WIDE_STATICS("cassandra.test.differential.wide.statics", "200"),
diff --git a/src/java/org/apache/cassandra/db/compaction/CursorCompactor.java b/src/java/org/apache/cassandra/db/compaction/CursorCompactor.java
index d49931d6b398..fda8c5376126 100644
--- a/src/java/org/apache/cassandra/db/compaction/CursorCompactor.java
+++ b/src/java/org/apache/cassandra/db/compaction/CursorCompactor.java
@@ -68,13 +68,13 @@
import org.apache.cassandra.io.sstable.format.SSTableWriter;
import org.apache.cassandra.io.sstable.format.SortedTableWriter;
import org.apache.cassandra.io.sstable.format.Version;
-import org.apache.cassandra.io.sstable.format.big.BigFormat;
import org.apache.cassandra.io.util.DataOutputBuffer;
import org.apache.cassandra.io.util.DataOutputPlus;
import org.apache.cassandra.schema.ColumnMetadata;
import org.apache.cassandra.schema.CompactionParams;
import org.apache.cassandra.schema.SchemaConstants;
import org.apache.cassandra.schema.TableMetadata;
+import org.apache.cassandra.utils.LazyToString;
import org.apache.cassandra.utils.Throwables;
import org.apache.cassandra.utils.TimeUUID;
@@ -112,10 +112,11 @@
* only purgable tombstones in the row cache.
*
Keeps track of the compaction progress.
*
- * This compaction implementation does not support 2ndary indexes, trie (BTI) sstable output,
- * counter columns, or a multi-cell column that the schema has dropped, and it stands aside for a
- * compaction that ignores gc grace for a key; see {@link #isSupported} and
- * {@link #unsupportedMetadata} for the full set of gates.
+ * This compaction implementation writes the BIG and BTI output formats and supports complex
+ * (collection and UDT) columns. It does not support 2ndary indexes, counter columns, or a
+ * multi-cell column that the schema has dropped. It also stands aside for a compaction that
+ * ignores gc grace for a key. See {@link #isSupported} and {@link #unsupportedMetadata} for
+ * the full list of checks.
*
* This compaction implementation avoids garbage creation per partition/row/cell by utilizing reader/writer code
* which supports reusable copies of sstable entry components. The implementation consolidates and duplicates code
@@ -131,10 +132,9 @@ public static boolean isSupported(AbstractCompactionStrategy.ScannerList scanner
if (unsupportedScanners(metadata, scanners))
return false;
- // BTI index writing is not supported yet
- if (!(DatabaseDescriptor.getSelectedSSTableFormat() instanceof BigFormat))
+ if (!DatabaseDescriptor.getSelectedSSTableFormat().supportsCursorCompaction())
{
- LOGGER.debug("Cursor compaction is not supported for {}.{}: only the BIG sstable output format is supported, not {}",
+ LOGGER.debug("Cursor compaction is not supported for {}.{}: the selected sstable output format {} does not support it",
metadata.keyspace, metadata.name, DatabaseDescriptor.getSelectedSSTableFormat());
return false;
}
@@ -328,13 +328,10 @@ private static boolean unsupportedHeaderColumns(TableMetadata metadata, SSTableR
{
if (isDroppedMultiCellOrCounterColumn(metadata, column, reader.header.getType(column)))
{
- LOGGER.atDebug()
- .setMessage("Cursor compaction for table: {} keyspace: {} is not supported. REASON: A multi-cell or counter column dropped from the schema is still carried in the header of {}, which the cursor path does not yet cover. column={}")
- .addArgument(metadata.name)
- .addArgument(metadata.keyspace)
- .addArgument(() -> reader.descriptor)
- .addArgument(() -> column)
- .log();
+ LOGGER.debug("Cursor compaction for table: {} keyspace: {} is not supported. REASON: A multi-cell " +
+ "or counter column dropped from the schema is still carried in the header of {}, which " +
+ "the cursor path does not yet cover. column={}",
+ metadata.name, metadata.keyspace, reader.descriptor, column);
return true;
}
}
@@ -356,6 +353,9 @@ private static boolean isDroppedMultiCellOrCounterColumn(TableMetadata metadata,
private static final Logger LOGGER = LoggerFactory.getLogger(CursorCompactor.class.getName());
+ /** Merged unfiltereds between progress refreshes, as {@link CompactionIterator} uses. */
+ private static final long UNFILTERED_TO_UPDATE_PROGRESS = 128;
+
private final OperationType type;
private final AbstractCompactionController controller;
private final ActiveCompactionsTracker activeCompactions;
@@ -372,8 +372,8 @@ private static boolean isDroppedMultiCellOrCounterColumn(TableMetadata metadata,
/**
* Scratch for {@link #anyMergedCellDeadAtNow}, which walks a row's cells and then puts the
* cursors back. The arrays hold the cursor ORDER and the equals-next flags that its sorts
- * overwrite, and the per-cursor state that tells it which cursors to rewind. All three are
- * null unless the table enforces strict liveness.
+ * overwrite, and the per-cursor state that tells it which cursors to rewind. Only a table
+ * that enforces strict liveness ever reads them.
*/
private final StatefulCursor[] probeCursorOrder;
private final boolean[] probeEqualsNext;
@@ -408,6 +408,8 @@ private static boolean isDroppedMultiCellOrCounterColumn(TableMetadata metadata,
private long totalBytesRead = 0;
private long totalSourceCQLRows;
private long totalDataBytesWritten;
+ /** Merged unfiltereds since the last progress refresh; see {@link #UNFILTERED_TO_UPDATE_PROGRESS}. */
+ private long compactedUnfiltered = 0;
// state
final Purger purger;
@@ -460,25 +462,11 @@ private CursorCompactor(OperationType type,
{
this.controller = controller;
this.type = type;
- // mirror CompactionIterator.purger(): accord-enabled (and accord-migrating) tables
- // purge and expire relative to gcBefore — derived from accord's durability bounds by
- // CompactionTask.getCompactionController — retaining data accord may still read at
- // earlier timestamps; every nowInSec use below is a purge/expiry decision
- TableMetadata tableMetadata = controller.cfs.metadata();
- this.nowInSec = tableMetadata.isAccordEnabled() || tableMetadata.migratingFromAccord()
- ? controller.gcBefore
- : nowInSec;
+ this.nowInSec = purgeTimestamp(controller, nowInSec);
this.compactionId = compactionId;
- long inputBytes = 0;
- long compressedInputBytes = 0;
- for (ISSTableScanner scanner : scanners)
- {
- inputBytes += scanner.getLengthInBytes();
- compressedInputBytes += scanner.getCompressedLengthInBytes();
- }
- this.totalInputBytes = inputBytes;
- this.totalCompressedInputBytes = compressedInputBytes;
+ this.totalInputBytes = sumLength(scanners);
+ this.totalCompressedInputBytes = sumCompressedLength(scanners);
this.partitionMergeCounters = new long[scanners.size()];
this.staticRowMergeCounters = new long[partitionMergeCounters.length];
this.rowMergeCounters = new long[partitionMergeCounters.length];
@@ -492,14 +480,7 @@ private CursorCompactor(OperationType type,
this.activeCompactions.beginCompaction(this); // note that CompactionTask also calls this, but CT only creates CompactionIterator with a NOOP ActiveCompactions
TableMetadata metadata = metadata();
- // the INPUT headers decide whether static rows can occur in this merge (and the output
- // header, SerializationHeader.make, is their union): after ALTER TABLE ... DROP of the
- // last static column, current metadata has no static columns but older sstables
- // legitimately still carry static rows
- boolean anyStaticColumns = false;
- for (SSTableReader sstable : this.sstables)
- anyStaticColumns |= sstable.header.hasStatic();
- this.hasStaticColumns = anyStaticColumns;
+ this.hasStaticColumns = anyStaticColumns(this.sstables);
/**
* Pipeline should end up similar to the one in {@link CompactionIterator}:
* [MERGED -> ?TopPartitionTracker -> GarbageSkipper -> Purger -> org.apache.cassandra.db.transform.DuplicateRowChecker -> Abortable] -> next()
@@ -516,10 +497,10 @@ private CursorCompactor(OperationType type,
this.sstableCursors = convertScannersToCursors(scanners, sstables, DatabaseDescriptor.getCompactionReadDiskAccessMode());
this.sstableCursorsEqualsNext = new boolean[sstables.size()];
this.enforceStrictLiveness = controller.cfs.metadata.get().enforceStrictLiveness();
- this.probeCursorOrder = enforceStrictLiveness ? new StatefulCursor[sstableCursors.length] : null;
- this.probeEqualsNext = enforceStrictLiveness ? new boolean[sstableCursors.length] : null;
- this.probeCursorState = enforceStrictLiveness ? new int[sstableCursors.length] : null;
- this.probeComplexDeletion = enforceStrictLiveness ? DeletionTime.ReusableDeletionTime.live() : null;
+ this.probeCursorOrder = new StatefulCursor[sstableCursors.length];
+ this.probeEqualsNext = new boolean[sstableCursors.length];
+ this.probeCursorState = new int[sstableCursors.length];
+ this.probeComplexDeletion = DeletionTime.ReusableDeletionTime.live();
purger = new Purger(type, controller);
@@ -531,6 +512,48 @@ private CursorCompactor(OperationType type,
assert clusteringParsingAgrees() : "the cursors disagree on how to parse a clustering: " + metadata;
}
+ /**
+ * Mirrors {@link CompactionIterator}'s purger: accord-enabled (and accord-migrating) tables
+ * purge and expire relative to gcBefore — derived from accord's durability bounds by
+ * CompactionTask.getCompactionController — retaining data accord may still read at earlier
+ * timestamps. Every nowInSec use in this class is a purge/expiry decision.
+ */
+ private static long purgeTimestamp(AbstractCompactionController controller, long nowInSec)
+ {
+ TableMetadata metadata = controller.cfs.metadata();
+ return metadata.isAccordEnabled() || metadata.migratingFromAccord() ? controller.gcBefore : nowInSec;
+ }
+
+ private static long sumLength(List scanners)
+ {
+ long bytes = 0;
+ for (ISSTableScanner scanner : scanners)
+ bytes += scanner.getLengthInBytes();
+ return bytes;
+ }
+
+ private static long sumCompressedLength(List scanners)
+ {
+ long bytes = 0;
+ for (ISSTableScanner scanner : scanners)
+ bytes += scanner.getCompressedLengthInBytes();
+ return bytes;
+ }
+
+ /**
+ * The INPUT headers decide whether static rows can occur in this merge, and the output header,
+ * SerializationHeader.make, is their union. After ALTER TABLE ... DROP of the last static
+ * column, current metadata has no static columns but older sstables legitimately still carry
+ * static rows.
+ */
+ private static boolean anyStaticColumns(Iterable sstables)
+ {
+ for (SSTableReader sstable : sstables)
+ if (sstable.header.hasStatic())
+ return true;
+ return false;
+ }
+
/** @see #lastWrittenUnfiltered */
private boolean clusteringParsingAgrees()
{
@@ -676,6 +699,9 @@ else if (UnfilteredSerializer.isTombstoneMarker(flags))
}
// move along
continueReadingAfterMerge(unfilteredMergeLimit, UNFILTERED_END);
+
+ if (++compactedUnfiltered % UNFILTERED_TO_UPDATE_PROGRESS == 0)
+ updateTotalBytesRead();
}
}
@@ -825,20 +851,28 @@ private boolean mergeRows(int rowMergeLimit, DeletionTime partitionActiveDeletio
skipRowsOnStrictLiveness(rowMergeLimit, isStatic);
}
else
- {
- isRowDropped = mergeRowCells(rowMergeLimit, rowActiveDeletion, isRowDropped, isStatic);
- if (!isRowDropped)
- ssTableCursorWriter.writeRowEnd(sstableCursors[0].unfiltered(), isFirstUnfiltered);
- }
- if (isRowDropped && isStatic &&
- isPartitionStarted())
- // if the partition write has not started, keep delaying it, might be an empty partition (purged+no data)
- {
- ssTableCursorWriter.writeEmptyStaticRow();
- }
+ isRowDropped = mergeAndWriteRow(rowMergeLimit, rowActiveDeletion, isRowDropped, isStatic, isFirstUnfiltered);
+
+ maybeWriteEmptyStaticRow(isRowDropped, isStatic);
return !isRowDropped;
}
+ /** @return true if the cell merge dropped the row, in which case nothing was written. */
+ private boolean mergeAndWriteRow(int rowMergeLimit, DeletionTime rowActiveDeletion, boolean isRowDropped, boolean isStatic, boolean isFirstUnfiltered) throws IOException
+ {
+ isRowDropped = mergeRowCells(rowMergeLimit, rowActiveDeletion, isRowDropped, isStatic);
+ if (!isRowDropped)
+ ssTableCursorWriter.writeRowEnd(sstableCursors[0].unfiltered(), isFirstUnfiltered);
+ return isRowDropped;
+ }
+
+ private void maybeWriteEmptyStaticRow(boolean isRowDropped, boolean isStatic) throws IOException
+ {
+ // if the partition write has not started, keep delaying it, might be an empty partition (purged+no data)
+ if (isRowDropped && isStatic && isPartitionStarted())
+ ssTableCursorWriter.writeEmptyStaticRow();
+ }
+
/**
* The merged row's liveness and deletion. One instance, reused for every row.
*/
@@ -1792,6 +1826,12 @@ private void maybeSwitchWriter(CompactionAwareWriter writerProvider)
ssTableCursorWriter = new SSTableCursorWriter((SortedTableWriter) newWriter);
ssTableCursorWriter.setFirst(partitionDescriptor.keyBuffer());
}
+ else
+ {
+ // The switch already opens the finished sstable early; this covers the interval between switches,
+ // where the legacy path gets it from SSTableRewriter.append.
+ writerProvider.maybeReopenEarly(partitionDescriptor.key());
+ }
assert ssTableCursorWriter != null;
}
@@ -2242,9 +2282,13 @@ public void setTargetDirectory(final String targetDirectory)
this.targetDirectory = targetDirectory;
}
+ /**
+ * Counts partitions, not rows, to match {@link CompactionIterator#getMergedRowCounts()}, which feeds
+ * {@code compaction_history.rows_merged}.
+ */
public long[] getMergedRowsCounts()
{
- return rowMergeCounters;
+ return partitionMergeCounters;
}
public long getTotalSourceCQLRows()
@@ -2262,6 +2306,17 @@ private void updateTotalBytesRead(StatefulCursor cursor)
totalBytesRead += cursor.bytesReadSinceSnapshot();
}
+ /**
+ * Refreshes progress from every cursor, so that a large partition moves
+ * {@code nodetool compactionstats} while it is being merged. Matches
+ * {@link CompactionIterator}, which refreshes on the same cadence.
+ */
+ private void updateTotalBytesRead()
+ {
+ for (StatefulCursor cursor : sstableCursors)
+ updateTotalBytesRead(cursor);
+ }
+
public String toString()
{
return this.getCompactionInfo().toString();
@@ -2365,24 +2420,22 @@ public void close()
activeCompactions.finishCompaction(this);
}
- // Every argument is a supplier: the builder is a no-op when INFO is off, so none of these
- // histograms is built or summed unless the line is actually logged.
- LOGGER.atInfo()
- .setMessage("Compaction ended {}: { data bytes read = {}, data bytes written = {}, input (keys = {}, static rows = {}, rows = {}, range tombstones = {}, cells = {}), output (keys = {}, static rows = {}, rows = {}, range tombstones = {}, cells = {})}")
- .addArgument(compactionId)
- .addArgument(this::getTotalBytesScanned)
- .addArgument(() -> totalDataBytesWritten)
- .addArgument(() -> mergeHistogramToString(partitionMergeCounters))
- .addArgument(() -> mergeHistogramToString(staticRowMergeCounters))
- .addArgument(() -> mergeHistogramToString(rowMergeCounters))
- .addArgument(() -> mergeHistogramToString(rangeTombstonesMergeCounters))
- .addArgument(() -> mergeHistogramToString(cellMergeCounters))
- .addArgument(() -> sumHistogram(partitionMergeCounters))
- .addArgument(() -> sumHistogram(staticRowMergeCounters))
- .addArgument(() -> sumHistogram(rowMergeCounters))
- .addArgument(() -> sumHistogram(rangeTombstonesMergeCounters))
- .addArgument(() -> sumHistogram(cellMergeCounters))
- .log();
+ LOGGER.info("Compaction ended {}: { data bytes read = {}, data bytes written = {}, " +
+ "input (keys = {}, static rows = {}, rows = {}, range tombstones = {}, cells = {}), " +
+ "output (keys = {}, static rows = {}, rows = {}, range tombstones = {}, cells = {})}",
+ compactionId,
+ LazyToString.lazy(() -> Long.toString(getTotalBytesScanned())),
+ totalDataBytesWritten,
+ LazyToString.lazy(() -> mergeHistogramToString(partitionMergeCounters)),
+ LazyToString.lazy(() -> mergeHistogramToString(staticRowMergeCounters)),
+ LazyToString.lazy(() -> mergeHistogramToString(rowMergeCounters)),
+ LazyToString.lazy(() -> mergeHistogramToString(rangeTombstonesMergeCounters)),
+ LazyToString.lazy(() -> mergeHistogramToString(cellMergeCounters)),
+ LazyToString.lazy(() -> Long.toString(sumHistogram(partitionMergeCounters))),
+ LazyToString.lazy(() -> Long.toString(sumHistogram(staticRowMergeCounters))),
+ LazyToString.lazy(() -> Long.toString(sumHistogram(rowMergeCounters))),
+ LazyToString.lazy(() -> Long.toString(sumHistogram(rangeTombstonesMergeCounters))),
+ LazyToString.lazy(() -> Long.toString(sumHistogram(cellMergeCounters))));
}
}
diff --git a/src/java/org/apache/cassandra/db/compaction/StatefulCursor.java b/src/java/org/apache/cassandra/db/compaction/StatefulCursor.java
index aeea5da967a9..722ad71fe927 100644
--- a/src/java/org/apache/cassandra/db/compaction/StatefulCursor.java
+++ b/src/java/org/apache/cassandra/db/compaction/StatefulCursor.java
@@ -42,6 +42,7 @@
import static org.apache.cassandra.io.sstable.SSTableCursorReader.State.CELL_END;
import static org.apache.cassandra.io.sstable.SSTableCursorReader.State.CELL_HEADER_START;
import static org.apache.cassandra.io.sstable.SSTableCursorReader.State.CELL_VALUE_START;
+import static org.apache.cassandra.io.sstable.SSTableCursorReader.State.DONE;
import static org.apache.cassandra.io.sstable.SSTableCursorReader.State.UNFILTERED_END;
import static org.apache.cassandra.io.sstable.SSTableCursorReader.State.isState;
@@ -87,6 +88,12 @@ public StatefulCursor(SSTableReader reader, Collection
public int readPartitionHeader()
{
+ // Rejected here rather than in readPartitionHeader(PartitionDescriptor), which is past the
+ // swap below: a DONE cursor has no next partition, and rotating the descriptors on a call
+ // that cannot succeed leaves prev holding content the write side never wrote.
+ if (state() == DONE)
+ throw new IllegalStateException("readPartitionHeader() on a cursor that is DONE");
+
// A range never spans a partition, so one left open belongs to the partition that ended.
// Reporting it here names that partition; carrying the flag forward would blame the next
// partition's first start bound instead, and would hide an unmatched close in it.
diff --git a/src/java/org/apache/cassandra/db/compaction/writers/CompactionAwareWriter.java b/src/java/org/apache/cassandra/db/compaction/writers/CompactionAwareWriter.java
index 2369b3cbb728..4f68f7ee5d2d 100644
--- a/src/java/org/apache/cassandra/db/compaction/writers/CompactionAwareWriter.java
+++ b/src/java/org/apache/cassandra/db/compaction/writers/CompactionAwareWriter.java
@@ -156,6 +156,16 @@ public final String getSStableDirectoryPath() throws IOException
return sstableDirectoryPath;
}
+ /**
+ * Publishes an early-opened partial sstable once enough has been written since the last one.
+ * {@link #append} gets this from {@link SSTableRewriter#append}; the cursor path, which does not append,
+ * calls it directly on each partition boundary.
+ */
+ public final void maybeReopenEarly(DecoratedKey key)
+ {
+ sstableWriter.maybeReopenEarly(key);
+ }
+
@Override
protected Throwable doPostCleanup(Throwable accumulate)
{
diff --git a/src/java/org/apache/cassandra/dht/ReusableDecoratedKey.java b/src/java/org/apache/cassandra/dht/ReusableDecoratedKey.java
index 7a9723e17c20..7bbdb95f9bc1 100644
--- a/src/java/org/apache/cassandra/dht/ReusableDecoratedKey.java
+++ b/src/java/org/apache/cassandra/dht/ReusableDecoratedKey.java
@@ -19,8 +19,10 @@
package org.apache.cassandra.dht;
import java.nio.ByteBuffer;
+import java.util.Arrays;
import org.apache.cassandra.db.BufferDecoratedKey;
+import org.apache.cassandra.db.DecoratedKey;
import org.apache.cassandra.utils.ByteBufferUtil;
public abstract class ReusableDecoratedKey extends BufferDecoratedKey
@@ -46,6 +48,25 @@ public void copyKey(ByteBuffer newKey)
recalculateToken();
}
+ public void copyKey(byte[] newKey, int length)
+ {
+ maybeResizeKey(length);
+ System.arraycopy(newKey, 0, keyBytes, 0, length);
+ keyLength = length;
+ key.limit(length);
+ recalculateToken();
+ }
+
+ /**
+ * Always a copy, token included: the next copyKey overwrites the bytes and moves the token, so
+ * this key is never safe to retain as it is.
+ */
+ @Override
+ public DecoratedKey retainable()
+ {
+ return getToken().getPartitioner().decorateKey(ByteBuffer.wrap(Arrays.copyOf(keyBytes, keyLength)));
+ }
+
/** WARNING: retains ref to external buffer */
public void shadowKey(ByteBuffer newKey, byte[] newKeyBytes, int newKeyLength)
{
diff --git a/src/java/org/apache/cassandra/io/sstable/BigCursorIndexWriter.java b/src/java/org/apache/cassandra/io/sstable/BigCursorIndexWriter.java
index 7a477f2a30ba..6aef14e76989 100644
--- a/src/java/org/apache/cassandra/io/sstable/BigCursorIndexWriter.java
+++ b/src/java/org/apache/cassandra/io/sstable/BigCursorIndexWriter.java
@@ -27,6 +27,7 @@
import org.apache.cassandra.config.DatabaseDescriptor;
import org.apache.cassandra.db.ClusteringPrefix;
import org.apache.cassandra.db.DeletionTime;
+import org.apache.cassandra.db.TypeSizes;
import org.apache.cassandra.io.FSWriteError;
import org.apache.cassandra.io.sstable.format.big.BigFormatPartitionWriter;
import org.apache.cassandra.io.sstable.format.big.BigTableWriter;
@@ -43,6 +44,7 @@
*/
public class BigCursorIndexWriter extends CursorIndexWriter
{
+ private final BigTableWriter writer;
private final BigTableWriter.IndexWriter indexWriter;
private final DeletionTime.Serializer deletionTimeSerializer;
// The garbage-free add() overload exists only on the concrete BloomFilter. With
@@ -59,9 +61,11 @@ public class BigCursorIndexWriter extends CursorIndexWriter
private int rowIndexEntryOffset;
private final int indexBlockThreshold;
- public BigCursorIndexWriter(BigTableWriter.IndexWriter indexWriter,
+ public BigCursorIndexWriter(BigTableWriter writer,
+ BigTableWriter.IndexWriter indexWriter,
DeletionTime.Serializer deletionTimeSerializer)
{
+ this.writer = writer;
this.indexWriter = indexWriter;
this.deletionTimeSerializer = deletionTimeSerializer;
this.indexBlockThreshold = DatabaseDescriptor.getColumnIndexSize(BigFormatPartitionWriter.DEFAULT_GRANULARITY);
@@ -158,8 +162,8 @@ private void writeClusteringToRowIndexEntries(ClusteringDescriptor clustering) t
}
@Override
- public void endPartition(byte[] key, int keyLength, int headerLength,
- DeletionTime partitionDeletionTime, long partitionEnd,
+ public void endPartition(org.apache.cassandra.db.DecoratedKey decoratedKey, byte[] key, int keyLength,
+ int headerLength, DeletionTime partitionDeletionTime, long partitionEnd,
ClusteringDescriptor lastName) throws IOException
{
/**
@@ -171,6 +175,8 @@ public void endPartition(byte[] key, int keyLength, int headerLength,
if (bloomFilter != null)
bloomFilter.add(key, 0, keyLength, reusableIndexes);
long indexStart = indexFileWriter.position();
+ int columnIndexCount = 0;
+ int indexedPartSize = 0;
try
{
ByteArrayUtil.writeWithShortLength(key, 0, keyLength, indexFileWriter);
@@ -212,6 +218,10 @@ public void endPartition(byte[] key, int keyLength, int headerLength,
int entriesAndOffsetsSize = rowIndexEntries.getLength() + rowIndexEntriesOffsets.size() * 4;
assert entriesAndOffsetsSize > 0;
+ columnIndexCount = rowIndexEntriesOffsets.size();
+ // What RowIndexEntry calls indexedPartSize: the entries and their offsets, without the
+ // header fields that entriesAndOffsetsSize also counts.
+ indexedPartSize = endOfEntries + rowIndexEntriesOffsets.size() * 4;
indexFileWriter.writeUnsignedVInt32(entriesAndOffsetsSize); // size != 0
// copy the header elements
indexFileWriter.write(rowIndexEntries.getData(), endOfEntries, rowIndexEntries.getLength() - endOfEntries);
@@ -227,6 +237,13 @@ public void endPartition(byte[] key, int keyLength, int headerLength,
{
throw new FSWriteError(e, indexFileWriter.getPath());
}
- indexWriter.summary.maybeAddEntry(key, 0, keyLength, indexStart);
+ // indexEnd and partitionEnd feed the readable boundary that openEarly needs; without them the
+ // preemptive reopen has nothing to publish and never fires.
+ indexWriter.summary.maybeAddEntry(decoratedKey, key, 0, keyLength,
+ indexStart, indexFileWriter.position(), partitionEnd);
+
+ // The entry starts after the key, which was written at indexStart with a short length prefix.
+ writer.maybeCacheKey(decoratedKey, partitionStart, indexStart + TypeSizes.SHORT_SIZE + keyLength,
+ partitionDeletionTime, headerLength, columnIndexCount, indexedPartSize);
}
}
diff --git a/src/java/org/apache/cassandra/io/sstable/ClusteringDescriptorPrefixView.java b/src/java/org/apache/cassandra/io/sstable/ClusteringDescriptorPrefixView.java
new file mode 100644
index 000000000000..094dce8d785b
--- /dev/null
+++ b/src/java/org/apache/cassandra/io/sstable/ClusteringDescriptorPrefixView.java
@@ -0,0 +1,249 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.io.sstable;
+
+import java.nio.ByteBuffer;
+import java.util.Arrays;
+
+import org.apache.cassandra.db.ClusteringBound;
+import org.apache.cassandra.db.ClusteringPrefix;
+import org.apache.cassandra.db.marshal.AbstractType;
+import org.apache.cassandra.db.marshal.ByteBufferAccessor;
+import org.apache.cassandra.db.marshal.ValueAccessor;
+import org.apache.cassandra.schema.TableMetadata;
+import org.apache.cassandra.utils.vint.VIntCoding;
+
+/**
+ * A reusable {@link ClusteringPrefix} view over a {@link ClusteringDescriptor}'s serialized
+ * clustering bytes. {@link #reset} wraps the descriptor's array in place and parses the
+ * component boundaries; it copies nothing.
+ *
+ * Only the methods {@link org.apache.cassandra.db.ClusteringComparator#asByteComparable} calls
+ * are supported, plus {@link #retainable}: kind, size, get and accessor. Every other method
+ * throws {@link UnsupportedOperationException}. get(i) returns one shared window, re-positioned
+ * per call, so a caller must consume one component at a time. Use two views to compare two
+ * prefixes.
+ */
+public class ClusteringDescriptorPrefixView implements ClusteringPrefix
+{
+ private final AbstractType>[] types;
+ private int size;
+ private ClusteringPrefix.Kind kind;
+ private int[] offsets = new int[8];
+ private int[] lengths = new int[8]; // -1 = null component, 0 = empty
+ private byte[] backing;
+ private ByteBuffer window;
+ private int limit;
+ /** True when this view owns its byte copy, so the bytes outlive the descriptor. */
+ private boolean owned;
+
+ public ClusteringDescriptorPrefixView(AbstractType>[] types)
+ {
+ this.types = types;
+ }
+
+ /**
+ * Returns a view that owns a copy of the descriptor's bytes. It stays valid after the
+ * descriptor is reused, so a consumer that retains the prefix must use this instead of
+ * {@link #reset}.
+ */
+ public static ClusteringDescriptorPrefixView snapshotOf(ClusteringDescriptor descriptor, AbstractType>[] types)
+ {
+ return snapshot(types,
+ descriptor.clusteringKind(),
+ descriptor.clusteringColumnsBound(),
+ descriptor.clusteringBytes(),
+ descriptor.clusteringLength());
+ }
+
+ private static ClusteringDescriptorPrefixView snapshot(AbstractType>[] types,
+ ClusteringPrefix.Kind kind,
+ int size,
+ byte[] bytes,
+ int length)
+ {
+ ClusteringDescriptorPrefixView view = new ClusteringDescriptorPrefixView(types);
+ byte[] copy = Arrays.copyOf(bytes, length);
+ view.kind = kind;
+ view.size = size;
+ view.owned = true;
+ view.backing = copy;
+ view.window = ByteBuffer.wrap(copy);
+ view.parse(copy.length);
+ return view;
+ }
+
+ /**
+ * Points this view at the descriptor's live bytes and parses them. The view stays correct only
+ * until the descriptor is written again, so a consumer that retains it must call
+ * {@link #retainable}.
+ *
+ * @throws IllegalStateException if this view owns a copy, which {@link #snapshotOf} returns
+ */
+ public ClusteringDescriptorPrefixView reset(ClusteringDescriptor descriptor)
+ {
+ if (owned)
+ throw new IllegalStateException("a snapshot owns its bytes and cannot be reset");
+
+ this.kind = descriptor.clusteringKind();
+ this.size = descriptor.clusteringColumnsBound();
+ byte[] bytes = descriptor.clusteringBytes();
+ int limit = descriptor.clusteringLength();
+ if (backing != bytes || window == null)
+ {
+ backing = bytes;
+ window = ByteBuffer.wrap(bytes);
+ }
+ parse(limit);
+ return this;
+ }
+
+ // Wire format, as the cursor reader stores it: one vint block header per 32 components
+ // (bit 2i = empty, bit 2i+1 = null), then each present component as fixed-width raw bytes,
+ // or as a vint length followed by the bytes.
+ private void parse(int limit)
+ {
+ this.limit = limit;
+ if (offsets.length < size)
+ {
+ offsets = new int[size];
+ lengths = new int[size];
+ }
+
+ int pos = 0;
+ long header = 0;
+ for (int i = 0; i < size; i++)
+ {
+ if (i % 32 == 0)
+ {
+ window.limit(limit).position(pos);
+ header = VIntCoding.readUnsignedVInt(window);
+ pos = window.position();
+ }
+ long flags = (header >>> ((i % 32) * 2)) & 0b11;
+ if (flags == 0)
+ {
+ AbstractType> type = types[i];
+ int len;
+ if (type.isValueLengthFixed())
+ {
+ len = type.valueLengthIfFixed();
+ }
+ else
+ {
+ window.limit(limit).position(pos);
+ len = (int) VIntCoding.readUnsignedVInt(window);
+ pos = window.position();
+ }
+ offsets[i] = pos;
+ lengths[i] = len;
+ pos += len;
+ }
+ else if ((flags & 0b10) != 0) // null bit (2i+1)
+ {
+ offsets[i] = pos;
+ lengths[i] = -1;
+ }
+ else // empty bit (2i)
+ {
+ offsets[i] = pos;
+ lengths[i] = 0;
+ }
+ }
+ }
+
+ @Override
+ public Kind kind()
+ {
+ return kind;
+ }
+
+ @Override
+ public int size()
+ {
+ return size;
+ }
+
+ @Override
+ public ByteBuffer get(int i)
+ {
+ if (lengths[i] < 0)
+ return null;
+ window.limit(offsets[i] + lengths[i]).position(offsets[i]);
+ return window;
+ }
+
+ @Override
+ public ValueAccessor accessor()
+ {
+ return ByteBufferAccessor.instance;
+ }
+
+ @Override
+ public String toString(TableMetadata metadata)
+ {
+ throw new UnsupportedOperationException();
+ }
+
+ @Override
+ public ClusteringBound asStartBound()
+ {
+ throw new UnsupportedOperationException();
+ }
+
+ @Override
+ public ClusteringBound asEndBound()
+ {
+ throw new UnsupportedOperationException();
+ }
+
+ @Override
+ public ByteBuffer[] getRawValues()
+ {
+ throw new UnsupportedOperationException();
+ }
+
+ @Override
+ public ByteBuffer[] getBufferArray()
+ {
+ throw new UnsupportedOperationException();
+ }
+
+ /**
+ * Returns a prefix whose bytes outlive the descriptor. A view that already owns its bytes
+ * returns itself.
+ */
+ @Override
+ public ClusteringPrefix> retainable()
+ {
+ return owned ? this : snapshot(types, kind, size, backing, limit);
+ }
+
+ @Override
+ public long unsharedHeapSize()
+ {
+ throw new UnsupportedOperationException();
+ }
+
+ @Override
+ public ClusteringPrefix clustering()
+ {
+ return this;
+ }
+}
diff --git a/src/java/org/apache/cassandra/io/sstable/CursorIndexWriter.java b/src/java/org/apache/cassandra/io/sstable/CursorIndexWriter.java
index f5f9be58dcc1..5dc2e77c36a0 100644
--- a/src/java/org/apache/cassandra/io/sstable/CursorIndexWriter.java
+++ b/src/java/org/apache/cassandra/io/sstable/CursorIndexWriter.java
@@ -79,14 +79,24 @@ public abstract void rowWritten(UnfilteredDescriptor descriptor, long rowStart,
DeletionTime openMarker) throws IOException;
/**
- * The partition ends at partitionEnd, which includes the end-of-partition marker.
+ * The partition ends at partitionEnd, which is past its end-of-partition marker.
*
- * @param lastName the clustering of the last non-static unfiltered in this partition. A
- * trailing index block uses it as the block's last name. Null if the
- * partition wrote no non-static unfiltered, which leaves no trailing block
- * to cut.
+ * @param key the partition key; a reusable instance, whose bytes and token the next partition
+ * overwrites. Anything that keeps it past this call, such as the index summary's
+ * readable boundary, the key cache or the BTI partition index, takes retainable().
+ * @param lastName the clustering of the last non-static unfiltered written to this
+ * partition, or null if the partition wrote none.
*/
- public abstract void endPartition(byte[] key, int keyLength, int headerLength,
- DeletionTime partitionDeletionTime, long partitionEnd,
- ClusteringDescriptor lastName) throws IOException;
+ public abstract void endPartition(org.apache.cassandra.db.DecoratedKey key, byte[] keyBytes, int keyLength,
+ int headerLength, DeletionTime partitionDeletionTime,
+ long partitionEnd, ClusteringDescriptor lastName) throws IOException;
+
+ /**
+ * Releases the per-instance index state. The owning writer calls this when it closes.
+ * The underlying file writers belong to the table writer, so an implementation must not
+ * close them.
+ */
+ public void close()
+ {
+ }
}
diff --git a/src/java/org/apache/cassandra/io/sstable/SSTableCursorWriter.java b/src/java/org/apache/cassandra/io/sstable/SSTableCursorWriter.java
index 49c61e22387c..b3a1dd5cc8bf 100644
--- a/src/java/org/apache/cassandra/io/sstable/SSTableCursorWriter.java
+++ b/src/java/org/apache/cassandra/io/sstable/SSTableCursorWriter.java
@@ -42,9 +42,9 @@
import org.apache.cassandra.db.rows.SerializationHelper;
import org.apache.cassandra.db.rows.UnfilteredSerializer;
import org.apache.cassandra.dht.IPartitioner;
+import org.apache.cassandra.dht.ReusableDecoratedKey;
import org.apache.cassandra.io.sstable.format.SSTableReader;
import org.apache.cassandra.io.sstable.format.SortedTableWriter;
-import org.apache.cassandra.io.sstable.format.big.BigTableWriter;
import org.apache.cassandra.io.sstable.metadata.MetadataCollector;
import org.apache.cassandra.io.util.DataOutputBuffer;
import org.apache.cassandra.io.util.DataOutputPlus;
@@ -149,6 +149,10 @@ public class SSTableCursorWriter implements AutoCloseable
// Format-specific index production. BIG writes promoted blocks, Index.db, a bloom filter and a
// summary.
private final CursorIndexWriter cursorIndexWriter;
+ // The last key written, copied in per partition. It is the underlying writer's last key, so an
+ // sstable opened early at a writer switch carries real bounds. Whatever keeps it past the next
+ // partition takes retainable(), which copies it.
+ private final ReusableDecoratedKey lastKey;
private SSTableCursorWriter(
Descriptor desc,
@@ -167,8 +171,8 @@ private SSTableCursorWriter(
hasStaticColumns = serializationHeader.hasStatic();
staticColumns = hasStaticColumns ? serializationHeader.columns(true).toArray(EMPTY_COL_META) : EMPTY_COL_META;
regularColumns = serializationHeader.columns(false).toArray(EMPTY_COL_META);
- this.cursorIndexWriter = new BigCursorIndexWriter((BigTableWriter.IndexWriter) indexWriter,
- this.deletionTimeSerializer);
+ this.cursorIndexWriter = ssTableWriter.newCursorIndexWriter(serializationHeader);
+ this.lastKey = ssTableWriter.getPartitioner().createReusableKey(0);
// Same two conditions SortedTableWriter settles once, in its own constructor and in
// guardCollectionSize: both guardrails off, or a system keyspace.
this.collectionGuardsDisabled =
@@ -189,6 +193,7 @@ public SSTableCursorWriter(SortedTableWriter,?> ssTableWriter)
@Override
public void close()
{
+ cursorIndexWriter.close();
SSTableReader finish = ssTableWriter.finish(false);
if (finish != null) {
Ref ref = finish.ref();
@@ -225,7 +230,8 @@ public int writePartitionStart(byte[] partitionKey, int partitionKeyLength, Dele
* @param lastName the clustering of the last non-static unfiltered written to this partition, needed as
* the last name of a trailing index block; null if the partition wrote none.
*/
- public void writePartitionEnd(byte[] partitionKey, int partitionKeyLength, DeletionTime partitionDeletionTime,
+ public void writePartitionEnd(byte[] partitionKey,
+ int partitionKeyLength, DeletionTime partitionDeletionTime,
int headerLength, ClusteringDescriptor lastName) throws IOException
{
SERIALIZER.writeEndOfPartition(dataWriter);
@@ -233,17 +239,18 @@ public void writePartitionEnd(byte[] partitionKey, int partitionKeyLength, Delet
long partitionSize = partitionEnd - partitionStart;
addPartitionMetadata(partitionKey, partitionKeyLength, partitionSize, partitionDeletionTime);
+ // Per partition, not once at rollover: BigTableWriter.openInternal reads this field, so an sstable
+ // opened early at a writer switch would otherwise carry a stale last. The copy is into the
+ // reusable key, not a new one; the readers of last take retainable() when they keep it.
+ lastKey.copyKey(partitionKey, partitionKeyLength);
+ ssTableWriter.setLast(lastKey);
+
/** {@link SortedTableWriter#endPartition(DecoratedKey, DeletionTime)}
lastWrittenKey = key; // tracked for verification, see {@link SortedTableWriter#verifyPartition(DecoratedKey)}, checking the key size and sorting
- // first/last are retained for metadata {@link org.apache.cassandra.io.sstable.format.SSTableWriter#finalizeMetadata()}. They are also exposed via
- // getters from the writer, but usage is unclear.
- last = lastWrittenKey;
- if (first == null)
- first = lastWrittenKey;
// this is implemented differently for BIG/BTI
createRowIndexEntry(key, partitionLevelDeletion, partitionEnd - 1);
*/
- cursorIndexWriter.endPartition(partitionKey, partitionKeyLength, headerLength, partitionDeletionTime, partitionEnd, lastName);
+ cursorIndexWriter.endPartition(lastKey, partitionKey, partitionKeyLength, headerLength, partitionDeletionTime, partitionEnd, lastName);
}
@@ -255,13 +262,16 @@ public void writePartitionEnd(byte[] partitionKey, int partitionKeyLength, Delet
*/
private void addPartitionMetadata(byte[] partitionKey, int partitionKeyLength, long partitionSize, DeletionTime partitionDeletionTime)
{
+ // Before the guardrail check: SortedTableWriter counts the partition deletion in startPartition, so it
+ // is already in totalTombstones by the time the guardrail runs at partition end.
+ metadataCollector.updatePartitionDeletion(partitionDeletionTime);
+
if (partitionSize > guardrailsPartitionSizeWarning)
guardPartitionThreshold(Guardrails.partitionSize, partitionKey, partitionKeyLength, partitionSize);
if (metadataCollector.totalTombstones > guardrailsPartitionTombstonesWarning)
guardPartitionThreshold(Guardrails.partitionTombstones, partitionKey, partitionKeyLength, metadataCollector.totalTombstones);
- metadataCollector.updatePartitionDeletion(partitionDeletionTime);
metadataCollector.addPartitionSizeInBytes(partitionSize);
metadataCollector.addKey(partitionKey, 0, partitionKeyLength);
metadataCollector.addCellPerPartitionCount();
@@ -919,8 +929,7 @@ static void encodeColumnsSubset(IntArrayList missingColumns, int supersetCount,
public void setLast(ByteBuffer key)
{
IPartitioner partitioner = ssTableWriter.getPartitioner();
- DecoratedKey last = partitioner.decorateKey(ByteBufferUtil.clone(key));
- ssTableWriter.setLast(last);
+ ssTableWriter.setLast(partitioner.decorateKey(ByteBufferUtil.clone(key)));
}
public void setFirst(ByteBuffer key)
diff --git a/src/java/org/apache/cassandra/io/sstable/SSTableRewriter.java b/src/java/org/apache/cassandra/io/sstable/SSTableRewriter.java
index aaae5ecab876..3bfac8449d6e 100644
--- a/src/java/org/apache/cassandra/io/sstable/SSTableRewriter.java
+++ b/src/java/org/apache/cassandra/io/sstable/SSTableRewriter.java
@@ -156,7 +156,12 @@ public AbstractRowIndexEntry tryAppend(UnfilteredRowIterator partition)
}
}
- private void maybeReopenEarly(DecoratedKey key)
+ /**
+ * Publishes a partial reader once {@code preemptiveOpenInterval} bytes have been written since the last one,
+ * and moves the originals' starts past what it covers. Call this only on a partition boundary: the cursor
+ * path has no {@link #append} to hang it off.
+ */
+ public void maybeReopenEarly(DecoratedKey key)
{
if (writer.getFilePointer() - currentlyOpenedEarlyAt > preemptiveOpenInterval)
{
diff --git a/src/java/org/apache/cassandra/io/sstable/format/SSTableFormat.java b/src/java/org/apache/cassandra/io/sstable/format/SSTableFormat.java
index bd6f630701f0..dc0a09d48cfb 100644
--- a/src/java/org/apache/cassandra/io/sstable/format/SSTableFormat.java
+++ b/src/java/org/apache/cassandra/io/sstable/format/SSTableFormat.java
@@ -53,6 +53,15 @@ public interface SSTableFormat
SSTableReaderFactory getReaderFactory();
+ /**
+ * Whether cursor compaction can write this format. A format that returns true must also
+ * override {@link org.apache.cassandra.io.sstable.format.SortedTableWriter#newCursorIndexWriter}.
+ */
+ default boolean supportsCursorCompaction()
+ {
+ return false;
+ }
+
/**
* All the components that the writter can produce when saving an sstable, as well as all the components
* that the reader can read.
diff --git a/src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java b/src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java
index ab3171200f17..71b6ac08e8f4 100644
--- a/src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java
+++ b/src/java/org/apache/cassandra/io/sstable/format/SSTableReader.java
@@ -923,6 +923,17 @@ protected abstract AbstractRowIndexEntry getRowIndexEntry(PartitionPosition key,
boolean updateStats,
SSTableReadsListener listener);
+ /**
+ * Retrieves the index entry for a key, with stats and cache updates on and no read listener.
+ *
+ * @return The index entry corresponding to the key, or null if the key is not present
+ */
+ @VisibleForTesting
+ public AbstractRowIndexEntry getRowIndexEntry(PartitionPosition key, Operator op)
+ {
+ return getRowIndexEntry(key, op, true, SSTableReadsListener.NOOP_LISTENER);
+ }
+
public UnfilteredRowIterator simpleIterator(FileDataInput file, DecoratedKey key, long dataPosition, boolean tombstoneOnly)
{
return SSTableIdentityIterator.create(this, file, dataPosition, key, tombstoneOnly);
diff --git a/src/java/org/apache/cassandra/io/sstable/format/SortedTableWriter.java b/src/java/org/apache/cassandra/io/sstable/format/SortedTableWriter.java
index ec6c210daeca..8d0d7950174d 100644
--- a/src/java/org/apache/cassandra/io/sstable/format/SortedTableWriter.java
+++ b/src/java/org/apache/cassandra/io/sstable/format/SortedTableWriter.java
@@ -35,6 +35,7 @@
import org.apache.cassandra.db.DecoratedKey;
import org.apache.cassandra.db.DeletionPurger;
import org.apache.cassandra.db.DeletionTime;
+import org.apache.cassandra.db.SerializationHeader;
import org.apache.cassandra.db.guardrails.Guardrails;
import org.apache.cassandra.db.guardrails.Threshold;
import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
@@ -53,6 +54,7 @@
import org.apache.cassandra.io.compress.CompressionMetadata;
import org.apache.cassandra.io.sstable.AbstractRowIndexEntry;
import org.apache.cassandra.io.sstable.Component;
+import org.apache.cassandra.io.sstable.CursorIndexWriter;
import org.apache.cassandra.io.sstable.Descriptor;
import org.apache.cassandra.io.sstable.SSTable;
import org.apache.cassandra.io.sstable.SSTableFlushObserver;
@@ -309,6 +311,18 @@ protected void onRangeTombstoneMarker(RangeTombstoneMarker marker)
notifyObservers(o -> o.nextUnfilteredCluster(marker));
}
+ /**
+ * Builds the index writer that cursor compaction uses for this format. A format whose
+ * {@link org.apache.cassandra.io.sstable.format.SSTableFormat#supportsCursorCompaction} is false
+ * does not override this and never reaches the call.
+ *
+ * @param header the header the cursor writer writes with, which need not be this writer's own
+ */
+ public CursorIndexWriter newCursorIndexWriter(SerializationHeader header)
+ {
+ throw new UnsupportedOperationException("cursor compaction has no index writer for " + getClass().getName());
+ }
+
protected abstract AbstractRowIndexEntry createRowIndexEntry(DecoratedKey key, DeletionTime partitionLevelDeletion, long finishResult) throws IOException;
protected final void notifyObservers(Consumer action)
diff --git a/src/java/org/apache/cassandra/io/sstable/format/big/BigFormat.java b/src/java/org/apache/cassandra/io/sstable/format/big/BigFormat.java
index e6b60c2a0656..c9f51af1ba90 100644
--- a/src/java/org/apache/cassandra/io/sstable/format/big/BigFormat.java
+++ b/src/java/org/apache/cassandra/io/sstable/format/big/BigFormat.java
@@ -234,6 +234,12 @@ public static boolean isSelected()
return is(DatabaseDescriptor.getSelectedSSTableFormat());
}
+ @Override
+ public boolean supportsCursorCompaction()
+ {
+ return true;
+ }
+
@Override
public Version getLatestVersion()
{
diff --git a/src/java/org/apache/cassandra/io/sstable/format/big/BigTableWriter.java b/src/java/org/apache/cassandra/io/sstable/format/big/BigTableWriter.java
index 7fc4d2604ac1..62983ea0ddc2 100644
--- a/src/java/org/apache/cassandra/io/sstable/format/big/BigTableWriter.java
+++ b/src/java/org/apache/cassandra/io/sstable/format/big/BigTableWriter.java
@@ -32,11 +32,14 @@
import org.apache.cassandra.config.DatabaseDescriptor;
import org.apache.cassandra.db.DecoratedKey;
import org.apache.cassandra.db.DeletionTime;
+import org.apache.cassandra.db.SerializationHeader;
import org.apache.cassandra.db.compaction.OperationType;
import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
import org.apache.cassandra.index.Index;
import org.apache.cassandra.io.FSWriteError;
import org.apache.cassandra.io.sstable.AbstractRowIndexEntry;
+import org.apache.cassandra.io.sstable.BigCursorIndexWriter;
+import org.apache.cassandra.io.sstable.CursorIndexWriter;
import org.apache.cassandra.io.sstable.Descriptor;
import org.apache.cassandra.io.sstable.Downsampling;
import org.apache.cassandra.io.sstable.SSTable;
@@ -72,7 +75,8 @@ public class BigTableWriter extends SortedTableWriter cachedKeys = new HashMap<>();
- private final boolean shouldMigrateKeyCache;
+ private static final SSTableReader[] NO_ORIGINALS = new SSTableReader[0];
+ private final SSTableReader[] originals;
public BigTableWriter(Builder builder, ILifecycleTransaction txn, SSTable.Owner owner)
{
@@ -81,8 +85,61 @@ public BigTableWriter(Builder builder, ILifecycleTransaction txn, SSTable.Owner
this.rowIndexEntrySerializer = builder.getRowIndexEntrySerializer();
checkNotNull(this.rowIndexEntrySerializer);
- this.shouldMigrateKeyCache = DatabaseDescriptor.shouldMigrateKeycacheOnCompaction()
- && !txn.isOffline();
+ boolean migrateKeyCache = DatabaseDescriptor.shouldMigrateKeycacheOnCompaction() && !txn.isOffline();
+ // Empty unless the key cache is being migrated, so shouldCacheKey needs no second guard.
+ // LifecycleTransaction.originals() wraps a fresh set on each call, and shouldCacheKey scans
+ // this per partition. Safe to snapshot: the only cancel that drops a compaction's originals
+ // runs in CompactionTask.runMayThrow before this writer.
+ this.originals = migrateKeyCache ? txn.originals().toArray(NO_ORIGINALS) : NO_ORIGINALS;
+ }
+
+ @Override
+ public CursorIndexWriter newCursorIndexWriter(SerializationHeader header)
+ {
+ return new BigCursorIndexWriter(this, indexWriter, DeletionTime.getSerializer(descriptor.version));
+ }
+
+ /**
+ * Carries a key that is hot in the originals into this sstable's key cache, as
+ * {@link #createRowIndexEntry} does on the iterator path.
+ *
+ * The cursor path serialises the promoted index straight into Index.db and never builds the
+ * IndexInfo list, so a multi-block partition caches a shallow entry where the iterator path
+ * would cache a full one. Both find the same rows; the shallow one reads its index blocks from
+ * Index.db on a hit.
+ *
+ * @param key the partition's key; may be a reusable instance, the cache keeps a retainable copy
+ */
+ public void maybeCacheKey(DecoratedKey key, long dataFilePosition, long indexFilePosition,
+ DeletionTime partitionLevelDeletion, long headerLength,
+ int columnIndexCount, int indexedPartSize)
+ {
+ if (!shouldCacheKey(key))
+ return;
+
+ cachedKeys.put(key.retainable(), RowIndexEntry.create(dataFilePosition,
+ indexFilePosition,
+ partitionLevelDeletion,
+ headerLength,
+ columnIndexCount,
+ indexedPartSize,
+ null,
+ null,
+ rowIndexEntrySerializer.indexInfoSerializer(),
+ descriptor.version));
+ }
+
+ /**
+ * True when one of the transaction's originals has a cached position for this key. The array is
+ * empty unless key cache migration is on, so that setting is already folded in.
+ */
+ private boolean shouldCacheKey(DecoratedKey key)
+ {
+ for (SSTableReader reader : originals)
+ if (reader instanceof KeyCacheSupport> && ((KeyCacheSupport>) reader).getCachedPosition(key, false) != null)
+ return true;
+
+ return false;
}
@Override
@@ -111,17 +168,8 @@ protected RowIndexEntry createRowIndexEntry(DecoratedKey key, DeletionTime parti
indexWriter.append(key, entry, dataWriter.position(), partitionWriter.buffer());
- if (shouldMigrateKeyCache)
- {
- for (SSTableReader reader : txn.originals())
- {
- if (reader instanceof KeyCacheSupport> && ((KeyCacheSupport>) reader).getCachedPosition(key, false) != null)
- {
- cachedKeys.put(key, entry);
- break;
- }
- }
- }
+ if (shouldCacheKey(key))
+ cachedKeys.put(key, entry);
return entry;
}
diff --git a/src/java/org/apache/cassandra/io/sstable/format/bti/BtiCursorIndexWriter.java b/src/java/org/apache/cassandra/io/sstable/format/bti/BtiCursorIndexWriter.java
new file mode 100644
index 000000000000..5f9d1b3f768a
--- /dev/null
+++ b/src/java/org/apache/cassandra/io/sstable/format/bti/BtiCursorIndexWriter.java
@@ -0,0 +1,140 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.io.sstable.format.bti;
+
+import java.io.IOException;
+
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.db.ClusteringComparator;
+import org.apache.cassandra.db.DecoratedKey;
+import org.apache.cassandra.db.DeletionTime;
+import org.apache.cassandra.db.marshal.AbstractType;
+import org.apache.cassandra.io.sstable.ClusteringDescriptor;
+import org.apache.cassandra.io.sstable.ClusteringDescriptorPrefixView;
+import org.apache.cassandra.io.sstable.CursorIndexWriter;
+import org.apache.cassandra.io.sstable.UnfilteredDescriptor;
+import org.apache.cassandra.io.sstable.format.bti.RowIndexReader.IndexInfo;
+
+/**
+ * BTI index production for the cursor writer: a row-index trie per partition
+ * ({@link RowIndexWriter}) and a partition index entry ({@link TrieIndexEntry}) appended
+ * through {@link BtiTableWriter.IndexWriter}.
+ *
+ * It cuts blocks exactly as {@link BtiFormatPartitionWriter} does. A partition with one block
+ * writes no row trie and takes a trie root of -1.
+ */
+public class BtiCursorIndexWriter extends CursorIndexWriter
+{
+ private final BtiTableWriter.IndexWriter indexWriter;
+ private final RowIndexWriter rowTrie;
+ private final int rowIndexBlockSize;
+
+ private final ClusteringDescriptor firstClustering;
+ private final ClusteringDescriptor lastClustering;
+ private final AbstractType>[] clusteringTypes;
+ private boolean blockOpen; // a first clustering has been captured for the current block
+ private int rowIndexBlockCount;
+ private DeletionTime blockStartOpenMarker = DeletionTime.LIVE;
+
+ public BtiCursorIndexWriter(BtiTableWriter writer,
+ ClusteringComparator comparator,
+ AbstractType>[] clusteringTypes)
+ {
+ this.indexWriter = writer.indexWriter;
+ this.rowTrie = new RowIndexWriter(comparator, indexWriter.rowIndexWriter, writer.descriptor.version);
+ this.rowIndexBlockSize = DatabaseDescriptor.getColumnIndexSize(BtiFormatPartitionWriter.DEFAULT_GRANULARITY);
+ this.firstClustering = new ClusteringDescriptor(clusteringTypes);
+ this.lastClustering = new ClusteringDescriptor(clusteringTypes);
+ this.clusteringTypes = clusteringTypes;
+ }
+
+ @Override
+ protected void reset()
+ {
+ rowTrie.reset();
+ rowIndexBlockCount = 0;
+ blockOpen = false;
+ blockStartOpenMarker = DeletionTime.LIVE;
+ }
+
+ @Override
+ public void rowWritten(UnfilteredDescriptor descriptor, long rowStart, long rowEnd,
+ DeletionTime openMarker) throws IOException
+ {
+ if (!blockOpen)
+ {
+ firstClustering.copy(descriptor);
+ blockOpen = true;
+ }
+ lastClustering.copy(descriptor);
+
+ /** {@link BtiFormatPartitionWriter#addUnfiltered} */
+ if (currentOffsetInPartition(rowEnd) - indexBlockStartOffset >= rowIndexBlockSize)
+ addIndexBlock(rowEnd, openMarker);
+ }
+
+ /** {@link BtiFormatPartitionWriter#addIndexBlock()} */
+ private void addIndexBlock(long endOfRowPosition, DeletionTime openMarkerAtEnd) throws IOException
+ {
+ IndexInfo info = new IndexInfo(indexBlockStartOffset, blockStartOpenMarker);
+ // snapshot: RowIndexWriter holds the prefixes lazily across add() calls (prevMax), so a
+ // reusable view must not escape into it; this copy per block boundary is deliberate
+ rowTrie.add(ClusteringDescriptorPrefixView.snapshotOf(firstClustering, clusteringTypes),
+ ClusteringDescriptorPrefixView.snapshotOf(lastClustering, clusteringTypes), info);
+ blockOpen = false;
+ ++rowIndexBlockCount;
+ notePosition(endOfRowPosition);
+ // copy: the trie holds the IndexInfo until complete(), and the caller's DeletionTime is
+ // a reusable instance
+ blockStartOpenMarker = openMarkerAtEnd.isLive() ? DeletionTime.LIVE
+ : DeletionTime.build(openMarkerAtEnd.markedForDeleteAt(),
+ openMarkerAtEnd.localDeletionTime());
+ }
+
+ @Override
+ public void endPartition(DecoratedKey key, byte[] keyBytes, int keyLength, int headerLength,
+ DeletionTime partitionDeletionTime, long partitionEnd,
+ ClusteringDescriptor lastName) throws IOException
+ {
+ /** {@link BtiFormatPartitionWriter#finish()} + {@link BtiTableWriter#createRowIndexEntry} */
+ // lastName goes unused: lastClustering already holds that clustering
+ // the last row may not fall on a block boundary; cut the final block here
+ if (rowIndexBlockCount > 0 && blockOpen)
+ addIndexBlock(partitionEnd, DeletionTime.LIVE);
+
+ // SortedTablePartitionWriter.finish measures the partition length before it writes the
+ // end-of-partition marker, and complete() takes that length. partitionEnd here is the
+ // position after the marker, so subtract its one byte
+ long trieRoot = rowIndexBlockCount > 1 ? rowTrie.complete(partitionEnd - 1 - partitionStart) : -1;
+ TrieIndexEntry entry = TrieIndexEntry.create(partitionStart, trieRoot,
+ partitionDeletionTime, rowIndexBlockCount);
+ // PartitionIndexBuilder keeps the previous key to compute the next separator, so the key must be
+ // a copy. The iterator path pays the same one for its merge key.
+ indexWriter.append(key.retainable(), entry);
+ }
+
+ @Override
+ public void close()
+ {
+ /** {@link BtiFormatPartitionWriter#close()} */
+ // clears the trie builder's in-heap stack and prev state; the Rows.db writer belongs to
+ // BtiTableWriter
+ rowTrie.close();
+ }
+}
diff --git a/src/java/org/apache/cassandra/io/sstable/format/bti/BtiFormat.java b/src/java/org/apache/cassandra/io/sstable/format/bti/BtiFormat.java
index ff7f11ce17d9..daf4c6ea58c4 100644
--- a/src/java/org/apache/cassandra/io/sstable/format/bti/BtiFormat.java
+++ b/src/java/org/apache/cassandra/io/sstable/format/bti/BtiFormat.java
@@ -127,6 +127,12 @@ public static boolean isSelected()
return is(DatabaseDescriptor.getSelectedSSTableFormat());
}
+ @Override
+ public boolean supportsCursorCompaction()
+ {
+ return true;
+ }
+
@Override
public Version getLatestVersion()
{
diff --git a/src/java/org/apache/cassandra/io/sstable/format/bti/BtiFormatPartitionWriter.java b/src/java/org/apache/cassandra/io/sstable/format/bti/BtiFormatPartitionWriter.java
index 6bb024ab7e42..33785372935f 100644
--- a/src/java/org/apache/cassandra/io/sstable/format/bti/BtiFormatPartitionWriter.java
+++ b/src/java/org/apache/cassandra/io/sstable/format/bti/BtiFormatPartitionWriter.java
@@ -38,7 +38,7 @@
*/
class BtiFormatPartitionWriter extends SortedTablePartitionWriter
{
- private static final int DEFAULT_GRANULARITY = 16 * 1024;
+ static final int DEFAULT_GRANULARITY = 16 * 1024;
private final RowIndexWriter rowTrie;
private final int rowIndexBlockSize;
private int rowIndexBlockCount;
diff --git a/src/java/org/apache/cassandra/io/sstable/format/bti/BtiTableWriter.java b/src/java/org/apache/cassandra/io/sstable/format/bti/BtiTableWriter.java
index b0b7b2455c2a..bd02389330ad 100644
--- a/src/java/org/apache/cassandra/io/sstable/format/bti/BtiTableWriter.java
+++ b/src/java/org/apache/cassandra/io/sstable/format/bti/BtiTableWriter.java
@@ -28,14 +28,18 @@
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
+import org.apache.cassandra.db.ClusteringComparator;
import org.apache.cassandra.db.DecoratedKey;
import org.apache.cassandra.db.DeletionTime;
+import org.apache.cassandra.db.SerializationHeader;
import org.apache.cassandra.db.compaction.OperationType;
import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
+import org.apache.cassandra.db.marshal.AbstractType;
import org.apache.cassandra.index.Index;
import org.apache.cassandra.io.FSReadError;
import org.apache.cassandra.io.FSWriteError;
import org.apache.cassandra.io.sstable.AbstractRowIndexEntry;
+import org.apache.cassandra.io.sstable.CursorIndexWriter;
import org.apache.cassandra.io.sstable.Descriptor;
import org.apache.cassandra.io.sstable.SSTable;
import org.apache.cassandra.io.sstable.format.DataComponent;
@@ -71,6 +75,15 @@ public BtiTableWriter(Builder builder, ILifecycleTransaction txn, SSTable.Owner
super(builder, txn, owner);
}
+ @Override
+ public CursorIndexWriter newCursorIndexWriter(SerializationHeader header)
+ {
+ ClusteringComparator comparator = header.clusteringTypes().isEmpty()
+ ? new ClusteringComparator()
+ : new ClusteringComparator(header.clusteringTypes());
+ return new BtiCursorIndexWriter(this, comparator, header.clusteringTypes().toArray(AbstractType[]::new));
+ }
+
@Override
protected TrieIndexEntry createRowIndexEntry(DecoratedKey key, DeletionTime partitionLevelDeletion, long finishResult) throws IOException
{
diff --git a/src/java/org/apache/cassandra/io/sstable/indexsummary/IndexSummaryBuilder.java b/src/java/org/apache/cassandra/io/sstable/indexsummary/IndexSummaryBuilder.java
index 6c9699a01377..20fb005b8c17 100644
--- a/src/java/org/apache/cassandra/io/sstable/indexsummary/IndexSummaryBuilder.java
+++ b/src/java/org/apache/cassandra/io/sstable/indexsummary/IndexSummaryBuilder.java
@@ -194,12 +194,21 @@ public IndexSummaryBuilder maybeAddEntry(DecoratedKey decoratedKey, long indexSt
return maybeAddEntry(decoratedKey, indexStart, 0, 0);
}
/**
+ * The garbage-free counterpart of {@link #maybeAddEntry(DecoratedKey, long, long, long)}: it samples from
+ * the key bytes, and takes a retainable copy of the key only for a readable boundary, which is one record
+ * per summary interval.
+ *
+ * @param decoratedKey the key for this record; may be a reusable instance
* @param keyBytes the key data for this record
* @param offset key data offset in the keyBytes array
* @param length key data length
* @param indexStart the position in the index file this record begins
+ * @param indexEnd the position in the index file we need to be able to read to (exclusive) to read this record
+ * @param dataEnd the position in the data file we need to be able to read to (exclusive) to read this record;
+ * a value of 0 indicates we are not tracking readable boundaries
*/
- public IndexSummaryBuilder maybeAddEntry(byte[] keyBytes, int offset, int length, long indexStart) throws IOException
+ public IndexSummaryBuilder maybeAddEntry(DecoratedKey decoratedKey, byte[] keyBytes, int offset, int length,
+ long indexStart, long indexEnd, long dataEnd) throws IOException
{
if (keysWritten == nextSamplePosition)
{
@@ -217,6 +226,14 @@ public IndexSummaryBuilder maybeAddEntry(byte[] keyBytes, int offset, int length
"you should increase min_sampling_level");
}
}
+ else if (dataEnd != 0 && keysWritten + 1 == nextSamplePosition)
+ {
+ // this is the last key in this summary interval, so stash it
+ ReadableBoundary boundary = new ReadableBoundary(decoratedKey.retainable(), indexEnd, dataEnd,
+ (int) (offsets.length() / 4), entries.length());
+ lastReadableByData.put(dataEnd, boundary);
+ lastReadableByIndex.put(indexEnd, boundary);
+ }
keysWritten++;
return this;
diff --git a/test/distributed/org/apache/cassandra/distributed/test/CursorCompactionPipelineTest.java b/test/distributed/org/apache/cassandra/distributed/test/CursorCompactionPipelineTest.java
new file mode 100644
index 000000000000..2c1042790364
--- /dev/null
+++ b/test/distributed/org/apache/cassandra/distributed/test/CursorCompactionPipelineTest.java
@@ -0,0 +1,98 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.distributed.test;
+
+import java.io.IOException;
+
+import org.junit.Test;
+
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.db.Keyspace;
+import org.apache.cassandra.distributed.Cluster;
+import org.apache.cassandra.distributed.api.ConsistencyLevel;
+
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertTrue;
+
+/**
+ * Runs a real compaction on a real node with cursor compaction enabled, and asserts the cursor
+ * pipeline is the one that ran.
+ *
+ * Every other cursor test drives a {@code CompactionTask} it built itself. This one goes through a
+ * started node: yaml parsing of {@code cursor_compaction_enabled}, the compaction manager, the
+ * strategy, and the executor thread. A silent fallback to the iterator pipeline anywhere in that
+ * chain leaves every unit test green, because none of them exercise it.
+ *
+ * The counters are read inside the instance. They are static state in the node's own classloader, so
+ * a read from the test JVM would see zero however the node behaved.
+ */
+public class CursorCompactionPipelineTest extends TestBaseImpl
+{
+ private static final int PARTITIONS = 2000;
+
+ @Test
+ public void cursorPipelineRunsOnARealNode() throws IOException
+ {
+ try (Cluster cluster = init(builder().withNodes(1)
+ .withConfig(config -> config.set("autocompaction_on_startup_enabled", false)
+ .set("cursor_compaction_enabled", true))
+ .start()))
+ {
+ cluster.schemaChange(withKeyspace("CREATE TABLE %s.tbl (pk int, ck int, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH compaction = {'class':'SizeTieredCompactionStrategy', 'enabled':'false'}"));
+
+ // The node must genuinely have read the yaml setting, not merely accept the config key.
+ cluster.get(1).runOnInstance(() ->
+ assertTrue("cursor_compaction_enabled did not reach DatabaseDescriptor on the node",
+ DatabaseDescriptor.cursorCompactionEnabled()));
+
+ String padding = "x".repeat(200);
+ for (int i = 0; i < PARTITIONS; i++)
+ {
+ cluster.coordinator(1).execute(withKeyspace("INSERT INTO %s.tbl (pk, ck, v) VALUES (?, ?, ?)"),
+ ConsistencyLevel.ALL, i, 0, padding);
+ if (i % 500 == 0)
+ cluster.get(1).flush(KEYSPACE);
+ }
+ cluster.get(1).flush(KEYSPACE);
+
+ long[] counts = cluster.get(1).callOnInstance(() -> {
+ long cursorBefore = org.apache.cassandra.db.compaction.CompactionPipelineCounts.cursorPipelines();
+ long iteratorBefore = org.apache.cassandra.db.compaction.CompactionPipelineCounts.iteratorPipelines();
+
+ Keyspace.open(KEYSPACE).getColumnFamilyStore("tbl").forceMajorCompaction();
+
+ return new long[]{ org.apache.cassandra.db.compaction.CompactionPipelineCounts.cursorPipelines() - cursorBefore,
+ org.apache.cassandra.db.compaction.CompactionPipelineCounts.iteratorPipelines() - iteratorBefore };
+ });
+
+ assertTrue("no cursor pipeline was created for the major compaction, so the node fell back " +
+ "to the iterator path; iterator pipelines created: " + counts[1],
+ counts[0] > 0);
+ assertEquals("the node created an iterator pipeline while cursor compaction was enabled",
+ 0, counts[1]);
+
+ // The compaction must also have produced correct data, not merely run the right pipeline.
+ Object[][] rows = cluster.coordinator(1).execute(
+ withKeyspace("SELECT count(*) FROM %s.tbl"), ConsistencyLevel.ALL);
+ assertEquals("the cursor compaction lost or duplicated partitions",
+ (long) PARTITIONS, rows[0][0]);
+ }
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/CompactionPipelineCounts.java b/test/unit/org/apache/cassandra/db/compaction/CompactionPipelineCounts.java
index 9c0e267c5f20..25e33462615a 100644
--- a/test/unit/org/apache/cassandra/db/compaction/CompactionPipelineCounts.java
+++ b/test/unit/org/apache/cassandra/db/compaction/CompactionPipelineCounts.java
@@ -65,6 +65,21 @@ public static CompactionPipelineCounts mark()
DatabaseDescriptor.cursorCompactionEnabled());
}
+ /**
+ * The raw cursor-pipeline counter. In-JVM dtests read this inside the instance, where the
+ * counters actually live; a read from the test JVM sees its own classloader's zero.
+ */
+ public static long cursorPipelines()
+ {
+ return AbstractCompactionPipeline.cursorPipelinesCreated();
+ }
+
+ /** The raw iterator-pipeline counter; see {@link #cursorPipelines()}. */
+ public static long iteratorPipelines()
+ {
+ return AbstractCompactionPipeline.iteratorPipelinesCreated();
+ }
+
/**
* Asserts that at least one compaction selecting the expected pipeline happened since
* {@code before}, and that no cursor pipeline was created at all if cursor compaction was
diff --git a/test/unit/org/apache/cassandra/db/compaction/MergedPartitionCountsDifferentialTest.java b/test/unit/org/apache/cassandra/db/compaction/MergedPartitionCountsDifferentialTest.java
new file mode 100644
index 000000000000..d7cb1ed606ea
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/MergedPartitionCountsDifferentialTest.java
@@ -0,0 +1,107 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction;
+
+import java.util.Arrays;
+import java.util.Collection;
+import java.util.Set;
+import java.util.concurrent.atomic.AtomicReference;
+
+import org.junit.Test;
+
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.compaction.differential.DifferentialCompactionTester;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+
+import static org.junit.Assert.assertArrayEquals;
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertNotNull;
+
+/**
+ * {@code getMergedRowCounts} counts PARTITIONS, not rows: index i holds the number of partitions that
+ * were merged from i+1 sources. It reaches {@code compaction_history.rows_merged} and the "total
+ * partitions merged" log line, so a cursor path returning row counts would inflate both by the width
+ * of every partition.
+ *
+ * Lives in {@code org.apache.cassandra.db.compaction} for {@link AbstractCompactionPipeline}, which is
+ * package-private.
+ */
+public class MergedPartitionCountsDifferentialTest extends DifferentialCompactionTester
+{
+ private static final int PARTITIONS = 8;
+ private static final int ROWS_PER_PARTITION = 50;
+
+ /** Captures the pipeline's merged counts at the point CompactionTask reads them. */
+ private static TaskFactory capturing(AtomicReference sink)
+ {
+ return (cfs, txn, gcBefore) -> new CompactionTask(cfs, txn, gcBefore, false)
+ {
+ @Override
+ protected Collection finish(AbstractCompactionPipeline pipeline)
+ {
+ Collection result = super.finish(pipeline);
+ sink.set(pipeline.getMergedRowCounts());
+ return result;
+ }
+ };
+ }
+
+ private long[] mergedCountsFromOneCompaction(boolean cursor) throws Exception
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck bigint, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH compression = {'enabled': 'false'}");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ // Every partition appears in both sstables, and every row of it in both, so a row count and a
+ // partition count differ by exactly ROWS_PER_PARTITION.
+ for (int round = 0; round < 2; round++)
+ {
+ for (long pk = 0; pk < PARTITIONS; pk++)
+ for (long ck = 0; ck < ROWS_PER_PARTITION; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", pk, ck, "v" + round + '-' + ck);
+ flush();
+ }
+ assertEquals("the fixture needs two sstables to merge", 2, cfs.getLiveSSTables().size());
+
+ AtomicReference sink = new AtomicReference<>();
+ Set inputs = cfs.getLiveSSTables();
+ commitThroughFactory(cfs, cursor, capturing(sink));
+ long[] counts = sink.get();
+ assertNotNull("the task never reported merged counts", counts);
+ assertEquals("one counter per input sstable", inputs.size(), counts.length);
+ return counts;
+ }
+
+ @Test
+ public void bothPipelinesCountPartitionsNotRows() throws Exception
+ {
+ assumeCursorSupportedFormatSelected();
+
+ long[] iterator = mergedCountsFromOneCompaction(false);
+ long[] cursor = mergedCountsFromOneCompaction(true);
+
+ // Every partition came from both sources, so the two-source bucket holds them all.
+ assertArrayEquals("the iterator path must count partitions, or this says nothing",
+ new long[]{ 0, PARTITIONS }, iterator);
+ assertArrayEquals("the cursor path counted " + Arrays.toString(cursor) +
+ " where the iterator path counted " + Arrays.toString(iterator),
+ iterator, cursor);
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/StatefulCursorPartialRangeTest.java b/test/unit/org/apache/cassandra/db/compaction/StatefulCursorPartialRangeTest.java
new file mode 100644
index 000000000000..a57699a178dc
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/StatefulCursorPartialRangeTest.java
@@ -0,0 +1,347 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction;
+
+import java.util.ArrayList;
+import java.util.Arrays;
+import java.util.Collections;
+import java.util.List;
+
+import org.junit.Test;
+
+import org.apache.cassandra.config.Config.DiskAccessMode;
+import org.apache.cassandra.cql3.CQLTester;
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.DecoratedKey;
+import org.apache.cassandra.dht.IPartitioner;
+import org.apache.cassandra.dht.Range;
+import org.apache.cassandra.dht.Token;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+import org.apache.cassandra.io.sstable.format.SSTableReader.PartitionPositionBounds;
+import org.apache.cassandra.utils.ByteBufferUtil;
+
+import static org.apache.cassandra.io.sstable.SSTableCursorReader.State.DONE;
+import static org.apache.cassandra.io.sstable.SSTableCursorReader.State.PARTITION_END;
+import static org.apache.cassandra.io.sstable.SSTableCursorReader.State.PARTITION_START;
+import static org.apache.cassandra.io.sstable.SSTableCursorReader.State.ROW_START;
+import static org.apache.cassandra.io.sstable.SSTableCursorReader.State.TOMBSTONE_START;
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertFalse;
+import static org.junit.Assert.assertTrue;
+import static org.junit.Assert.fail;
+
+/**
+ * Low-level tests directly against {@link StatefulCursor}'s partial-range bound support
+ * (the bounds constructor), driving the cursor's own state machine by hand -
+ * {@link CursorCompactor} is not involved at all. Bounds are computed via
+ * {@link SSTableReader#getPositionsForRanges}, the same production entry point repair
+ * validation will use.
+ */
+public class StatefulCursorPartialRangeTest extends CQLTester
+{
+ private static final int PARTITION_COUNT = 6;
+
+ private SSTableReader flushSinglePartitionPerRowTable() throws Throwable
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck bigint, v1 bigint, PRIMARY KEY (pk, ck))");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+ for (long pk = 0; pk < PARTITION_COUNT; pk++)
+ execute("INSERT INTO %s (pk, ck, v1) VALUES (?, ?, ?)", pk, 0L, pk);
+ flush();
+ assertEquals(1, cfs.getLiveSSTables().size());
+ return cfs.getLiveSSTables().iterator().next();
+ }
+
+ /**
+ * Drives a single cursor from wherever {@code state} indicates it currently sits through
+ * the rest of the current partition (if any) to the next {@code PARTITION_START} or
+ * {@code DONE} - no merging, just enough of the state machine to move past row/tombstone
+ * content without inspecting it (mirrors {@code CursorCompactor.skipRowsOnStrictLiveness}'s
+ * use of the same {@code skipUnfiltered} primitive, but with {@code autoContinue=true} so it
+ * collapses straight through to the next meaningful state).
+ */
+ private static int finishPartition(StatefulCursor cursor, int state)
+ {
+ while (state == ROW_START || state == TOMBSTONE_START)
+ state = cursor.skipUnfiltered(true);
+ if (state == PARTITION_END)
+ state = cursor.continueReading();
+ return state;
+ }
+
+ private static List readAllPartitionKeys(StatefulCursor cursor, IPartitioner partitioner)
+ {
+ List keys = new ArrayList<>();
+ int state = cursor.state();
+ while (state != DONE)
+ {
+ assertEquals(PARTITION_START, state);
+ // Never DONE: readPartitionHeader(PartitionDescriptor) returns a header state or throws,
+ // and bound exhaustion is reported by finishPartition's continueReading() below.
+ state = cursor.readPartitionHeader();
+ // currentKey() is backed by a reusable, mutated-in-place key AND token (see
+ // IPartitioner.createReusableKey / Murmur3Partitioner.ReusableLongToken) - snapshot
+ // via decorateKey() immediately, or every entry in this list ends up aliasing the
+ // same mutable objects and reflecting whatever was read last.
+ keys.add(partitioner.decorateKey(ByteBufferUtil.clone(cursor.currentKey().getKey())));
+ state = finishPartition(cursor, state);
+ }
+ return keys;
+ }
+
+ /** {@code (exclusiveStart, inclusiveEnd]}, matching Cassandra's Range convention. */
+ private static Range rangeBetween(Token exclusiveStart, DecoratedKey inclusiveEnd)
+ {
+ return new Range<>(exclusiveStart, inclusiveEnd.getToken());
+ }
+
+ @Test
+ public void singleSegmentCoveringWholeFileMatchesFullRangeRead() throws Throwable
+ {
+ SSTableReader sstable = flushSinglePartitionPerRowTable();
+ List allKeysInTokenOrder = readAllPartitionKeys(new StatefulCursor(sstable, DiskAccessMode.standard), sstable.getPartitioner());
+ assertEquals(PARTITION_COUNT, allKeysInTokenOrder.size());
+
+ List bounds = Collections.singletonList(sstable.getPositionsForFullRange());
+
+ StatefulCursor bounded = new StatefulCursor(sstable, bounds, DiskAccessMode.standard);
+ assertEquals(allKeysInTokenOrder, readAllPartitionKeys(bounded, sstable.getPartitioner()));
+ }
+
+ @Test
+ public void multipleDisjointSegmentsReadOnlyTheSelectedPartitions() throws Throwable
+ {
+ SSTableReader sstable = flushSinglePartitionPerRowTable();
+ List allKeysInTokenOrder = readAllPartitionKeys(new StatefulCursor(sstable, DiskAccessMode.standard), sstable.getPartitioner());
+ assertEquals(PARTITION_COUNT, allKeysInTokenOrder.size());
+
+ // Isolate index 1 and index 4 (out of 6) as two separate, non-adjacent byte segments in
+ // the same sstable - exactly the shape a repair session covering disjoint token ranges
+ // produces against one file.
+ Range firstSegment = rangeBetween(allKeysInTokenOrder.get(0).getToken(), allKeysInTokenOrder.get(1));
+ Range secondSegment = rangeBetween(allKeysInTokenOrder.get(3).getToken(), allKeysInTokenOrder.get(4));
+ List bounds = sstable.getPositionsForRanges(Arrays.asList(firstSegment, secondSegment));
+ assertEquals("expected two disjoint byte segments for two disjoint token ranges", 2, bounds.size());
+
+ // The first segment starts mid-file (after key[0]'s partition) - a prerequisite for the
+ // byte-accounting assertion below to actually be able to catch a stale snapshot bug.
+ long firstSegmentLowerPosition = bounds.get(0).lowerPosition;
+ assertTrue("test setup: first segment must not start at file offset 0, or the assertion below can't expose a stale snapshot",
+ firstSegmentLowerPosition > 0);
+
+ StatefulCursor bounded = new StatefulCursor(sstable, bounds, DiskAccessMode.standard);
+ assertEquals("bytesReadSinceSnapshot() immediately after construction must not count the skipped prefix before the first segment",
+ 0L, bounded.bytesReadSinceSnapshot());
+ assertEquals(Arrays.asList(allKeysInTokenOrder.get(1), allKeysInTokenOrder.get(4)), readAllPartitionKeys(bounded, sstable.getPartitioner()));
+ }
+
+ /**
+ * Reads every assigned partition, accumulating {@link StatefulCursor#bytesReadSinceSnapshot()}
+ * the same incremental way {@code CursorCompactor.updateBytesRead} does (once per partition),
+ * and returns the total bytes reported as read.
+ */
+ private static long readAllAccumulatingBytesRead(StatefulCursor cursor)
+ {
+ long totalBytesRead = 0;
+ int state = cursor.state();
+ while (state != DONE)
+ {
+ state = cursor.readPartitionHeader();
+ if (state == DONE)
+ break;
+ state = finishPartition(cursor, state);
+ totalBytesRead += cursor.bytesReadSinceSnapshot();
+ }
+ totalBytesRead += cursor.bytesReadSinceSnapshot();
+ return totalBytesRead;
+ }
+
+ @Test
+ public void byteAccountingStaysSaneAcrossSegmentHop() throws Throwable
+ {
+ SSTableReader sstable = flushSinglePartitionPerRowTable();
+ List allKeysInTokenOrder = readAllPartitionKeys(new StatefulCursor(sstable, DiskAccessMode.standard), sstable.getPartitioner());
+ assertEquals(PARTITION_COUNT, allKeysInTokenOrder.size());
+
+ // Two disjoint segments (index 1 and index 4 of 6) with skipped partitions BETWEEN them -
+ // the seek from the first segment's end to the second segment's start jumps over partitions
+ // 2 and 3, whose bytes must never be counted as read.
+ Range firstSegment = rangeBetween(allKeysInTokenOrder.get(0).getToken(), allKeysInTokenOrder.get(1));
+ Range secondSegment = rangeBetween(allKeysInTokenOrder.get(3).getToken(), allKeysInTokenOrder.get(4));
+ List bounds = sstable.getPositionsForRanges(Arrays.asList(firstSegment, secondSegment));
+ assertEquals("expected two disjoint byte segments for two disjoint token ranges", 2, bounds.size());
+ assertTrue("test setup: segments must have a byte gap between them to expose a stale snapshot",
+ bounds.get(1).lowerPosition > bounds.get(0).upperPosition);
+
+ long estimatedBytes = (bounds.get(0).upperPosition - bounds.get(0).lowerPosition)
+ + (bounds.get(1).upperPosition - bounds.get(1).lowerPosition);
+
+ StatefulCursor bounded = new StatefulCursor(sstable, bounds, DiskAccessMode.standard);
+ long totalBytesRead = readAllAccumulatingBytesRead(bounded);
+
+ // Must equal exactly the sum of the two segments' sizes (getEstimatedBytes). Before the
+ // segment-hop snapshot fix, the inter-segment gap was counted too, pushing this past the
+ // estimate (>100% progress).
+ assertEquals("bytes read across a multi-segment read must equal the summed segment sizes, not include the skipped gap",
+ estimatedBytes, totalBytesRead);
+ }
+
+ @Test
+ public void boundExhaustionBeforeEndOfFileReportsDoneShortOfTheFileEnd() throws Throwable
+ {
+ SSTableReader sstable = flushSinglePartitionPerRowTable();
+ List allKeysInTokenOrder = readAllPartitionKeys(new StatefulCursor(sstable, DiskAccessMode.standard), sstable.getPartitioner());
+
+ // A segment covering only the first 3 (of 6) partitions - stops well before true EOF.
+ Range earlySegment = rangeBetween(sstable.getPartitioner().getMinimumToken(), allKeysInTokenOrder.get(2));
+ List bounds = sstable.getPositionsForRanges(Collections.singletonList(earlySegment));
+
+ StatefulCursor bounded = new StatefulCursor(sstable, bounds, DiskAccessMode.standard);
+ List keysRead = readAllPartitionKeys(bounded, sstable.getPartitioner());
+
+ assertEquals(allKeysInTokenOrder.subList(0, 3), keysRead);
+ assertEquals("bound-exhausted cursor must report DONE", DONE, bounded.state());
+ assertTrue("bound-exhausted cursor must report DONE (isEOF)", bounded.isEOF());
+ assertTrue("position must be short of the sstable's full uncompressed length - the file has more data past the assigned bounds",
+ bounded.position() < bounded.uncompressedLength());
+
+ // Byte accounting must reflect the actual position reached, not the whole file's length.
+ assertEquals(bounded.position(), bounded.bytesReadSinceSnapshot());
+ }
+
+ @Test
+ public void lastSegmentReachingTheEndOfFileStopsAtTheFileEnd() throws Throwable
+ {
+ SSTableReader sstable = flushSinglePartitionPerRowTable();
+
+ List bounds = Collections.singletonList(sstable.getPositionsForFullRange());
+
+ StatefulCursor bounded = new StatefulCursor(sstable, bounds, DiskAccessMode.standard);
+ readAllPartitionKeys(bounded, sstable.getPartitioner());
+
+ assertEquals(DONE, bounded.state());
+ assertTrue(bounded.isEOF());
+ assertEquals("a bound that extends to the end of the file must stop at the file's end",
+ bounded.uncompressedLength(), bounded.position());
+ }
+
+ /** Reads to DONE via bound exhaustion, stopping short of true end of file. */
+ private static StatefulCursor exhaustBounds(SSTableReader sstable, List allKeysInTokenOrder)
+ {
+ Range firstTwo = rangeBetween(sstable.getPartitioner().getMinimumToken(), allKeysInTokenOrder.get(1));
+ List bounds = sstable.getPositionsForRanges(Collections.singletonList(firstTwo));
+ StatefulCursor bounded = new StatefulCursor(sstable, bounds, DiskAccessMode.standard);
+ readAllPartitionKeys(bounded, sstable.getPartitioner());
+ assertEquals(DONE, bounded.state());
+ assertTrue("test setup: cursor must stop on bound exhaustion, short of the file's end",
+ bounded.position() < bounded.uncompressedLength());
+ return bounded;
+ }
+
+ /**
+ * A cursor in DONE must reject a repeat {@code readPartitionHeader()}, and must reject it before
+ * the curr/prev swap. {@code readPartitionHeader(PartitionDescriptor)} throws on its
+ * {@code state != PARTITION_START} check, but that is past the swap, so without the guard every
+ * rejected call still rotated the descriptors and left prev on content nothing had written.
+ *
+ * At DONE the slots hold the last partition read in curr and the one before it in prev, on both
+ * DONE routes: {@code readPartitionHeader(PartitionDescriptor)} never returns DONE, so the
+ * transition always happens in {@code advanceSegment}, with no swap.
+ */
+ @Test
+ public void readPartitionHeaderRejectsReentryAfterBoundExhaustedDone() throws Throwable
+ {
+ SSTableReader sstable = flushSinglePartitionPerRowTable();
+ IPartitioner partitioner = sstable.getPartitioner();
+ List allKeysInTokenOrder = readAllPartitionKeys(new StatefulCursor(sstable, DiskAccessMode.standard), partitioner);
+ StatefulCursor bounded = exhaustBounds(sstable, allKeysInTokenOrder);
+
+ assertEquals("test setup: currentKey must be the last partition inside the bounds",
+ allKeysInTokenOrder.get(1),
+ partitioner.decorateKey(ByteBufferUtil.clone(bounded.currentKey().getKey())));
+ assertEquals("test setup: prevKey must be the partition before it",
+ allKeysInTokenOrder.get(0),
+ partitioner.decorateKey(ByteBufferUtil.clone(bounded.prevKey().getKey())));
+ long swapsAtDone = bounded.partitionSwaps();
+
+ for (int attempt = 0; attempt < 3; attempt++)
+ {
+ try
+ {
+ bounded.readPartitionHeader();
+ fail("readPartitionHeader() must reject re-entry once the cursor is DONE");
+ }
+ catch (IllegalStateException expected)
+ {
+ // a rejected call must not have disturbed the cursor's descriptors
+ assertEquals("a rejected re-entry must leave currentKey() untouched",
+ allKeysInTokenOrder.get(1),
+ partitioner.decorateKey(ByteBufferUtil.clone(bounded.currentKey().getKey())));
+ assertEquals("a rejected re-entry must leave prevKey() untouched",
+ allKeysInTokenOrder.get(0),
+ partitioner.decorateKey(ByteBufferUtil.clone(bounded.prevKey().getKey())));
+ assertEquals("a rejected re-entry must not advance the slots",
+ swapsAtDone, bounded.partitionSwaps());
+ }
+ }
+ }
+
+ /**
+ * Pins the POST-reset invariant {@code CursorCompactor} depends on: whichever route a cursor took
+ * to DONE, {@code resetAfterDone()} leaves {@code prevKey()} on the last partition actually read
+ * (consulted as the last key written to an output sstable) and {@code currPartition} cleared.
+ *
+ * What this does NOT prove, spelled out so it is not over-trusted: on its own it does not
+ * discriminate {@code readPartitionHeader()}'s swap-ORDERING fix. Revert that fix AND
+ * {@code resetAfterDone()}'s conditional swap together, and the bound-exhaustion route still nets
+ * exactly one swap by the time this test looks - the two bugs cancel - so it passes against the
+ * fully pre-fix code. {@link #readPartitionHeaderRejectsReentryAfterBoundExhaustedDone} is what
+ * catches that, because it inspects {@code prevKey()} BEFORE any reset, where the cancellation
+ * has not happened yet. Verified by reverting both fixes and observing exactly that split. This
+ * test does discriminate a revert of the conditional swap alone.
+ */
+ @Test
+ public void resetAfterDonePreservesLastReadKeyOnBothDoneRoutes() throws Throwable
+ {
+ SSTableReader sstable = flushSinglePartitionPerRowTable();
+ IPartitioner partitioner = sstable.getPartitioner();
+ List allKeysInTokenOrder = readAllPartitionKeys(new StatefulCursor(sstable, DiskAccessMode.standard), partitioner);
+
+ // route 1: bound exhaustion
+ StatefulCursor bounded = exhaustBounds(sstable, allKeysInTokenOrder);
+ assertTrue(bounded.resetAfterDone());
+ assertEquals("bound-exhausted cursor must keep the last partition it read in prevKey()",
+ allKeysInTokenOrder.get(1),
+ partitioner.decorateKey(ByteBufferUtil.clone(bounded.prevKey().getKey())));
+ assertEquals("the stale current partition must be cleared", 0, bounded.currPartition().keyLength());
+ assertFalse("resetAfterDone() is once-only", bounded.resetAfterDone());
+
+ // route 2: true end of file (DONE returned by the read itself)
+ StatefulCursor unbounded = new StatefulCursor(sstable, DiskAccessMode.standard);
+ readAllPartitionKeys(unbounded, partitioner);
+ assertEquals(DONE, unbounded.state());
+ assertEquals(unbounded.uncompressedLength(), unbounded.position());
+ assertTrue(unbounded.resetAfterDone());
+ assertEquals("EOF cursor must keep the last partition it read in prevKey()",
+ allKeysInTokenOrder.get(PARTITION_COUNT - 1),
+ partitioner.decorateKey(ByteBufferUtil.clone(unbounded.prevKey().getKey())));
+ assertEquals("the stale current partition must be cleared", 0, unbounded.currPartition().keyLength());
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/BtiCursorCompactionAllocationGateTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/BtiCursorCompactionAllocationGateTest.java
new file mode 100644
index 000000000000..de30e3c5ea09
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/BtiCursorCompactionAllocationGateTest.java
@@ -0,0 +1,81 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+
+/**
+ * Runs the inherited allocation tests with the BTI format selected. The measured region then
+ * covers BTI's index path: per-block boundary prefix copies, IndexInfo, and open-marker
+ * snapshots.
+ *
+ * The wide-schema sparse-row ceiling is deliberately NOT overridden. BTI measures 0.505-0.514
+ * B/B there against BIG's 0.352-0.353, and the inherited 0.6 leaves the same headroom the
+ * complex-column ceiling below keeps. An override would only loosen it.
+ */
+public class BtiCursorCompactionAllocationGateTest extends CursorCompactionAllocationGateTest
+{
+ @Override
+ protected String formatName()
+ {
+ return "bti";
+ }
+
+ @Override
+ protected long ceilingBytes()
+ {
+ return 768 * 1024;
+ }
+
+ /**
+ * The same per-partition BTI index cost, expressed per input byte. Marker-dense partitions
+ * are small, about 10KB, so 2KB per partition adds 0.2 to 0.3 B/B over the BIG residual.
+ * Measured 1.012 B/B under BTI against 0.684 under BIG. A leak of one small object per
+ * marker costs more than 1.5 B/B, so a ceiling of 1.3 still fails.
+ */
+ @Override
+ protected double rtPerInputByteCeiling()
+ {
+ return 1.3;
+ }
+
+ /**
+ * The complex-column test runs at multi-MB scale, unlike the range-tombstone one. BTI's
+ * 2KB per partition then spreads across far more input bytes and barely moves the ratio.
+ * Measured 0.511 B/B under BTI against about 0.5 under BIG, whose ceiling is 0.5. A
+ * ceiling of 0.6 keeps headroom comparable to the other ceilings and still fails on a
+ * per-row regression.
+ */
+ @Override
+ protected double complexPerInputByteCeiling()
+ {
+ return 0.6;
+ }
+
+ /**
+ * The large-file test compacts ~40MB, so BTI's per-partition index cost spreads thin: measured
+ * 0.247-0.248 B/B over three runs, against 0.178-0.179 under BIG. The inherited 0.5 was
+ * calibrated for BIG and leaves BTI room to double its allocation unnoticed. 0.32 keeps the
+ * ~30% headroom the range-tombstone ceiling above uses.
+ */
+ @Override
+ protected double largeFilePerInputByteCeiling()
+ {
+ return 0.32;
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/BtiCursorEarlyOpenBoundaryTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/BtiCursorEarlyOpenBoundaryTest.java
new file mode 100644
index 000000000000..2f70141710ec
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/BtiCursorEarlyOpenBoundaryTest.java
@@ -0,0 +1,51 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+/**
+ * The BTI half of {@link CursorEarlyOpenBoundaryTest}.
+ *
+ * BTI reaches no index summary, so it cannot carry the boundary defect the parent class is named
+ * for. It has its own retention instead: {@code PartitionIndexBuilder} holds {@code firstKey} for
+ * the whole build and the previous key across {@code addEntry} to compute each separator, then
+ * writes both bounds into the Partitions.db footer. Handed a key the next partition overwrites,
+ * both bounds collapse onto the final partition and every separator derives from a key that has
+ * since moved, so the trie routes seeks to the wrong place.
+ *
+ * The parent's committed-bounds assertion is what catches that here. Its early-open assertions still
+ * run on whatever BTI publishes, but nothing requires a publication: {@code BtiTableWriter.openEarly}
+ * defers through {@code PartitionIndexBuilder.buildPartial} until the data, row index and partition
+ * index writers have all flushed past the recorded ends, refuses a second request while one is
+ * pending, and {@code openFinalEarly} cancels what is still outstanding. Zero reopens is correct
+ * behaviour, not a failure, so a count-based oracle is unsound on this format.
+ */
+public class BtiCursorEarlyOpenBoundaryTest extends CursorEarlyOpenBoundaryTest
+{
+ @Override
+ protected String formatName()
+ {
+ return "bti";
+ }
+
+ @Override
+ protected boolean requiresMidStreamReopen()
+ {
+ return false;
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/BtiCursorReadTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/BtiCursorReadTest.java
new file mode 100644
index 000000000000..c513a3e9fc1f
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/BtiCursorReadTest.java
@@ -0,0 +1,48 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+import org.junit.After;
+import org.junit.Before;
+
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.io.sstable.format.SSTableFormat;
+
+/**
+ * Runs every reader-level scenario, including the allocation walk, with the BTI format
+ * selected, so the flushed inputs are BTI sstables. SSTableCursorReader opens only Data.db,
+ * which BTI writes exactly as BIG does, so every scenario must hold unchanged.
+ */
+public class BtiCursorReadTest extends ComplexColumnCursorReadTest
+{
+ private SSTableFormat, ?> originalFormat;
+
+ @Before
+ public void selectBti()
+ {
+ originalFormat = DatabaseDescriptor.getSelectedSSTableFormat();
+ DatabaseDescriptor.setSelectedSSTableFormat("bti");
+ }
+
+ @After
+ public void restoreFormat()
+ {
+ DatabaseDescriptor.setSelectedSSTableFormat(originalFormat);
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/BtiDifferentialCompactionTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/BtiDifferentialCompactionTest.java
new file mode 100644
index 000000000000..687703d22bee
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/BtiDifferentialCompactionTest.java
@@ -0,0 +1,49 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+import org.junit.After;
+import org.junit.Before;
+
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.io.sstable.format.SSTableFormat;
+
+/**
+ * Runs the full edge-case corpus with the BTI format selected, so both pipelines read and
+ * write BTI sstables. The harness byte-compares every component, which under BTI includes the
+ * partition and row index tries (Partitions.db, Rows.db). A trie is deterministic given
+ * identical inputs, so byte identity is the correct bar for it too.
+ */
+public class BtiDifferentialCompactionTest extends EdgeCaseDifferentialCompactionTest
+{
+ private SSTableFormat, ?> originalFormat;
+
+ @Before
+ public void selectBti()
+ {
+ originalFormat = DatabaseDescriptor.getSelectedSSTableFormat();
+ DatabaseDescriptor.setSelectedSSTableFormat("bti");
+ }
+
+ @After
+ public void restoreFormat()
+ {
+ DatabaseDescriptor.setSelectedSSTableFormat(originalFormat);
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/BtiMultiOutputDifferentialCompactionTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/BtiMultiOutputDifferentialCompactionTest.java
new file mode 100644
index 000000000000..44cd06fdbf16
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/BtiMultiOutputDifferentialCompactionTest.java
@@ -0,0 +1,78 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+import org.junit.After;
+import org.junit.Before;
+
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.io.sstable.format.SSTableFormat;
+
+/**
+ * Runs the size-capped multi-output scenarios with the BTI format selected. A writer switch is the
+ * only route to any of the following, and nothing else in this suite takes it under BTI:
+ *
+ * - {@code PartitionIndexBuilder.firstKey}, {@code lastKey} and {@code complete()} per output
+ * ({@code PartitionIndexBuilder.java:130-183}). With one output those run once over the whole
+ * key range; with several they run over a partial range whose bounds the switch decided. All
+ * three inherited scenarios reach this.
+ * - {@code BtiTableWriter.openFinalEarly} on a SWITCHED writer. {@code SSTableRewriter} early-opens
+ * the finished writer as it switches ({@code SSTableRewriter.java:271-282}), and early open stays
+ * enabled in this harness deliberately. All three inherited scenarios reach this too.
+ * - A fresh {@code SSTableCursorWriter} per output, and so a fresh {@code BtiCursorIndexWriter}
+ * and a fresh {@code RowIndexWriter} — {@code CursorCompactor} builds them when it switches
+ * ({@code CursorCompactor.java:1788-1793}). Every other BTI scenario constructs exactly one of
+ * each per compaction, so per-output construction and per-output {@code close()} are otherwise
+ * unexercised.
+ *
+ * Only {@code widePartitionsForceFrequentSwitches} builds a ROW TRIE here: its partitions hold about
+ * 6 KiB, above the 4 KiB column_index_size, so {@code BtiCursorIndexWriter.endPartition} cuts a block
+ * plus a tail and calls {@code RowIndexWriter.complete}. {@code manyPartitionsSplitAcrossOutputs} and
+ * {@code tombstonesAndStaticsAcrossOutputs} hold about 1 KiB per partition, so their block count never
+ * exceeds one and every partition takes the {@code trieRoot = -1} arm. Those two cover the switch and
+ * the per-output partition index; they say nothing about the row trie, and per-output
+ * {@code RowIndexWriter.reset()} does nothing observable on them.
+ *
+ * Beyond the switch itself, the format selection also puts BTI's own components — {@code Partitions.db}
+ * and {@code Rows.db} — into the per-output byte comparison; under BIG the same scenarios compare
+ * {@code Index.db} and {@code Summary.db} instead.
+ *
+ * The inherited scenarios each assert {@code out.sstables.size() >= 2}, so a data volume that stopped
+ * rolling over under BTI fails here rather than passing vacuously. The switch decision reads
+ * {@code SortedTableWriter.getEstimatedOnDiskBytesWritten()}, which is the DATA file's position only,
+ * so it is the same number under both formats for the same rows; the index components differ in size
+ * but do not enter the decision.
+ */
+public class BtiMultiOutputDifferentialCompactionTest extends MultiOutputDifferentialCompactionTest
+{
+ private SSTableFormat, ?> originalFormat;
+
+ @Before
+ public void selectBti()
+ {
+ originalFormat = DatabaseDescriptor.getSelectedSSTableFormat();
+ DatabaseDescriptor.setSelectedSSTableFormat("bti");
+ }
+
+ @After
+ public void restoreFormat()
+ {
+ DatabaseDescriptor.setSelectedSSTableFormat(originalFormat);
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/BtiRandomDifferentialCompactionTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/BtiRandomDifferentialCompactionTest.java
new file mode 100644
index 000000000000..f9c03e6e02f4
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/BtiRandomDifferentialCompactionTest.java
@@ -0,0 +1,47 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+import org.junit.After;
+import org.junit.Before;
+
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.io.sstable.format.SSTableFormat;
+
+/**
+ * Runs the randomized differential soak with the BTI format selected: both pipelines read and
+ * write BTI sstables, and the byte comparison covers the index tries.
+ */
+public class BtiRandomDifferentialCompactionTest extends RandomDifferentialCompactionTest
+{
+ private SSTableFormat, ?> originalFormat;
+
+ @Before
+ public void selectBti()
+ {
+ originalFormat = DatabaseDescriptor.getSelectedSSTableFormat();
+ DatabaseDescriptor.setSelectedSSTableFormat("bti");
+ }
+
+ @After
+ public void restoreFormat()
+ {
+ DatabaseDescriptor.setSelectedSSTableFormat(originalFormat);
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/CursorCompactionAllocationGateTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/CursorCompactionAllocationGateTest.java
index f6a6afe2a156..ce0529c65c12 100644
--- a/test/unit/org/apache/cassandra/db/compaction/differential/CursorCompactionAllocationGateTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/CursorCompactionAllocationGateTest.java
@@ -23,7 +23,9 @@
import java.util.List;
import java.util.Set;
+import org.junit.After;
import org.junit.Assume;
+import org.junit.Before;
import org.junit.Test;
import org.apache.cassandra.config.DatabaseDescriptor;
@@ -33,6 +35,7 @@
import org.apache.cassandra.db.compaction.OperationType;
import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
import org.apache.cassandra.io.sstable.Descriptor;
+import org.apache.cassandra.io.sstable.format.SSTableFormat;
import org.apache.cassandra.io.sstable.format.SSTableReader;
import org.apache.cassandra.utils.FBUtilities;
import org.apache.cassandra.utils.ThreadStats;
@@ -75,6 +78,34 @@ public class CursorCompactionAllocationGateTest extends DifferentialCompactionTe
private static final int MEASURED_ITERATIONS = 3;
private static final long CEILING_BYTES = 512 * 1024;
+ private SSTableFormat, ?> originalFormat;
+
+ /** The format these ceilings were measured against. A format subclass overrides it. */
+ protected String formatName()
+ {
+ return "big";
+ }
+
+ @Before
+ public void selectFormat()
+ {
+ originalFormat = DatabaseDescriptor.getSelectedSSTableFormat();
+ DatabaseDescriptor.setSelectedSSTableFormat(formatName());
+ }
+
+ @After
+ public void restoreFormat()
+ {
+ DatabaseDescriptor.setSelectedSSTableFormat(originalFormat);
+ }
+
+ /** A format subclass raises this: another sstable format may allocate more in its
+ * index path. */
+ protected long ceilingBytes()
+ {
+ return CEILING_BYTES;
+ }
+
private interface ThrowingRunnable
{
void run() throws Exception;
@@ -168,13 +199,13 @@ public void allocationDoesNotScaleWithRows() throws Exception
logger.info("cursor compaction allocation: small={}B big={}B delta={}B ceiling={}B " +
"(iterator path for context: small={}B big={}B delta={}B)",
- smallAlloc, bigAlloc, delta, CEILING_BYTES,
+ smallAlloc, bigAlloc, delta, ceilingBytes(),
smallIter, bigIter, bigIter - smallIter);
assertTrue(String.format("cursor compaction allocation scales with data: " +
"%,dB (small) -> %,dB (big), delta %,dB exceeds ceiling %,dB. " +
"A per-row/cell allocation has been introduced on the cursor hot path.",
- smallAlloc, bigAlloc, delta, CEILING_BYTES),
- delta <= CEILING_BYTES);
+ smallAlloc, bigAlloc, delta, ceilingBytes()),
+ delta <= ceilingBytes());
});
}
@@ -239,22 +270,33 @@ public void allocationAtLargeFileSizes() throws Exception
long bigIter = measureSteadyStateAllocation(192, false, 4, padding, 2, 2);
logger.info("LARGE-FILE cursor compaction allocation (4 files, ~10MB each big): " +
- "cursor small={}B big={}B delta={}B over {}B extra input = {}B/B; " +
- "iterator small={}B big={}B delta={}B",
+ "cursor small={}B big={}B delta={}B over {}B extra input = {}B/B " +
+ "(ceiling {} B/B); iterator small={}B big={}B delta={}B",
smallAlloc, bigAlloc, delta, extraBytes, String.format("%.3f", perInputByte),
- smallIter, bigIter, bigIter - smallIter);
+ largeFilePerInputByteCeiling(), smallIter, bigIter, bigIter - smallIter);
// The residual scales with data VOLUME, not row count. JFR decomposition at this
// scale: 62% Ref$Debug stack captures (test env only,
// -Dcassandra.debugrefcount=true), then chunk-cache machinery and per-compaction
// constants. ZERO cursor-owned sites. Measured ~0.27 B allocated per extra input
- // byte in the test env. Ceiling 0.5 B/B trips on any real per-element regression
- // and absorbs the volume-proportional test-env noise.
- assertTrue(String.format("cursor allocation per input byte too high: %.3f B/B (delta %,dB over %,dB)",
- perInputByte, delta, extraBytes),
- perInputByte <= 0.5);
+ // byte in the test env. The ceiling trips on any real per-element regression and
+ // absorbs the volume-proportional test-env noise.
+ assertTrue(String.format("cursor allocation per input byte too high: %.3f B/B (delta %,dB over %,dB, " +
+ "ceiling %.2f B/B)",
+ perInputByte, delta, extraBytes, largeFilePerInputByteCeiling()),
+ perInputByte <= largeFilePerInputByteCeiling());
});
}
+ /** Ceiling for {@link #allocationAtLargeFileSizes}, calibrated on BIG: measured ~0.27 B/B in
+ * the test env, all of it volume-proportional residual (Ref$Debug, chunk cache) by JFR
+ * attribution, with 0.5 B/B leaving room for that noise and none for a per-element
+ * regression. A format subclass raises this: BTI adds a row trie and a partition index,
+ * ~2KB per partition, which this number does not include. */
+ protected double largeFilePerInputByteCeiling()
+ {
+ return 0.5;
+ }
+
/** Compacts all live sstables on the configured path, measuring ONLY execute(); restores inputs. */
private long compactOnceMeasured(ColumnFamilyStore cfs, long gcBefore) throws Exception
{
@@ -307,11 +349,11 @@ public void allocationDoesNotScaleWithSparseRows() throws Exception
long bigAlloc = measureSparse(SMALL_PARTITIONS * SCALE);
long delta = bigAlloc - smallAlloc;
logger.info("sparse-row cursor compaction allocation: small={}B big={}B delta={}B ceiling={}B",
- smallAlloc, bigAlloc, delta, CEILING_BYTES);
+ smallAlloc, bigAlloc, delta, ceilingBytes());
assertTrue(String.format("sparse-row cursor compaction allocation scales with data: " +
"%,dB -> %,dB, delta %,dB exceeds ceiling %,dB",
- smallAlloc, bigAlloc, delta, CEILING_BYTES),
- delta <= CEILING_BYTES);
+ smallAlloc, bigAlloc, delta, ceilingBytes()),
+ delta <= ceilingBytes());
});
}
@@ -371,21 +413,31 @@ public void allocationDoesNotScaleWithWideSchemaSparseRows() throws Exception
long extraBytes = bigBytes - smallBytes;
double perInputByte = (double) delta / extraBytes;
logger.info("wide-schema sparse-row cursor compaction allocation: small={}B big={}B delta={}B " +
- "over {}B extra input = {} B/B",
- smallAlloc, bigAlloc, delta, extraBytes, String.format("%.3f", perInputByte));
+ "over {}B extra input = {} B/B (ceiling {} B/B)",
+ smallAlloc, bigAlloc, delta, extraBytes, String.format("%.3f", perInputByte),
+ wideSchemaPerInputByteCeiling());
// Calibrated per INPUT BYTE: the mixed 3-of-69 and 67-of-69 rows make multi-MB
// inputs whose volume-proportional test-env residual (Ref$Debug, chunk cache)
- // dwarfs any fixed ceiling. Measured ~0.37 B/B on the BIG run. The per-row Columns
- // cascade this gate guards measured ~3.8 B/B. One small object leaked per row costs
- // about +0.2 B/B at this row size, lands at ~0.57 B/B, and still passes. The gate
- // catches a whole-pipeline regression, not a single re-introduced per-row object.
+ // dwarfs any fixed ceiling. The per-row Columns cascade this gate guards measured
+ // ~3.8 B/B. One small object leaked per row costs about +0.2 B/B at this row size,
+ // lands at ~0.57 B/B, and still passes. The gate catches a whole-pipeline
+ // regression, not a single re-introduced per-row object.
assertTrue(String.format("wide-schema (>=64 col) sparse-row cursor allocation per input byte too high: " +
- "%.3f B/B (delta %,dB over %,dB extra input)",
- perInputByte, delta, extraBytes),
- perInputByte <= 0.6);
+ "%.3f B/B (delta %,dB over %,dB extra input, ceiling %.2f B/B)",
+ perInputByte, delta, extraBytes, wideSchemaPerInputByteCeiling()),
+ perInputByte <= wideSchemaPerInputByteCeiling());
});
}
+ /** Ceiling for {@link #allocationDoesNotScaleWithWideSchemaSparseRows}, calibrated on BIG:
+ * measured ~0.37 B/B, against the ~3.8 B/B the per-row Columns cascade cost when it was
+ * present. A format subclass raises this: BTI adds a row trie and a partition index,
+ * ~2KB per partition, which this number does not include. */
+ protected double wideSchemaPerInputByteCeiling()
+ {
+ return 0.6;
+ }
+
private long measureWideSparse(int partitions) throws Exception
{
DatabaseDescriptor.setCursorCompactionEnabled(true);
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/CursorCompactionGateTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/CursorCompactionGateTest.java
index 5c36e07cb45f..66ee1dcef787 100644
--- a/test/unit/org/apache/cassandra/db/compaction/differential/CursorCompactionGateTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/CursorCompactionGateTest.java
@@ -39,7 +39,6 @@
import org.apache.cassandra.dht.Token;
import org.apache.cassandra.io.sstable.ISSTableScanner;
import org.apache.cassandra.io.sstable.format.SSTableReader;
-import org.apache.cassandra.io.sstable.format.big.BigFormat;
import org.apache.cassandra.schema.CompactionParams.TombstoneOption;
import org.apache.cassandra.schema.TableMetadata;
import org.apache.cassandra.utils.FBUtilities;
@@ -83,13 +82,13 @@ private ColumnFamilyStore twoSSTableTable()
}
/**
- * Whether the gate can accept any compaction at all under the running configuration. The cursor
- * path writes the BIG format only, and {@code test/conf/latest_diff.yaml} selects BTI, so an
- * assertion that the gate opens has to read the format rather than assume it.
+ * Whether the gate can accept any compaction at all under the running configuration. Which
+ * formats the cursor path writes changes as the patch series lands, and the test configs
+ * differ in what they select, so the oracle asks the format rather than naming one.
*/
private static boolean cursorSupportsSelectedFormat()
{
- return DatabaseDescriptor.getSelectedSSTableFormat() instanceof BigFormat;
+ return DatabaseDescriptor.getSelectedSSTableFormat().supportsCursorCompaction();
}
private boolean isSupportedWith(ColumnFamilyStore cfs, TombstoneOption tombstoneOption) throws Exception
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/CursorCompactionProgressTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/CursorCompactionProgressTest.java
new file mode 100644
index 000000000000..e52694e91aab
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/CursorCompactionProgressTest.java
@@ -0,0 +1,169 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+import java.util.ArrayList;
+import java.util.List;
+
+import org.junit.Test;
+
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.compaction.ActiveCompactionsTracker;
+import org.apache.cassandra.db.compaction.CompactionInfo;
+import org.apache.cassandra.db.compaction.CompactionTask;
+
+import static org.junit.Assert.assertTrue;
+
+/**
+ * Pins that compaction progress advances inside one large partition on the cursor path.
+ *
+ * {@code CursorCompactor} used to update {@code totalBytesRead} only when a partition header was
+ * read or a cursor reset, so {@code nodetool compactionstats} showed nothing moving while a single
+ * large partition was merged, and partition N's bytes only landed once partition N+1's header was
+ * read. The iterator path refreshes {@code CompactionInfo.bytesRead} every 100 unfiltereds
+ * ({@code CompactionIterator}); the cursor path now refreshes every
+ * {@code CursorCompactor.UNFILTERED_TO_UPDATE_PROGRESS}.
+ *
+ * The scenario is one partition with many rows, so a per-partition update produces exactly one
+ * distinct intermediate value and this test fails, while a per-unfiltered update produces many.
+ *
+ * The cadence itself is deliberately not asserted. It is a private constant on both paths and a
+ * test that pins it would break on any retuning without anything being wrong.
+ */
+public class CursorCompactionProgressTest extends DifferentialCompactionTester
+{
+ /** One partition, enough rows that a per-unfiltered cadence has many chances to fire. */
+ private static final int ROWS = 120_000;
+
+ /**
+ * The iterator path's granularity is the yardstick, so both paths are measured in one run: the
+ * sampler is timing-sensitive and two separate runs would compare counts taken under different
+ * machine load.
+ *
+ * The margin is an order of magnitude, which is far wider than the two cadences differ (128
+ * against 100) and far narrower than the defect. Measured on this fixture: with the in-partition
+ * refresh the cursor path reports 917 distinct values against the iterator's 1079; with it
+ * disabled, 2 against 1663.
+ */
+ private static final int GRANULARITY_MARGIN = 10;
+
+ @Test
+ public void progressAdvancesWithinOnePartitionAsOftenAsTheIteratorPath() throws Throwable
+ {
+ long iterator = distinctProgressValues(false);
+ long cursor = distinctProgressValues(true);
+
+ assertTrue("the iterator path is the yardstick and it reported no progress inside the " +
+ "partition, so this scenario cannot judge the cursor path", iterator > 2);
+
+ assertTrue("compaction progress barely moved inside the partition on the cursor path: " +
+ cursor + " distinct values against the iterator path's " + iterator + ". A reader " +
+ "of nodetool compactionstats would watch one large partition merge with the " +
+ "counter stuck.",
+ cursor >= iterator / GRANULARITY_MARGIN);
+ }
+
+ private long distinctProgressValues(boolean cursor) throws Throwable
+ {
+ ColumnFamilyStore cfs = oneLargePartitionInTwoSSTables();
+
+ SamplingTracker tracker = new SamplingTracker();
+ commitThroughFactory(cfs, cursor,
+ (store, txn, gcBefore) -> new CompactionTask(store, txn, gcBefore, false),
+ tracker);
+
+ return tracker.distinctIntermediateValues();
+ }
+
+ private ColumnFamilyStore oneLargePartitionInTwoSSTables() throws Throwable
+ {
+ createTable("CREATE TABLE %s (pk int, ck int, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH compression = {'enabled': 'false'}");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ String padding = "x".repeat(60);
+ for (int round = 0; round < 2; round++)
+ {
+ for (int ck = 0; ck < ROWS; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", 1, ck + round * ROWS, padding);
+ flush();
+ }
+ assertTrue("the fixture needs inputs", cfs.getLiveSSTables().size() >= 2);
+ return cfs;
+ }
+
+ /**
+ * Samples {@code getCompleted()} from a thread while the compaction runs, and keeps the values
+ * that fall strictly between zero and the total. A path that only updates per partition reports
+ * one such value for a single-partition compaction; a path that updates per unfiltered reports
+ * many.
+ */
+ private static final class SamplingTracker implements ActiveCompactionsTracker
+ {
+ private final List samples = new ArrayList<>();
+ private volatile boolean running;
+ private volatile long total;
+ private Thread sampler;
+
+ @Override
+ public void beginCompaction(CompactionInfo.Holder holder)
+ {
+ total = holder.getCompactionInfo().getTotal();
+ running = true;
+ sampler = new Thread(() -> {
+ while (running)
+ {
+ long completed = holder.getCompactionInfo().getCompleted();
+ synchronized (samples)
+ {
+ samples.add(completed);
+ }
+ Thread.onSpinWait();
+ }
+ }, "compaction-progress-sampler");
+ sampler.setDaemon(true);
+ sampler.start();
+ }
+
+ @Override
+ public void finishCompaction(CompactionInfo.Holder holder)
+ {
+ running = false;
+ try
+ {
+ if (sampler != null)
+ sampler.join(10_000);
+ }
+ catch (InterruptedException e)
+ {
+ Thread.currentThread().interrupt();
+ }
+ }
+
+ /** Sampling can outlive the join timeout, so the lock stays. */
+ long distinctIntermediateValues()
+ {
+ synchronized (samples)
+ {
+ return samples.stream().filter(v -> v > 0 && v < total).distinct().count();
+ }
+ }
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/CursorDiskBoundaryDifferentialTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/CursorDiskBoundaryDifferentialTest.java
new file mode 100644
index 000000000000..5feeba5a6393
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/CursorDiskBoundaryDifferentialTest.java
@@ -0,0 +1,221 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+import java.util.ArrayList;
+import java.util.Comparator;
+import java.util.List;
+import java.util.Set;
+
+import org.junit.Test;
+
+import org.apache.cassandra.Util;
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.DecoratedKey;
+import org.apache.cassandra.db.Directories;
+import org.apache.cassandra.db.DiskBoundaries;
+import org.apache.cassandra.db.PartitionPosition;
+import org.apache.cassandra.db.compaction.CompactionTask;
+import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
+import org.apache.cassandra.db.compaction.writers.DefaultCompactionWriter;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+import org.apache.cassandra.io.util.File;
+import org.apache.cassandra.tcm.Epoch;
+import org.apache.cassandra.utils.ByteBufferUtil;
+
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertTrue;
+
+/**
+ * The multi-data-directory boundary switch on the cursor path.
+ *
+ * {@code CompactionAwareWriter.maybeSwitchLocation} splits a compaction's output across data
+ * directories at fixed token positions, finishing one sstable and starting the next in another
+ * directory. {@code DefaultCompactionWriter.shouldSwitchWriterInCurrentLocation} always returns
+ * false, so nothing else in this scenario can split the output: an output count above one is proof
+ * the boundary path ran, and the count is exactly the number of directories the keys span.
+ *
+ * The switch decision itself is shared between the two compaction paths. What is not shared is when
+ * it is consulted: the iterator path asks from {@code SSTableRewriter.append}, the cursor path from
+ * {@code CursorCompactor.maybeSwitchWriter} at a partition boundary, handing it the cursor's own
+ * reusable key. So this pins that the split lands on the same partitions on both paths, and that
+ * every committed output holds only keys belonging to its directory.
+ */
+public class CursorDiskBoundaryDifferentialTest extends DifferentialCompactionTester
+{
+ private static final int DIRECTORIES = 3;
+ private static final int PARTITIONS = 600;
+
+ @Test
+ public void diskBoundarySwitchMatchesIteratorOnBothPaths() throws Throwable
+ {
+ ColumnFamilyStore cfs = populated();
+ ColumnFamilyStore bounded = withDirectories(cfs, boundaryPositions(cfs));
+
+ assertCursorMatchesIterator(cfs, cfs.getLiveSSTables(), splitAcrossDirectories(bounded, true));
+ }
+
+ @Test
+ public void committedOutputsRespectTheirDirectoryBoundaries() throws Throwable
+ {
+ ColumnFamilyStore cfs = populated();
+ List positions = boundaryPositions(cfs);
+ ColumnFamilyStore bounded = withDirectories(cfs, positions);
+
+ commitThroughFactory(cfs, true, splitAcrossDirectories(bounded, false));
+
+ List outputs = new ArrayList<>(cfs.getLiveSSTables());
+ assertEquals("one output per directory the keys span; DefaultCompactionWriter cannot split " +
+ "for any other reason, so a single output means the boundary path never ran",
+ DIRECTORIES, outputs.size());
+
+ for (SSTableReader output : outputs)
+ {
+ int first = boundaryIndexOf(positions, output.getFirst());
+ int last = boundaryIndexOf(positions, output.getLast());
+ assertEquals("output " + output.descriptor.id + " spans a disk boundary: its first key " +
+ "belongs to directory " + first + " and its last to directory " + last,
+ first, last);
+
+ // It must also physically sit in that directory, which is the half of the switch that
+ // maybeSwitchLocation performs rather than decides.
+ // Absolute on both sides: the configured data directory may be a relative path.
+ String expected = new File(bounded.getDirectories().getLocationForDisk(
+ bounded.getDiskBoundaries().directories.get(first)).path()).absolutePath();
+ assertTrue("output " + output.descriptor.id + " belongs to directory " + first +
+ " but was written to " + output.descriptor.directory.absolutePath(),
+ output.descriptor.directory.absolutePath().startsWith(expected));
+ }
+ }
+
+ /** The index of the first boundary at or above this key: the directory the key belongs to. */
+ private static int boundaryIndexOf(List positions, DecoratedKey key)
+ {
+ for (int i = 0; i < positions.size(); i++)
+ if (key.compareTo(positions.get(i)) <= 0)
+ return i;
+ return positions.size();
+ }
+
+ /**
+ * Two positions taken from the fixture's own keys, plus the partitioner's maximum, so the output
+ * must split into exactly {@link #DIRECTORIES} pieces.
+ */
+ private static List boundaryPositions(ColumnFamilyStore cfs)
+ {
+ List keys = new ArrayList<>(PARTITIONS);
+ for (int pk = 0; pk < PARTITIONS; pk++)
+ keys.add(cfs.getPartitioner().decorateKey(ByteBufferUtil.bytes(pk)));
+ keys.sort(Comparator.naturalOrder());
+
+ // maxKeyBound, as DiskBoundaryManager itself builds them, not the keys themselves.
+ // CompactionAwareWriter.maybeSwitchLocation early-returns on `< 0` but advances on `> 0`, so a
+ // key exactly equal to a boundary takes neither branch and switches to the directory it is
+ // already in. A bound sits above every key of that token, so equality cannot arise.
+ List positions = new ArrayList<>(DIRECTORIES);
+ positions.add(keys.get(PARTITIONS / 3).getToken().maxKeyBound());
+ positions.add(keys.get(2 * PARTITIONS / 3).getToken().maxKeyBound());
+ positions.add(cfs.getPartitioner().getMaximumTokenForSplitting().maxKeyBound());
+ return positions;
+ }
+
+ /**
+ * {@code Directories.dataDirectories} is a static final built at class load from the yaml, so a
+ * test cannot give the real table more data directories after the fact. This builds a second view
+ * of the same table with directories of its own and boundaries of its own.
+ *
+ * The boundaries are fabricated rather than derived from local ranges by
+ * {@code DiskBoundaryManager}, so they fall on known keys of this fixture and the scenario
+ * controls where the switch must happen instead of asserting against whatever ownership produced.
+ */
+ private static ColumnFamilyStore withDirectories(ColumnFamilyStore real, List positions)
+ {
+ Directories.DataDirectory[] dirs = new Directories.DataDirectory[DIRECTORIES];
+ for (int i = 0; i < DIRECTORIES; i++)
+ {
+ File dir = new File(DatabaseDescriptor.getAllDataFileLocations()[0],
+ "boundary-" + real.getTableName() + '-' + i);
+ dir.tryCreateDirectories();
+ dirs[i] = new Directories.DataDirectory(dir);
+ }
+ return new BoundedCFS(real, new Directories(real.metadata(), dirs), dirs, positions);
+ }
+
+ /** A view of one table carrying its own data directories and its own disk boundaries. */
+ private static final class BoundedCFS extends ColumnFamilyStore
+ {
+ private final DiskBoundaries boundaries;
+
+ BoundedCFS(ColumnFamilyStore real, Directories directories,
+ Directories.DataDirectory[] dirs, List positions)
+ {
+ super(real.keyspace, real.getTableName(), Util.newSeqGen(), real.metadata.get(),
+ directories, false, false);
+ this.boundaries = new DiskBoundaries(this, dirs, positions, Epoch.EMPTY, 0);
+ }
+
+ @Override
+ public DiskBoundaries getDiskBoundaries()
+ {
+ // ColumnFamilyStore's constructor reaches this override before the field is assigned.
+ return boundaries == null ? super.getDiskBoundaries() : boundaries;
+ }
+ }
+
+ /**
+ * The writer is built over the bounded view so it sees several directories; the transaction
+ * stays on the real table, so the outputs are tracked and asserted there. The parameter cannot
+ * be named keepOriginals; see {@link TaskFactory}.
+ */
+ private static TaskFactory splitAcrossDirectories(ColumnFamilyStore bounded, boolean retainOriginals)
+ {
+ return (cfs, txn, gcBefore) -> new CompactionTask(cfs, txn, gcBefore, retainOriginals)
+ {
+ @Override
+ public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore ignored,
+ Directories directories,
+ ILifecycleTransaction transaction,
+ Set nonExpiredSSTables)
+ {
+ return new DefaultCompactionWriter(bounded, bounded.getDirectories(), transaction,
+ nonExpiredSSTables, retainOriginals, 0);
+ }
+ };
+ }
+
+ private ColumnFamilyStore populated() throws Throwable
+ {
+ createTable("CREATE TABLE %s (pk int, ck int, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH compression = {'enabled': 'false'}");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ String padding = "x".repeat(200);
+ for (int round = 0; round < 2; round++)
+ {
+ for (int pk = 0; pk < PARTITIONS; pk++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", pk, round, padding);
+ flush();
+ }
+ assertTrue("the fixture needs inputs", cfs.getLiveSSTables().size() >= 2);
+ return cfs;
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/CursorEarlyOpenBoundaryTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/CursorEarlyOpenBoundaryTest.java
new file mode 100644
index 000000000000..21aa72f0e5b5
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/CursorEarlyOpenBoundaryTest.java
@@ -0,0 +1,339 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+import java.nio.ByteBuffer;
+import java.util.ArrayList;
+import java.util.Collection;
+import java.util.Comparator;
+import java.util.List;
+import java.util.Set;
+
+import org.junit.After;
+import org.junit.Before;
+import org.junit.Test;
+
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.DecoratedKey;
+import org.apache.cassandra.db.Directories;
+import org.apache.cassandra.db.compaction.CompactionTask;
+import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
+import org.apache.cassandra.db.compaction.writers.DefaultCompactionWriter;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
+import org.apache.cassandra.db.lifecycle.WrappedLifecycleTransaction;
+import org.apache.cassandra.dht.IPartitioner;
+import org.apache.cassandra.dht.Token;
+import org.apache.cassandra.io.sstable.format.SSTableFormat;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+import org.apache.cassandra.utils.ByteBufferUtil;
+
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertTrue;
+
+/**
+ * Pins the key an early-opened sstable carries as its {@code last}, on the BIG format.
+ *
+ * {@code sstable_preemptive_open_interval} publishes a partial reader every so many bytes.
+ * {@code BigTableWriter.openEarly} takes that reader's bounds from
+ * {@code IndexSummaryBuilder.ReadableBoundary.lastKey}, which the index writer stashes per sampled
+ * entry through {@code retainable()}. {@code retainable()} clones the bytes but keeps the caller's
+ * {@code Token} object, and returns the caller's key untouched when the buffer is already an exact
+ * fit. The cursor path's key and token are both reusable instances that the next partition
+ * overwrites, so a boundary that retains either one reports the partition the compaction has
+ * reached now, not the flushed boundary.
+ *
+ * That matters because {@code SSTableRewriter.maybeReopenEarly} calls
+ * {@code moveStarts(reader.getLast())}. A key that runs ahead trims the originals past partitions
+ * the partial sstable cannot serve yet, and reads in that window return nothing until the
+ * compaction commits.
+ *
+ * BIG only. {@code BtiTableWriter.openEarly} takes first and last from the partition index, and its
+ * publication is deferred until the data, row index and partition index writers have all flushed
+ * past the recorded ends, so it does not reach this boundary at all.
+ */
+public class CursorEarlyOpenBoundaryTest extends DifferentialCompactionTester
+{
+ /** The smallest interval the config accepts. The scenario writes several times this. */
+ private static final int OPEN_INTERVAL_MIB = 1;
+
+ private SSTableFormat, ?> originalFormat;
+ private int originalInterval;
+
+ /** The format this scenario runs under. */
+ protected String formatName()
+ {
+ return "big";
+ }
+
+ /**
+ * True when the preemptive reopen must fire inside the output. BIG publishes synchronously from
+ * BigTableWriter.openEarly. BTI defers publication through PartitionIndexBuilder.buildPartial
+ * until the data, row index and partition index writers have all flushed past the recorded ends,
+ * and openFinalEarly cancels whatever is still pending, so zero is a legitimate outcome there.
+ */
+ protected boolean requiresMidStreamReopen()
+ {
+ return true;
+ }
+
+ @Before
+ public void selectBigAndShrinkOpenInterval()
+ {
+ originalFormat = DatabaseDescriptor.getSelectedSSTableFormat();
+ DatabaseDescriptor.setSelectedSSTableFormat(formatName());
+ originalInterval = DatabaseDescriptor.getSSTablePreemptiveOpenIntervalInMiB();
+ DatabaseDescriptor.setSSTablePreemptiveOpenIntervalInMiB(OPEN_INTERVAL_MIB);
+ }
+
+ @After
+ public void restoreFormatAndOpenInterval()
+ {
+ DatabaseDescriptor.setSSTablePreemptiveOpenIntervalInMiB(originalInterval);
+ DatabaseDescriptor.setSelectedSSTableFormat(originalFormat);
+ }
+
+ @Test
+ public void earlyOpenedBoundaryIsDetachedOnCursorPath() throws Throwable
+ {
+ assertBoundaryIsDetached(true);
+ }
+
+ /** The same expectation on the iterator path, so a failure above reads as a cursor defect. */
+ @Test
+ public void earlyOpenedBoundaryIsDetachedOnIteratorPath() throws Throwable
+ {
+ assertBoundaryIsDetached(false);
+ }
+
+ private void assertBoundaryIsDetached(boolean cursor) throws Throwable
+ {
+ ColumnFamilyStore cfs = severalMegabytesInTwoSSTables();
+
+ List boundaries = new ArrayList<>();
+ commitThroughFactory(cfs, cursor, singleOutputCapturing(boundaries));
+
+ long midStream = boundaries.stream().filter(b -> b.midStream).count();
+ if (requiresMidStreamReopen())
+ assertTrue("the preemptive reopen never fired inside the output, so this scenario asserted " +
+ "nothing; mid-stream reopens=" + midStream + " of " + boundaries.size() +
+ " early opens, interval MiB=" + OPEN_INTERVAL_MIB + ", output bytes=" +
+ cfs.getLiveSSTables().stream().mapToLong(SSTableReader::onDiskLength).sum(),
+ midStream > 0);
+
+ // Not assumeCursorSupportedFormatSelected: @Before forces BIG, which always supports the
+ // cursor path, so that guard can never fire. This pins what the scenario actually needs.
+ cfs.getLiveSSTables().forEach(DifferentialCompactionTester::assertOutputFormatIsSelected);
+
+ for (CapturedBoundary boundary : boundaries)
+ boundary.assertDetached();
+
+ assertCommittedBoundsAreTheExtremes(cfs);
+ }
+
+ /**
+ * The committed output must span the whole key range it was given.
+ *
+ * On BTI this is where a regressed key shows first: PartitionIndexBuilder holds firstKey and the
+ * previous key across addEntry to compute each separator, and writes both bounds into the
+ * Partitions.db footer. Handed a key the next partition overwrites, both bounds collapse onto the
+ * final partition and every separator is derived from a key that has since moved.
+ */
+ private void assertCommittedBoundsAreTheExtremes(ColumnFamilyStore cfs)
+ {
+ List written = everyKeyWritten(cfs);
+ DecoratedKey min = written.stream().min(Comparator.naturalOrder()).orElseThrow();
+ DecoratedKey max = written.stream().max(Comparator.naturalOrder()).orElseThrow();
+
+ for (SSTableReader output : cfs.getLiveSSTables())
+ {
+ assertEquals("the committed output's first key is not the lowest key written", min, output.getFirst());
+ assertEquals("the committed output's last key is not the highest key written", max, output.getLast());
+ }
+ }
+
+ /**
+ * Records what an early-opened reader carried as its {@code last} at the moment it was
+ * published, and holds the reader so the same field can be read again after the compaction.
+ */
+ private static final class CapturedBoundary
+ {
+ private final SSTableReader reader;
+ private final ByteBuffer keyAtPublication;
+ /** False for the reader openFinalEarly publishes at prepare time, which never reopens. */
+ private final boolean midStream;
+ /** Where the reader could find its own last key when it was published; negative is a miss. */
+ private final long positionOfLast;
+ /** The reader's other bound, which nothing else in the tree asserts. */
+ private final ByteBuffer firstAtPublication;
+ private final long positionOfFirst;
+
+ CapturedBoundary(SSTableReader reader, boolean midStream)
+ {
+ this.reader = reader;
+ this.midStream = midStream;
+ DecoratedKey last = reader.getLast();
+ this.keyAtPublication = ByteBufferUtil.clone(last.getKey());
+ // Taken here, one line before SSTableRewriter.maybeReopenEarly hands this same key to
+ // moveStarts. updateStats false so the probe does not warm the key cache.
+ this.positionOfLast = reader.getPosition(last, SSTableReader.Operator.EQ, false);
+ DecoratedKey first = reader.getFirst();
+ this.firstAtPublication = ByteBufferUtil.clone(first.getKey());
+ this.positionOfFirst = reader.getPosition(first, SSTableReader.Operator.EQ, false);
+ }
+
+ void assertDetached()
+ {
+ DecoratedKey last = reader.getLast();
+
+ // Absolute, not a consistency check: the partial sstable must be able to serve the key it
+ // claims as its last, because moveStarts trims the originals to exactly that key. A
+ // boundary that ran ahead sits past the reader's own indexLength override and misses here.
+ assertTrue("the early-opened sstable cannot find the key it published as its last, so " +
+ "moveStarts trimmed the originals to a key this sstable cannot serve; reads in " +
+ "that window return nothing until the compaction commits",
+ positionOfLast >= 0);
+
+ // ReadableBoundary also carries indexLength, dataLength, summaryCount and entriesLength,
+ // which become this reader's length overrides. first is not derived from the boundary at
+ // all, so it must be both stable and reachable under those overrides.
+ assertTrue("the early-opened sstable cannot find the key it published as its first",
+ positionOfFirst >= 0);
+ assertEquals("the early-opened sstable's first key changed after publication",
+ firstAtPublication, reader.getFirst().getKey());
+
+ // Catches a retained reusable Token: the key's own bytes and its token disagree.
+ assertEquals("the early-opened sstable's last key carries a token that does not belong to " +
+ "its own bytes, so it was built from a reusable token that has since moved; " +
+ "moveStarts trimmed the originals past partitions this sstable cannot serve",
+ reader.getPartitioner().getToken(last.getKey()), last.getToken());
+
+ // Catches a retained reusable key buffer, where bytes and token move together and so
+ // agree with each other while both describe the wrong partition.
+ assertEquals("the early-opened sstable's last key changed after publication, so the " +
+ "boundary retained the writer's reusable key rather than a copy",
+ keyAtPublication, last.getKey());
+ }
+ }
+
+ /** One output, with the transaction wrapped so every early-opened reader is captured. */
+ private static TaskFactory singleOutputCapturing(List boundaries)
+ {
+ return (cfs, txn, gcBefore) -> new CompactionTask(cfs, txn, gcBefore, false)
+ {
+ @Override
+ public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs,
+ Directories directories,
+ ILifecycleTransaction transaction,
+ Set nonExpiredSSTables)
+ {
+ return new DefaultCompactionWriter(cfs, directories,
+ new EarlyOpenCapturing(transaction, boundaries),
+ nonExpiredSSTables, false, 0);
+ }
+ };
+ }
+
+ // Both files must outgrow a buffer, not just Data.db. IndexSummaryBuilder.refreshReadableBoundary
+ // takes the lower of the boundaries below the data and the index sync positions, and a sync
+ // position only advances when that writer flushes a buffer. A table with short partition keys
+ // writes an Index.db smaller than one buffer however large Data.db grows, so the index sync
+ // position stays at zero, no boundary is ever readable, and openEarly publishes nothing on
+ // either path. Padding the partition key is what makes this scenario exercise the reopen at all.
+ /** Partitions per flushed sstable. Two rounds, so the compaction has two inputs. */
+ private static final int PARTITIONS_PER_ROUND = 4000;
+ /** Long enough that Index.db outgrows the index writer's buffer several times over. */
+ private static final int KEY_PADDING = 200;
+ private static final String KEY_PREFIX = "k".repeat(KEY_PADDING);
+ /** Long enough that Data.db outgrows the preemptive open interval several times over. */
+ private static final int VALUE_PADDING = 300;
+
+ /** Every partition key the fixture wrote, decorated for comparison. */
+ private List everyKeyWritten(ColumnFamilyStore cfs)
+ {
+ List keys = new ArrayList<>(PARTITIONS_PER_ROUND);
+ for (long pk = 0; pk < PARTITIONS_PER_ROUND; pk++)
+ keys.add(cfs.getPartitioner().decorateKey(ByteBufferUtil.bytes(KEY_PREFIX + pk)));
+ return keys;
+ }
+
+ private ColumnFamilyStore severalMegabytesInTwoSSTables() throws Throwable
+ {
+ createTable("CREATE TABLE %s (pk text, ck bigint, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH compression = {'enabled': 'false'}");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ String valuePadding = "x".repeat(VALUE_PADDING);
+ for (int round = 0; round < 2; round++)
+ {
+ for (long pk = 0; pk < PARTITIONS_PER_ROUND; pk++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)",
+ KEY_PREFIX + pk, 0L, valuePadding + round);
+ flush();
+ }
+ assertTrue("the fixture needs inputs", cfs.getLiveSSTables().size() >= 2);
+ return cfs;
+ }
+
+ /** Captures the readers published with {@code OpenReason.EARLY}; delegates everything else. */
+ private static final class EarlyOpenCapturing extends WrappedLifecycleTransaction
+ {
+ private final List boundaries;
+ private boolean preparing;
+
+ EarlyOpenCapturing(ILifecycleTransaction delegate, List boundaries)
+ {
+ super(delegate);
+ this.boundaries = boundaries;
+ }
+
+ private void capture(SSTableReader reader)
+ {
+ if (reader.openReason == SSTableReader.OpenReason.EARLY)
+ boundaries.add(new CapturedBoundary(reader, !preparing));
+ }
+
+ /**
+ * SSTableRewriter.switchWriter publishes one last EARLY reader here through openFinalEarly.
+ * Its last comes from SSTableWriter.setLast, which was already a copy before this change, so
+ * counting it as a reopen would let the scenario pass with no reopen at all.
+ */
+ @Override
+ public void prepareToCommit()
+ {
+ preparing = true;
+ super.prepareToCommit();
+ }
+
+ @Override
+ public void update(SSTableReader reader, boolean original)
+ {
+ capture(reader);
+ super.update(reader, original);
+ }
+
+ @Override
+ public void update(Collection readers, boolean original)
+ {
+ readers.forEach(this::capture);
+ super.update(readers, original);
+ }
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/CursorKeyCacheMigrationTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/CursorKeyCacheMigrationTest.java
new file mode 100644
index 000000000000..2f9f1145e410
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/CursorKeyCacheMigrationTest.java
@@ -0,0 +1,186 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+import java.util.ArrayList;
+import java.util.List;
+
+import org.junit.After;
+import org.junit.Before;
+import org.junit.Test;
+
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.compaction.CompactionTask;
+import org.apache.cassandra.db.DecoratedKey;
+import org.apache.cassandra.io.sstable.AbstractRowIndexEntry;
+import org.apache.cassandra.io.sstable.format.SSTableFormat;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+import org.apache.cassandra.io.sstable.keycache.KeyCacheSupport;
+import org.apache.cassandra.service.CacheService;
+import org.apache.cassandra.utils.ByteBufferUtil;
+
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertNotNull;
+import static org.junit.Assert.assertTrue;
+
+/**
+ * Covers {@code BigTableWriter.maybeCacheKey}, which carries a key that is hot in the originals into
+ * the compaction output's key cache. BIG format only; it is the only format with a key cache.
+ *
+ * The method reaches its {@code cachedKeys.put} only when an original already holds the key, which
+ * needs a point read before the compaction. No other test does that on the cursor path, so the put
+ * ran in no test at all: every {@code getCachedPosition} returned null and the loop fell through.
+ *
+ * That matters because the writer stores the key it is handed. The cursor's own key is a reusable
+ * instance the next partition overwrites, so a key stored without a copy would corrupt the map it
+ * is a key of, and then the real key cache that {@code BigTableWriter.openInternal} drains it into.
+ * Nothing about the sstable's own bounds would look wrong, so no other assertion in the tree
+ * catches it.
+ *
+ * The oracle here is per key and positional, not a count. A count survives a corrupted key: the map
+ * still holds one entry per put.
+ */
+public class CursorKeyCacheMigrationTest extends DifferentialCompactionTester
+{
+ /** Enough partitions that a corrupted key shows up as a miss rather than by luck. */
+ private static final int PARTITIONS = 200;
+
+ private SSTableFormat, ?> originalFormat;
+ private boolean originalMigrate;
+ private int originalColumnIndexCacheSize;
+
+ @Before
+ public void selectBigAndEnableMigration()
+ {
+ originalFormat = DatabaseDescriptor.getSelectedSSTableFormat();
+ DatabaseDescriptor.setSelectedSSTableFormat("big");
+ originalMigrate = DatabaseDescriptor.shouldMigrateKeycacheOnCompaction();
+ DatabaseDescriptor.setMigrateKeycacheOnCompaction(true);
+ // Forces a promoted row index, so the cached entry is the shallow kind the cursor path
+ // writes rather than the full one the iterator path builds.
+ // The getter returns bytes and the setter takes KiB; save the KiB one or the restore overflows.
+ originalColumnIndexCacheSize = DatabaseDescriptor.getColumnIndexCacheSizeInKiB();
+ DatabaseDescriptor.setColumnIndexCacheSize(0);
+ CacheService.instance.invalidateKeyCache();
+ }
+
+ @After
+ public void restore()
+ {
+ DatabaseDescriptor.setColumnIndexCacheSize(originalColumnIndexCacheSize);
+ DatabaseDescriptor.setMigrateKeycacheOnCompaction(originalMigrate);
+ DatabaseDescriptor.setSelectedSSTableFormat(originalFormat);
+ CacheService.instance.invalidateKeyCache();
+ }
+
+ @Test
+ public void hotKeysMigrateIntoTheOutputOnCursorPath() throws Throwable
+ {
+ assertHotKeysMigrate(true);
+ }
+
+ /** The same expectation on the iterator path, so a failure above reads as a cursor defect. */
+ @Test
+ public void hotKeysMigrateIntoTheOutputOnIteratorPath() throws Throwable
+ {
+ assertHotKeysMigrate(false);
+ }
+
+ private void assertHotKeysMigrate(boolean cursor) throws Throwable
+ {
+ ColumnFamilyStore cfs = twoSSTablesWithMultiBlockPartitions();
+
+ List hot = readEveryPartition(cfs);
+ assertTrue("the scenario must warm the key cache before it compacts, or maybeCacheKey " +
+ "never reaches its put", cachedAnywhere(cfs, hot) > 0);
+
+ // keepOriginals false: the originals must really be replaced, so the live set is the output.
+ commitThroughFactory(cfs, cursor,
+ (store, txn, gcBefore) -> new CompactionTask(store, txn, gcBefore, false));
+
+ List outputs = new ArrayList<>(cfs.getLiveSSTables());
+ assertEquals("expected one compaction output", 1, outputs.size());
+ SSTableReader output = outputs.get(0);
+ assertOutputFormatIsSelected(output);
+
+ int migrated = 0;
+ for (DecoratedKey key : hot)
+ {
+ AbstractRowIndexEntry cached = ((KeyCacheSupport>) output).getCachedPosition(key, false);
+ if (cached == null)
+ continue;
+ migrated++;
+
+ // The cached entry must name the same data position a fresh index lookup does. A key
+ // stored while it was still reusable lands under whatever bytes it later held, so its
+ // entry describes a different partition.
+ AbstractRowIndexEntry looked = output.getRowIndexEntry(key, SSTableReader.Operator.EQ);
+ assertNotNull("the migrated key " + key + " is not in the output's index at all", looked);
+ assertEquals("the key cache entry for " + key + " points at a different partition than " +
+ "the output's own index does", looked.position, cached.position);
+ }
+
+ assertTrue("no hot key reached the output's key cache, so BigTableWriter.maybeCacheKey " +
+ "never stored anything and this scenario proved nothing", migrated > 0);
+ }
+
+ /** How many of these keys any live sstable currently holds a cached position for. */
+ private static int cachedAnywhere(ColumnFamilyStore cfs, List keys)
+ {
+ int found = 0;
+ for (SSTableReader reader : cfs.getLiveSSTables())
+ if (reader instanceof KeyCacheSupport>)
+ for (DecoratedKey key : keys)
+ if (((KeyCacheSupport>) reader).getCachedPosition(key, false) != null)
+ found++;
+ return found;
+ }
+
+ /** Point-reads every partition, which is what puts its position in the originals' key cache. */
+ private List readEveryPartition(ColumnFamilyStore cfs) throws Throwable
+ {
+ List keys = new ArrayList<>(PARTITIONS);
+ for (int pk = 0; pk < PARTITIONS; pk++)
+ {
+ execute("SELECT * FROM %s WHERE pk = ?", pk);
+ keys.add(cfs.getPartitioner().decorateKey(ByteBufferUtil.bytes(pk)));
+ }
+ return keys;
+ }
+
+ private ColumnFamilyStore twoSSTablesWithMultiBlockPartitions() throws Throwable
+ {
+ createTable("CREATE TABLE %s (pk int, ck int, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH compression = {'enabled': 'false'}");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ String padding = "v".repeat(400);
+ for (int round = 0; round < 2; round++)
+ {
+ for (int pk = 0; pk < PARTITIONS; pk++)
+ for (int ck = 0; ck < 8; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", pk, ck + round * 8, padding);
+ flush();
+ }
+ assertTrue("the fixture needs inputs", cfs.getLiveSSTables().size() >= 2);
+ return cfs;
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/CursorPartialRangeGateTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/CursorPartialRangeGateTest.java
index 001d5fbc5d2d..0f7308152c9b 100644
--- a/test/unit/org/apache/cassandra/db/compaction/differential/CursorPartialRangeGateTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/CursorPartialRangeGateTest.java
@@ -35,7 +35,6 @@
import org.apache.cassandra.dht.Token;
import org.apache.cassandra.io.sstable.format.SSTableReader;
import org.apache.cassandra.io.sstable.format.SSTableReader.PartitionPositionBounds;
-import org.apache.cassandra.io.sstable.format.big.BigFormat;
import org.apache.cassandra.utils.FBUtilities;
import static org.junit.Assert.assertEquals;
@@ -90,13 +89,13 @@ private List> halfRange(ColumnFamilyStore cfs)
}
/**
- * Whether the gate can accept any compaction at all under the running configuration. The cursor
- * path writes the BIG format only, and {@code test/conf/latest_diff.yaml} selects BTI, so an
- * assertion that the gate opens has to read the format rather than assume it.
+ * Whether the gate can accept any compaction at all under the running configuration. Which
+ * formats the cursor path writes changes as the patch series lands, and the test configs
+ * differ in what they select, so the oracle asks the format rather than naming one.
*/
private static boolean cursorSupportsSelectedFormat()
{
- return DatabaseDescriptor.getSelectedSSTableFormat() instanceof BigFormat;
+ return DatabaseDescriptor.getSelectedSSTableFormat().supportsCursorCompaction();
}
private boolean isSupportedOver(ColumnFamilyStore cfs, List> ranges) throws Exception
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/CursorSupportMatrixTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/CursorSupportMatrixTest.java
index e3bfebd45570..b9b54b5f82f6 100644
--- a/test/unit/org/apache/cassandra/db/compaction/differential/CursorSupportMatrixTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/CursorSupportMatrixTest.java
@@ -24,14 +24,18 @@
import org.junit.Assume;
import org.junit.Test;
+import org.mockito.Mockito;
+import org.apache.cassandra.config.DatabaseDescriptor;
import org.apache.cassandra.cql3.CQLTester;
import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.compaction.AbstractCompactionStrategy;
import org.apache.cassandra.db.compaction.CompactionController;
import org.apache.cassandra.db.compaction.CursorCompactor;
+import org.apache.cassandra.io.sstable.format.SSTableFormat;
import org.apache.cassandra.io.sstable.format.SSTableReader;
import org.apache.cassandra.io.sstable.format.big.BigFormat;
+import org.apache.cassandra.io.sstable.format.bti.BtiFormat;
import org.apache.cassandra.notifications.INotificationConsumer;
import org.apache.cassandra.notifications.SSTableListChangedNotification;
import org.apache.cassandra.schema.ColumnMetadata;
@@ -146,6 +150,111 @@ public void vectorAndDurationSupported()
"PRIMARY KEY (pk, ck))");
}
+ /**
+ * BTI output is inside the supported surface, asserted through the gate production calls.
+ *
+ * {@link #assertSupported} cannot carry this claim: it reaches only
+ * {@code CursorCompactor.unsupportedMetadata}, which reads {@link TableMetadata} and never the
+ * selected format, so the assertion would read the same with the format selection deleted. The
+ * format gate is in {@code CursorCompactor.isSupported}, which {@link #isSupportedNow} drives.
+ */
+ @Test
+ public void btiFormatSupported() throws Exception
+ {
+ SSTableFormat, ?> original = DatabaseDescriptor.getSelectedSSTableFormat();
+ DatabaseDescriptor.setSelectedSSTableFormat(BtiFormat.NAME);
+ try
+ {
+ assertTrue("the BTI format must report cursor compaction support",
+ DatabaseDescriptor.getSelectedSSTableFormat().supportsCursorCompaction());
+
+ ColumnFamilyStore cfs =
+ twoSSTableTable("CREATE TABLE %s (pk bigint, ck bigint, m map, v text, " +
+ "PRIMARY KEY (pk, ck))",
+ "INSERT INTO %s (pk, ck, m, v) VALUES (1, 1, {'a': 1}, 'x')",
+ "INSERT INTO %s (pk, ck, m, v) VALUES (1, 2, {'b': 2}, 'y')");
+
+ // the inputs have to be in the format under test, or the gate would be reading a
+ // selection nothing in this table reflects
+ for (SSTableReader reader : cfs.getLiveSSTables())
+ assertTrue("expected BTI input sstables, got " + reader.descriptor.version.format.name(),
+ BtiFormat.is(reader.descriptor.version.format));
+
+ assertTrue("cursor compaction must accept a BTI table", isSupportedNow(cfs));
+ }
+ finally
+ {
+ DatabaseDescriptor.setSelectedSSTableFormat(original);
+ }
+ }
+
+ /**
+ * The negative half of the format gate: a selected format that does not support cursor
+ * compaction is refused.
+ *
+ * No such format exists in tree. BIG and BTI both override
+ * {@link SSTableFormat#supportsCursorCompaction()} to return true, so the only way into the
+ * branch is the interface default at {@code SSTableFormat:60}, which is false and is what
+ * gates a format added later. The stand-in below is that default and nothing else: every
+ * other method is left at Mockito's default, and it is selected only across the
+ * {@code isSupported} call, after the inputs and the scanners have been built by a real format.
+ *
+ * What this cannot see: {@code supportsCursorCompaction()} returning false is not
+ * distinguishable here from Mockito's own default for a boolean, because it is the sole
+ * default method on the interface. The assertion is about the gate's branch, not about where
+ * the false came from.
+ */
+ @Test
+ public void formatWithoutCursorSupportUnsupported() throws Exception
+ {
+ Assume.assumeTrue("requires the BIG sstable format", BigFormat.isSelected());
+
+ ColumnFamilyStore cfs =
+ twoSSTableTable("CREATE TABLE %s (pk bigint, ck bigint, v text, PRIMARY KEY (pk, ck))",
+ "INSERT INTO %s (pk, ck, v) VALUES (1, 1, 'x')",
+ "INSERT INTO %s (pk, ck, v) VALUES (1, 2, 'y')");
+
+ // control: the same table and the same two sstables under the real selected format are
+ // supported, so the rejection below is attributable to the format alone
+ assertTrue("expected a plain two-sstable table to be cursor-supported", isSupportedNow(cfs));
+
+ SSTableFormat, ?> original = DatabaseDescriptor.getSelectedSSTableFormat();
+ SSTableFormat, ?> noCursorSupport = Mockito.mock(SSTableFormat.class, Mockito.CALLS_REAL_METHODS);
+ assertFalse("the stand-in must report no cursor compaction support, or the gate below is " +
+ "not the thing being observed",
+ noCursorSupport.supportsCursorCompaction());
+
+ DatabaseDescriptor.setSelectedSSTableFormat(noCursorSupport);
+ try
+ {
+ assertFalse("cursor compaction must refuse a table whose selected output format does " +
+ "not support it",
+ isSupportedNow(cfs));
+ }
+ finally
+ {
+ DatabaseDescriptor.setSelectedSSTableFormat(original);
+ }
+
+ // the gate reopens once the real format is back, so the helper is not hardwired to one answer
+ assertTrue("expected the table to be cursor-supported again under the real format",
+ isSupportedNow(cfs));
+ }
+
+ /** Creates {@code ddl} with auto-compaction off and flushes each insert into its own sstable. */
+ private ColumnFamilyStore twoSSTableTable(String ddl, String firstInsert, String secondInsert)
+ {
+ createTable(ddl);
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+ execute(firstInsert);
+ flush();
+ execute(secondInsert);
+ flush();
+ assertEquals("expected one sstable per flush", 2, cfs.getLiveSSTables().size());
+ return cfs;
+ }
+
/** Counter columns are a planned gap in the supported surface, not a permanent limit. */
@Test
public void countersUnsupported()
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/DifferentialCompactionTester.java b/test/unit/org/apache/cassandra/db/compaction/differential/DifferentialCompactionTester.java
index 6f4c0be0361f..25e1cf476d1e 100644
--- a/test/unit/org/apache/cassandra/db/compaction/differential/DifferentialCompactionTester.java
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/DifferentialCompactionTester.java
@@ -24,6 +24,7 @@
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
+import java.util.ArrayDeque;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.HashSet;
@@ -45,22 +46,34 @@
import org.apache.cassandra.config.DatabaseDescriptor;
import org.apache.cassandra.cql3.CQLTester;
import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.DecoratedKey;
+import org.apache.cassandra.db.DeletionTime;
+import org.apache.cassandra.db.Slice;
+import org.apache.cassandra.db.Slices;
import org.apache.cassandra.db.compaction.AbstractCompactionStrategy;
import org.apache.cassandra.db.compaction.ActiveCompactionsTracker;
import org.apache.cassandra.db.compaction.CompactionController;
import org.apache.cassandra.db.compaction.CompactionPipelineCounts;
+import org.apache.cassandra.db.compaction.CompactionInfo;
import org.apache.cassandra.db.compaction.CompactionTask;
import org.apache.cassandra.db.compaction.CursorCompactor;
import org.apache.cassandra.db.compaction.OperationType;
+import org.apache.cassandra.db.filter.ColumnFilter;
import org.apache.cassandra.db.lifecycle.LifecycleTransaction;
import org.apache.cassandra.db.rows.Cell;
+import org.apache.cassandra.db.rows.Row;
+import org.apache.cassandra.db.rows.Unfiltered;
+import org.apache.cassandra.db.rows.UnfilteredRowIterator;
+import org.apache.cassandra.io.sstable.AbstractRowIndexEntry;
import org.apache.cassandra.io.sstable.Component;
import org.apache.cassandra.io.sstable.Descriptor;
import org.apache.cassandra.io.sstable.ISSTableScanner;
import org.apache.cassandra.io.sstable.IVerifier;
+import org.apache.cassandra.io.sstable.SSTableReadsListener;
+import org.apache.cassandra.io.sstable.format.SSTableFormat;
import org.apache.cassandra.io.sstable.format.SSTableReader;
-import org.apache.cassandra.io.sstable.format.big.BigFormat;
import org.apache.cassandra.io.sstable.metadata.StatsMetadata;
+import org.apache.cassandra.schema.TableMetadata;
import org.apache.cassandra.tools.JsonTransformer;
import org.apache.cassandra.tools.Util;
import org.apache.cassandra.utils.FBUtilities;
@@ -112,6 +125,30 @@ public abstract class DifferentialCompactionTester extends CQLTester
private static final boolean KEEP_SCRATCH_ON_FAILURE =
CassandraRelevantProperties.TEST_DIFFERENTIAL_KEEP_SCRATCH_ON_FAILURE.getBoolean();
+ /**
+ * Whether {@link #capture} reads every captured output back through single-row slices; see
+ * {@link #assertEveryRowReadableThroughASlice}. Defaults ON, because the slice path is the only
+ * reader that opens a BTI row trie and nothing else in this suite touches one. Turn it off for a
+ * local run that only wants the byte comparison; a CI run that has it off proves less than the
+ * suite claims.
+ */
+ private static final boolean SLICE_READBACK =
+ CassandraRelevantProperties.TEST_DIFFERENTIAL_SLICE_READBACK.getBoolean();
+
+ /**
+ * Ceiling on how many rows of ONE partition {@link #assertEveryRowReadableThroughASlice} probes.
+ * A probe is two seeks (forward and reverse) plus a reader open, the cost is linear in row count,
+ * and it is paid once per captured output — four times over a cross-generation scenario. The
+ * default clears every scenario in the tree today (the widest single partition is the 4000-row
+ * one in {@code EdgeCaseDifferentialCompactionTest}), so nothing is currently sampled down.
+ *
+ * The cap is per PARTITION, so total cost still scales with partition count. That is affordable
+ * only because no scenario outside scale mode has both many partitions and many rows per
+ * partition; the two multi-GB burn scenarios are in scale mode and skip the read-back entirely.
+ */
+ private static final int SLICE_READBACK_MAX_ROWS_PER_PARTITION =
+ CassandraRelevantProperties.TEST_DIFFERENTIAL_SLICE_READBACK_MAX_ROWS.getInt();
+
// sstabledump renders "expired" from the WALL CLOCK, not from the fixed nowInSec above, so
// the two paths' captures can differ on it. Every capture normalizes it away; see capture().
private static final Pattern EXPIRED_FLAG =
@@ -223,7 +260,14 @@ protected static long maxTombstoneLocalDeletionTime(Iterable ssta
return max;
}
- /** Creates the CompactionTask for one differential run. MUST honor keepOriginals=true. */
+ /**
+ * Creates the CompactionTask for one differential run. MUST honor keepOriginals=true.
+ *
+ * A factory that subclasses a CompactionTask to override getCompactionAwareWriter cannot name its
+ * own flag {@code keepOriginals}: inside the subclass that name resolves to CompactionTask's
+ * inherited field, which most task classes leave false, and the harness then loses the inputs it
+ * needs to restore. The failure reads as "input sstable lost during compaction".
+ */
public interface TaskFactory
{
CompactionTask create(ColumnFamilyStore cfs, LifecycleTransaction txn, long gcBefore);
@@ -405,6 +449,39 @@ protected CapturedOutput assertCursorMatchesIteratorAcrossGenerations(ColumnFami
return gen1;
}
+ /**
+ * Commits one compaction over the whole live set through the given task factory and path,
+ * WITHOUT restore: the live set genuinely becomes the outputs.
+ *
+ * {@link #assertCursorMatchesIterator} restores the originals, so a scenario asserting on a
+ * committed output — its level, its bounds, its key cache — cannot use it. The factory must
+ * build its writer with keepOriginals false here.
+ */
+ protected void commitThroughFactory(ColumnFamilyStore cfs, boolean cursor, TaskFactory taskFactory) throws Exception
+ {
+ commitThroughFactory(cfs, cursor, taskFactory, ActiveCompactionsTracker.NOOP);
+ }
+
+ /**
+ * As above, with a tracker of the caller's choosing, for a scenario that asserts on what
+ * {@link CompactionInfo} reports while the compaction runs rather than on its output.
+ */
+ protected void commitThroughFactory(ColumnFamilyStore cfs, boolean cursor, TaskFactory taskFactory,
+ ActiveCompactionsTracker tracker) throws Exception
+ {
+ DatabaseDescriptor.setCursorCompactionEnabled(cursor);
+ long gcBefore = cfs.getDefaultGcBefore(FBUtilities.nowInSeconds());
+ Set inputs = cfs.getLiveSSTables();
+ assertFalse("scenario produced no input sstables", inputs.isEmpty());
+ if (cursor)
+ assertCursorPathWillRun(cfs, inputs, gcBefore);
+ LifecycleTransaction txn = cfs.getTracker().tryModify(inputs, OperationType.COMPACTION);
+ assertNotNull("unable to mark inputs compacting for commit", txn);
+ CompactionPipelineCounts before = CompactionPipelineCounts.mark();
+ taskFactory.create(cfs, txn, gcBefore).execute(tracker);
+ CompactionPipelineCounts.assertPipelineRan(cursor, before);
+ }
+
/**
* Commits one compaction over the given inputs through the selected path WITHOUT restore:
* the live set genuinely becomes the outputs. Used by the cross-generation rung so the
@@ -554,7 +631,11 @@ else if (inputDescs.contains(reader.descriptor))
*/
protected void assertCursorPathWillRun(ColumnFamilyStore cfs, Set inputs, long gcBefore) throws Exception
{
- assumeBigFormatSelected();
+ // A format the cursor path cannot write is an unsupported configuration, not a defect:
+ // skip instead of failing the assertion below.
+ Assume.assumeTrue("cursor compaction cannot write the selected sstable format; selected=" +
+ DatabaseDescriptor.getSelectedSSTableFormat().name(),
+ DatabaseDescriptor.getSelectedSSTableFormat().supportsCursorCompaction());
try (CompactionController controller = new CompactionController(cfs, inputs, gcBefore);
AbstractCompactionStrategy.ScannerList scanners =
cfs.getCompactionStrategyManager().getScanners(new ArrayList<>(inputs), null))
@@ -567,21 +648,18 @@ protected void assertCursorPathWillRun(ColumnFamilyStore cfs, Set
}
/**
- * Cursor compaction only supports BIG output (CursorCompactor.isSupported). Under a non-BIG
- * format — `ant test-latest` selects BTI — every scenario in this suite would fail for a reason
- * that is not a defect. Skip instead, and keep the supportability assertion for every other
- * unsupported-ness reason so the iterator-vs-iterator trap still fires.
+ * Skips a scenario unless the BIG sstable format is selected; `ant test-latest` selects BTI.
*
* Separate from {@link #assertCursorPathWillRun} so a scenario that drives the harness from
- * inside a callback can raise it OUTSIDE that callback. JUnit decides skip-versus-fail on the
- * type it receives, so an AssumptionViolatedException crossing a broad catch that rewraps — as
- * Harry's TestHelper.withRandom does — arrives as a failure.
+ * inside a callback can raise the assumption OUTSIDE that callback. JUnit decides
+ * skip-versus-fail on the type it receives, so an AssumptionViolatedException crossing a broad
+ * catch that rewraps, as Harry's TestHelper.withRandom does, arrives as a failure.
*/
- protected static void assumeBigFormatSelected()
+ protected static void assumeCursorSupportedFormatSelected()
{
- Assume.assumeTrue("cursor compaction requires the BIG sstable format; selected=" +
+ Assume.assumeTrue("cursor compaction does not support the selected sstable format; selected=" +
DatabaseDescriptor.getSelectedSSTableFormat().name(),
- BigFormat.isSelected());
+ DatabaseDescriptor.getSelectedSSTableFormat().supportsCursorCompaction());
}
private static String listDataDir(Descriptor desc)
@@ -598,15 +676,362 @@ private static String listDataDir(Descriptor desc)
}
}
+ /**
+ * Asserts this compaction output was written in the sstable format the JVM currently has selected.
+ *
+ * The FORMAT counterpart of {@link CompactionPipelineCounts#assertPipelineRan}, which closes the
+ * same silent-fallback trap for the PIPELINE. A {@code Bti*} subclass selects BTI in
+ * {@code @Before}, restores in {@code @After}, and asserts nothing about the format in between.
+ * Meanwhile the byte comparison in {@link #assertEquivalentOutputs} walks
+ * {@code descriptor.discoverComponents()} — whatever components happen to be on disk. Under BIG
+ * that is {@code Index.db} and {@code Summary.db}, which compare equal and pass, so if the
+ * selection ever stopped taking effect every {@code Bti*} class would quietly become a duplicate
+ * of its base class and stay green. Nothing in the suite would say so.
+ *
+ * Compares the format NAME so a failure names both formats rather than printing two object
+ * identities.
+ *
+ * What it cannot see: that the writer for the selected format produced the right CONTENT. It
+ * pins only which writer ran. The byte comparison and
+ * {@link #assertEveryRowReadableThroughASlice} carry that.
+ */
+ protected static void assertOutputFormatIsSelected(SSTableReader sstable)
+ {
+ SSTableFormat, ?> selected = DatabaseDescriptor.getSelectedSSTableFormat();
+ SSTableFormat, ?> written = sstable.descriptor.getFormat();
+ assertEquals("compaction output " + sstable.descriptor + " was written in the '" + written.name() +
+ "' format while '" + selected.name() + "' is selected: this scenario is not testing the " +
+ "format it claims, and the byte comparison below would compare that other format's " +
+ "components and pass",
+ selected.name(), written.name());
+ }
+
+ /**
+ * ABSOLUTE, not differential: opens a single-row slice for every row of every partition and
+ * asserts the row that comes back is the row a plain sequential walk of the same sstable
+ * returned. Returns how many partitions actually carried a promoted row index.
+ *
+ * Byte identity between the two compaction paths says they agree on what to write. It cannot say
+ * the index they wrote ROUTES A SEEK to the right place: a row trie with wrong separators is
+ * written identically by both paths and compares equal, and the data is then unreachable. Nothing
+ * else in the tree reads a cursor-written BTI row trie. {@code sstable.getScanner()} walks
+ * {@code Partitions.db} and {@code Data.db} end to end; {@code BtiTableVerifier.verifyPartition}
+ * is a no-op; {@code SortedTableVerifier} opens partitions with {@code SSTableIdentityIterator},
+ * a straight data-file read. The trie is read only from
+ * {@code bti.SSTableIterator.ForwardIndexedReader.setForSlice} and from
+ * {@code bti.SSTableReversedIterator.ReverseIndexedReader.setForSlice}, and only on a real slice.
+ *
+ * Both directions are probed because they take different routes through the same trie: forward
+ * calls {@code RowIndexReader.separatorFloor}, reverse drives a {@code RowIndexReverseIterator}
+ * and then walks blocks backwards.
+ *
+ * One slice per reader instance is deliberate. {@code ForwardIndexedReader.setForSlice} seeks
+ * only when the target is ahead of the current file pointer, so packing many slices into one
+ * iterator would let later slices ride on wherever the first one landed and stop exercising the
+ * trie at all.
+ *
+ * The reference walk is {@code getScanner()} — {@code SSTableSimpleScanner} over
+ * {@code SSTableIdentityIterator}, a sequential data-file read that consults no index and applies
+ * no column filter. That independence is the point: a reference read through
+ * {@code partitionIterator} would itself seek through the trie for its first block.
+ *
+ * WHAT THIS CANNOT SEE:
+ *
+ * - A partition with fewer than two index blocks has no trie. {@code isIndexed()} is
+ * {@code blockCount() > 1}, and below that the reader falls back to a plain forward scan of
+ * the partition, so those partitions prove nothing here beyond retrievability. The return
+ * value is how many partitions were actually indexed; zero means this call said nothing
+ * whatsoever about any index.
+ * - Cell content, when the sstable header carries columns the schema no longer has. The
+ * probe reads with {@code ColumnFilter.all(metadata)} and the reference reads unfiltered, so
+ * on a dropped-column sstable the two legitimately disagree about cells. There the comparison
+ * drops to clustering, liveness and row deletion — enough for routing, which is what this
+ * assertion exists for. The byte comparison and the JSON dump cover the cells.
+ * - The middle of a partition wider than {@link #SLICE_READBACK_MAX_ROWS_PER_PARTITION}.
+ * The head and the TAIL are always kept, because the tail block is where
+ * {@code BtiFormatPartitionWriter.finish}'s "the last row may have fallen on a boundary
+ * already" decision lands.
+ * - An encoding both sides get wrong. This reads back through the same deserializer the
+ * writer serialized with. It pins ROUTING, not encoding.
+ *
+ *
+ * @return how many partitions carried a promoted row index, i.e. how many of these probes opened
+ * a trie at all. A scenario that means to test the index should assert this is non-zero.
+ */
+ protected int assertEveryRowReadableThroughASlice(SSTableReader sstable)
+ {
+ TableMetadata metadata = sstable.metadata();
+ // No clustering columns: one row per partition, never indexable, and Slice.make over an empty
+ // clustering degenerates to the whole partition. There is no seek to route.
+ if (metadata.comparator.size() == 0 || SLICE_READBACK_MAX_ROWS_PER_PARTITION <= 0)
+ return 0;
+
+ ColumnFilter fetchAll = ColumnFilter.all(metadata);
+ // Cells are only comparable when the probe's filter fetches exactly what the sequential read
+ // deserializes. A dropped column lives on in the sstable header but not in the schema.
+ boolean cellsComparable = sstable.header.columns().equals(metadata.regularAndStaticColumns());
+ int headCap = (SLICE_READBACK_MAX_ROWS_PER_PARTITION + 1) / 2;
+ int tailCap = SLICE_READBACK_MAX_ROWS_PER_PARTITION / 2;
+ int granularity = DatabaseDescriptor.getColumnIndexSize(-1);
+ int indexedPartitions = 0;
+
+ PendingPartition pending = new PendingPartition();
+
+ try (ISSTableScanner scanner = sstable.getScanner())
+ {
+ while (scanner.hasNext())
+ {
+ List probes = new ArrayList<>();
+ int unfiltereds;
+ DecoratedKey key;
+ DeletionTime partitionDeletion;
+ try (UnfilteredRowIterator partition = scanner.next())
+ {
+ key = partition.partitionKey();
+ partitionDeletion = partition.partitionLevelDeletion();
+ unfiltereds = collectProbeRows(partition, probes, headCap, tailCap);
+ }
+
+ AbstractRowIndexEntry entry = sstable.getRowIndexEntry(key, SSTableReader.Operator.EQ);
+ if (entry == null)
+ throw new AssertionError("a partition the sequential walk returned has no index entry: " +
+ key + " in " + sstable.descriptor);
+ if (entry.blockCount() > 1)
+ indexedPartitions++;
+
+ pending.advance(sstable, key, entry, unfiltereds, granularity);
+
+ assertPartitionDeletionReadableFromIndexEntry(sstable, key, entry, partitionDeletion);
+ assertEveryProbeReturnsExactly(sstable, metadata, fetchAll, key, probes, cellsComparable);
+ }
+ }
+ pending.finish(sstable, granularity);
+ return indexedPartitions;
+ }
+
+ /**
+ * A partition's serialized length is only known once the NEXT partition's start is read, so each
+ * partition's block-count bound is checked one iteration late, and the last one after the walk.
+ */
+ private static final class PendingPartition
+ {
+ private DecoratedKey key;
+ private long position = -1;
+ private int blockCount;
+ private int unfiltereds;
+
+ /** Bounds the partition held here against the next one's start, then holds the next one. */
+ void advance(SSTableReader sstable, DecoratedKey nextKey, AbstractRowIndexEntry nextEntry,
+ int nextUnfiltereds, int granularity)
+ {
+ if (key != null && nextEntry.position > position)
+ assertBlockCountWithinBounds(sstable, key, blockCount, unfiltereds,
+ nextEntry.position - position, granularity);
+ key = nextKey;
+ position = nextEntry.position;
+ blockCount = nextEntry.blockCount();
+ unfiltereds = nextUnfiltereds;
+ }
+
+ /** The last partition's bound, measured against the end of the file. */
+ void finish(SSTableReader sstable, int granularity)
+ {
+ if (key != null && sstable.uncompressedLength() > position)
+ assertBlockCountWithinBounds(sstable, key, blockCount, unfiltereds,
+ sstable.uncompressedLength() - position, granularity);
+ }
+ }
+
+ /** Every probe row of one partition, seeked in both directions. */
+ private static void assertEveryProbeReturnsExactly(SSTableReader sstable, TableMetadata metadata,
+ ColumnFilter fetchAll, DecoratedKey key,
+ List probes, boolean cellsComparable)
+ {
+ for (Row expected : probes)
+ {
+ assertSliceReturnsExactly(sstable, metadata, fetchAll, key, expected, false, cellsComparable);
+ assertSliceReturnsExactly(sstable, metadata, fetchAll, key, expected, true, cellsComparable);
+ }
+ }
+
+ /**
+ * The rows one partition is probed with: the first {@code headCap}, then the last {@code tailCap},
+ * appended to {@code probes} in partition order.
+ *
+ * @return how many unfiltereds the partition held, markers included
+ */
+ private static int collectProbeRows(UnfilteredRowIterator partition, List probes,
+ int headCap, int tailCap)
+ {
+ ArrayDeque tail = new ArrayDeque<>();
+ int unfiltereds = 0;
+ while (partition.hasNext())
+ {
+ Unfiltered unfiltered = partition.next();
+ unfiltereds++;
+ if (!unfiltered.isRow())
+ continue;
+ Row row = (Row) unfiltered;
+ if (probes.size() < headCap)
+ probes.add(row);
+ else if (tailCap > 0)
+ {
+ tail.addLast(row);
+ if (tail.size() > tailCap)
+ tail.removeFirst();
+ }
+ }
+ probes.addAll(tail);
+ return unfiltereds;
+ }
+
+ /**
+ * One single-row slice, in one direction, through the reader's index. Asserts exactly one row
+ * comes back and that it is {@code expected}.
+ *
+ * Range tombstone markers are skipped: a slice bounded to one clustering still emits the open and
+ * close markers of any range covering it, and those are not what this is about.
+ *
+ * Every failure message is built only once something has already failed. This runs once per row
+ * per direction over the whole corpus, and rendering a clustering decodes its values, so an eager
+ * message would cost more than the seek it describes.
+ */
+ private static void assertSliceReturnsExactly(SSTableReader sstable,
+ TableMetadata metadata,
+ ColumnFilter fetchAll,
+ DecoratedKey key,
+ Row expected,
+ boolean reversed,
+ boolean cellsComparable)
+ {
+ Slices slices = Slices.with(metadata.comparator, Slice.make(expected.clustering()));
+ try (UnfilteredRowIterator probe = sstable.rowIterator(key, slices, fetchAll, reversed,
+ SSTableReadsListener.NOOP_LISTENER))
+ {
+ Row found = null;
+ int rows = 0;
+ while (probe.hasNext())
+ {
+ Unfiltered unfiltered = probe.next();
+ if (!unfiltered.isRow())
+ continue;
+ rows++;
+ found = (Row) unfiltered;
+ }
+
+ if (rows == 1 && (cellsComparable ? expected.equals(found) : sameRowIdentity(expected, found)))
+ return;
+
+ String where = (reversed ? "reverse" : "forward") + " slice of " +
+ expected.clustering().toString(metadata) + " in partition " + key +
+ " of " + sstable.descriptor;
+ if (rows != 1)
+ fail("the index routed a " + where + " to " + rows + " rows; a single-clustering slice " +
+ "must return exactly one");
+ fail("the index routed a " + where + " to the wrong row" +
+ (cellsComparable ? "" : " (this sstable's header carries columns the schema does not, " +
+ "so only clustering, liveness and row deletion are compared)") +
+ "\n sequential walk: " + expected.toString(metadata, true) +
+ "\n slice returned: " + found.toString(metadata, true));
+ }
+ }
+
+ /** Clustering, primary key liveness and row deletion — everything a misrouted seek would change. */
+ private static boolean sameRowIdentity(Row expected, Row found)
+ {
+ return expected.clustering().equals(found.clustering())
+ && expected.primaryKeyLivenessInfo().equals(found.primaryKeyLivenessInfo())
+ && expected.deletion().equals(found.deletion());
+ }
+
+ /**
+ * Asserts the partition-level deletion the INDEX ENTRY carries matches the one in the data file.
+ *
+ * {@code AbstractSSTableIterator} skips the seek to the partition header when the entry is
+ * indexed and the column filter fetches no statics, and takes {@code indexEntry.deletionTime()}
+ * instead. So a probe with {@link ColumnFilter#NONE} reads the deletion out of
+ * {@code TrieIndexEntry} (or out of BIG's promoted entry) rather than out of {@code Data.db},
+ * and the sequential walk gives the data file's own copy to compare it against. Nothing else in
+ * the suite reads that field back.
+ *
+ * It only means something for an indexed partition, and it is dormant while no scenario writes a
+ * non-LIVE partition deletion on one: LIVE compared against LIVE passes for free.
+ */
+ private static void assertPartitionDeletionReadableFromIndexEntry(SSTableReader sstable,
+ DecoratedKey key,
+ AbstractRowIndexEntry entry,
+ DeletionTime fromDataFile)
+ {
+ if (entry.blockCount() <= 1)
+ return;
+ try (UnfilteredRowIterator probe = sstable.rowIterator(key, Slices.ALL, ColumnFilter.NONE, false,
+ SSTableReadsListener.NOOP_LISTENER))
+ {
+ assertEquals("the index entry's partition-level deletion differs from the data file's for " +
+ key + " in " + sstable.descriptor,
+ fromDataFile, probe.partitionLevelDeletion());
+ }
+ }
+
+ /**
+ * Bounds {@code blockCount()} by what the partition's own block structure implies.
+ *
+ * {@code BtiFormatPartitionWriter.addUnfiltered} cuts a block the moment the current one reaches
+ * {@code column_index_size}, and {@code finish} adds a tail block only if a cut already happened.
+ * So of {@code n} blocks at least {@code n - 1} are cut blocks, each at least one granularity of
+ * serialized data, which puts the partition's length at or above {@code (n - 1) * granularity}.
+ * Every block also begins at its own unfiltered — the static row is part of the partition header
+ * and never opens one — so {@code n} can never exceed the unfiltered count. Finally, a one-block
+ * index is not promoted at all: BTI's {@code finish} returns a {@code -1} trie root and
+ * {@code TrieIndexEntry.create} maps that to zero, and BIG's {@code RowIndexEntry.create}
+ * promotes only above one block. A block count of exactly 1 is unreadable by construction.
+ *
+ * These bound the count from ABOVE only. An index with too FEW blocks — a writer that stopped
+ * cutting halfway down a partition — satisfies all three, because the last block may be any
+ * length. Nothing outside the format can compute the exact count: that needs the serialized size
+ * of each individual unfiltered, which no reader exposes. The slice read-back is what catches an
+ * under-split index, by landing on the wrong row.
+ *
+ * {@code partitionLength} is measured from this partition's data-file position to the next one's
+ * (or to the end of the data file), so it includes the partition header and the end-of-partition
+ * marker. That only ever makes the bound looser.
+ */
+ private static void assertBlockCountWithinBounds(SSTableReader sstable,
+ DecoratedKey key,
+ int blockCount,
+ int unfiltereds,
+ long partitionLength,
+ int granularity)
+ {
+ // granularity is -1 when column_index_size is unset in the yaml, in which case each format
+ // falls back to its own default and the length bound cannot be stated.
+ boolean lengthBoundHolds = granularity <= 0 || blockCount < 2
+ || partitionLength >= (long) (blockCount - 1) * granularity;
+ if (blockCount != 1 && blockCount <= unfiltereds && lengthBoundHolds)
+ return;
+
+ String where = " for partition " + key + " in " + sstable.descriptor;
+ assertFalse("a promoted row index of exactly one block cannot be written: the writer drops it" + where,
+ blockCount == 1);
+ assertTrue("row index claims " + blockCount + " blocks but the partition holds only " + unfiltereds +
+ " unfiltereds, and every block begins at its own unfiltered" + where,
+ blockCount <= unfiltereds);
+ assertTrue("row index claims " + blockCount + " blocks, so at least " + (blockCount - 1) +
+ " of them were cut at the " + granularity + "-byte column_index_size, but the " +
+ "partition is only " + partitionLength + " bytes long" + where,
+ lengthBoundHolds);
+ }
+
private CapturedSSTable capture(ColumnFamilyStore cfs, SSTableReader sstable, Path dir) throws IOException
{
- // 1. structural verification of the output. In scale mode the verifier's debug
- // stream must be silenced: the extended index walk debug-logs EVERY index block
- // (~560K lines for a >2GiB partition), and ant's junit formatter buffers all test
- // output in memory — the log volume, not the verification, OOMs the fork.
- OutputHandler verifyOutput = scaleCapture()
- ? new OutputHandler.LogOutput() { @Override public void debug(String msg) {} }
- : new OutputHandler.LogOutput();
+ // 1. the output really is in the format this scenario selected
+ assertOutputFormatIsSelected(sstable);
+
+ // 2. structural verification of the output. The verifier's debug stream is always
+ // silenced: the extended index walk debug-logs EVERY index block, and ant's junit
+ // formatter buffers all test output in memory, so the log volume OOMs the fork.
+ // Verification is unaffected; a failure arrives as an exception, not as narration.
+ OutputHandler verifyOutput = new OutputHandler.LogOutput() { @Override public void debug(String msg) {} };
try (IVerifier verifier = sstable.getVerifier(cfs, verifyOutput, false,
IVerifier.options().invokeDiskFailurePolicy(true)
.extendedVerification(true).build()))
@@ -614,7 +1039,13 @@ private CapturedSSTable capture(ColumnFamilyStore cfs, SSTableReader sstable, Pa
verifier.verify();
}
- // 2. canonical logical dump
+ // 3. every row is retrievable through a real slice, i.e. the index routes seeks correctly.
+ // Skipped in scale mode for the same reason the verifier is muted there: those scenarios hold
+ // millions of rows in one partition, and a seek per row is not affordable.
+ if (SLICE_READBACK && !scaleCapture())
+ assertEveryRowReadableThroughASlice(sstable);
+
+ // 4. canonical logical dump
// JsonTransformer computes its "expired" fields from WALL CLOCK (currentTimeMillis),
// ignoring the fixed nowInSec passed below. Byte-identical outputs therefore render
// differently when a localExpirationTime falls between the two paths' captures, which run
@@ -652,7 +1083,7 @@ private CapturedSSTable capture(ColumnFamilyStore cfs, SSTableReader sstable, Pa
.replaceAll("\"expired\":\"normalized\"");
}
- // 3. stats spot-check summary
+ // 5. stats spot-check summary
StatsMetadata stats = sstable.getSSTableMetadata();
String statsSummary = "minTimestamp=" + stats.minTimestamp +
" maxTimestamp=" + stats.maxTimestamp +
@@ -663,10 +1094,16 @@ private CapturedSSTable capture(ColumnFamilyStore cfs, SSTableReader sstable, Pa
" totalColumnsSet=" + stats.totalColumnsSet +
" encodingStats=" + sstable.header.stats() +
" metaEncodingStats=" + stats.encodingStats.minTimestamp + "/" + stats.encodingStats.minLocalDeletionTime + "/" + stats.encodingStats.minTTL +
- " tombstoneHist=" + stats.estimatedTombstoneDropTime +
- " cellsPerPartition=" + stats.estimatedCellPerPartitionCount.mean() + "/" + stats.estimatedCellPerPartitionCount.count();
-
- // 4. copy components for byte comparison
+ " tombstoneHist=" + tombstoneHistogram(stats) +
+ " cellsPerPartition=" + stats.estimatedCellPerPartitionCount.mean() + "/" + stats.estimatedCellPerPartitionCount.count() +
+ " partitionSize=" + stats.estimatedPartitionSize.mean() + "/" + stats.estimatedPartitionSize.count() +
+ " sstableLevel=" + stats.sstableLevel +
+ " coveredClustering=" + stats.coveredClustering.toString(sstable.metadata().comparator) +
+ " tokenSpaceCoverage=" + stats.tokenSpaceCoverage +
+ " minTTL=" + stats.minTTL + " maxTTL=" + stats.maxTTL +
+ " hasPartitionLevelDeletions=" + stats.hasPartitionLevelDeletions;
+
+ // 6. copy components for byte comparison
Files.createDirectories(dir);
CapturedSSTable captured = new CapturedSSTable(dir, json, statsSummary);
for (Component c : sstable.descriptor.discoverComponents())
@@ -679,55 +1116,77 @@ private CapturedSSTable capture(ColumnFamilyStore cfs, SSTableReader sstable, Pa
return captured;
}
+ /**
+ * The histogram's CONTENT. TombstoneHistogram has no toString, and its hashCode covers the
+ * backing array's capacity, so two logically equal empty histograms print differently
+ * depending on whether they were built or defaulted. The exact serialized bins are still
+ * pinned, by the byte comparison of Statistics.db.
+ */
+ private static String tombstoneHistogram(StatsMetadata stats)
+ {
+ return "size=" + stats.estimatedTombstoneDropTime.size() +
+ ",sum=" + stats.estimatedTombstoneDropTime.sum(Integer.MAX_VALUE);
+ }
+
protected void assertEquivalentOutputs(CapturedOutput iterator, CapturedOutput cursor)
{
assertEquals("output sstable count differs between paths", iterator.sstables.size(), cursor.sstables.size());
for (int i = 0; i < iterator.sstables.size(); i++)
- {
- CapturedSSTable it = iterator.sstables.get(i);
- CapturedSSTable cu = cursor.sstables.get(i);
-
- // logical first: a row-level diff is far more debuggable than a stats mismatch.
- // In scale mode the dump is a digest — defer it below the byte comparison, which
- // still localizes divergences to exact offsets.
- boolean digestMode = it.json.startsWith("sha256:");
- if (!digestMode && !it.json.equals(cu.json))
- fail("LOGICAL divergence in output sstable " + i + " (iterator vs cursor):\n" + firstJsonDiff(it.json, cu.json) +
- "\niterator stats: " + it.statsSummary + "\ncursor stats: " + cu.statsSummary);
+ assertEquivalentSSTable(i, iterator.sstables.get(i), cursor.sstables.get(i));
+ }
- assertEquals("stats summary divergence in output sstable " + i, it.statsSummary, cu.statsSummary);
+ /** One output sstable of each path: logical dump, stats summary, then every component's bytes. */
+ private void assertEquivalentSSTable(int i, CapturedSSTable it, CapturedSSTable cu)
+ {
+ // logical first: a row-level diff is far more debuggable than a stats mismatch.
+ // In scale mode the dump is a digest — defer it below the byte comparison, which
+ // still localizes divergences to exact offsets.
+ boolean digestMode = it.json.startsWith("sha256:");
+ if (!digestMode && !it.json.equals(cu.json))
+ fail("LOGICAL divergence in output sstable " + i + " (iterator vs cursor):\n" + firstJsonDiff(it.json, cu.json) +
+ "\niterator stats: " + it.statsSummary + "\ncursor stats: " + cu.statsSummary);
+
+ assertEquals("stats summary divergence in output sstable " + i +
+ "\n iterator: " + it.statsSummary + "\n cursor: " + cu.statsSummary,
+ it.statsSummary, cu.statsSummary);
+
+ List divergences = componentDivergences(it, cu);
+ if (!divergences.isEmpty())
+ fail("BYTE divergence in output sstable " + i + " (iterator vs cursor):\n" + String.join("\n", divergences) +
+ "\nNothing is allowed to diverge: every divergence found to date has been a bug in one of the paths");
+
+ if (digestMode)
+ assertEquals("logical dump digest divergence in output sstable " + i +
+ " (scale mode; rerun a reduced scenario without scale mode for a row-level diff)",
+ it.json, cu.json);
+ }
- SortedSet components = new TreeSet<>();
- components.addAll(it.componentSizes.keySet());
- components.addAll(cu.componentSizes.keySet());
- List divergences = new ArrayList<>();
- for (String comp : components)
+ /** One description per component whose bytes differ, or that only one path wrote. */
+ private static List componentDivergences(CapturedSSTable it, CapturedSSTable cu)
+ {
+ SortedSet components = new TreeSet<>();
+ components.addAll(it.componentSizes.keySet());
+ components.addAll(cu.componentSizes.keySet());
+ List divergences = new ArrayList<>();
+ for (String comp : components)
+ {
+ Path a = it.dir.resolve(comp);
+ Path b = cu.dir.resolve(comp);
+ boolean hasA = Files.exists(a);
+ boolean hasB = Files.exists(b);
+ if (hasA != hasB)
{
- Path a = it.dir.resolve(comp);
- Path b = cu.dir.resolve(comp);
- boolean hasA = Files.exists(a);
- boolean hasB = Files.exists(b);
- if (hasA != hasB)
- {
- divergences.add(String.format(" %s: present only in %s path", comp, hasA ? "iterator" : "cursor"));
- continue;
- }
- if (!hasA)
- continue;
- long firstDiff = firstFileDifference(a, b);
- if (firstDiff < 0)
- continue;
- divergences.add(describeFileDiff(comp, a, b, firstDiff));
+ divergences.add(String.format(" %s: present only in %s path", comp, hasA ? "iterator" : "cursor"));
+ continue;
}
- if (!divergences.isEmpty())
- fail("BYTE divergence in output sstable " + i + " (iterator vs cursor):\n" + String.join("\n", divergences) +
- "\nNothing is allowed to diverge: every divergence found to date has been a bug in one of the paths");
-
- if (digestMode)
- assertEquals("logical dump digest divergence in output sstable " + i +
- " (scale mode; rerun a reduced scenario without scale mode for a row-level diff)",
- it.json, cu.json);
+ if (!hasA)
+ continue;
+ long firstDiff = firstFileDifference(a, b);
+ if (firstDiff < 0)
+ continue;
+ divergences.add(describeFileDiff(comp, a, b, firstDiff));
}
+ return divergences;
}
/** Streaming comparison: -1 if byte-identical, else the offset of the first difference
@@ -883,22 +1342,28 @@ private static String firstJsonDiff(String a, String b)
int max = Math.max(linesA.length, linesB.length);
for (int i = 0; i < max; i++)
{
- String la = i < linesA.length ? linesA[i] : "";
- String lb = i < linesB.length ? linesB[i] : "";
- if (!la.equals(lb))
- {
- StringBuilder sb = new StringBuilder();
- sb.append("first differing line ").append(i + 1).append(" of ").append(max).append(":\n");
- for (int j = Math.max(0, i - 2); j < Math.min(max, i + 3); j++)
- {
- String ja = j < linesA.length ? linesA[j] : "";
- String jb = j < linesB.length ? linesB[j] : "";
- sb.append(j == i ? ">>" : " ").append(" iterator: ").append(ja).append('\n');
- sb.append(j == i ? ">>" : " ").append(" cursor: ").append(jb).append('\n');
- }
- return sb.toString();
- }
+ if (!lineAt(linesA, i).equals(lineAt(linesB, i)))
+ return renderDiffContext(linesA, linesB, i, max);
}
return "(no line diff found despite string inequality — check line endings)";
}
+
+ /** Line {@code i} of one dump, or a placeholder where that dump is the shorter one. */
+ private static String lineAt(String[] lines, int i)
+ {
+ return i < lines.length ? lines[i] : "";
+ }
+
+ /** The differing line marked, with two lines either side, both dumps interleaved. */
+ private static String renderDiffContext(String[] linesA, String[] linesB, int i, int max)
+ {
+ StringBuilder sb = new StringBuilder();
+ sb.append("first differing line ").append(i + 1).append(" of ").append(max).append(":\n");
+ for (int j = Math.max(0, i - 2); j < Math.min(max, i + 3); j++)
+ {
+ sb.append(j == i ? ">>" : " ").append(" iterator: ").append(lineAt(linesA, j)).append('\n');
+ sb.append(j == i ? ">>" : " ").append(" cursor: ").append(lineAt(linesB, j)).append('\n');
+ }
+ return sb.toString();
+ }
}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/EdgeCaseDifferentialCompactionTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/EdgeCaseDifferentialCompactionTest.java
index a8541cd0511c..948be346feb3 100644
--- a/test/unit/org/apache/cassandra/db/compaction/differential/EdgeCaseDifferentialCompactionTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/EdgeCaseDifferentialCompactionTest.java
@@ -27,6 +27,8 @@
import org.junit.Test;
+import org.apache.cassandra.config.CassandraRelevantProperties;
+import org.apache.cassandra.config.DatabaseDescriptor;
import org.apache.cassandra.db.Clustering;
import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.Mutation;
@@ -38,8 +40,7 @@
import org.apache.cassandra.db.rows.CellPath;
import org.apache.cassandra.db.rows.Row;
import org.apache.cassandra.io.sstable.format.SSTableReader;
-import org.apache.cassandra.io.sstable.format.big.BigTableReader;
-import org.apache.cassandra.io.sstable.format.big.RowIndexEntry;
+import org.apache.cassandra.io.sstable.AbstractRowIndexEntry;
import org.apache.cassandra.schema.ColumnMetadata;
import org.apache.cassandra.schema.TableMetadata;
import org.apache.cassandra.utils.ByteBufferUtil;
@@ -73,6 +74,13 @@ public class EdgeCaseDifferentialCompactionTest extends DifferentialCompactionTe
* written for those partitions but must not be counted in stats (totalRows/totalColumnsSet).
* The staticRows scenario gives every partition static data, so it never writes an empty
* static row.
+ *
+ * pk 0 additionally carries a static row LARGER than column_index_size, which pins the other
+ * half of the same rule: {@code SSTableCursorWriter} routes a static row to
+ * {@code CursorIndexWriter.staticRowWritten}, which moves the next block's start past it,
+ * because a static row belongs to the partition header and not to a row index block. The
+ * block count below is what states that; the size of a static row cannot otherwise be seen,
+ * since {@code staticRowWritten} is final and branch-free.
*/
@Test
public void emptyStaticRows() throws Exception
@@ -81,12 +89,15 @@ public void emptyStaticRows() throws Exception
ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
cfs.disableAutoCompaction();
+ // pk 0's static alone exceeds the 4 KiB column_index_size; its two regular rows are tiny
+ String bigStatic = "s".repeat(5000);
for (int round = 0; round < 2; round++)
{
for (long pk = 0; pk < 8; pk++)
{
if (pk % 2 == 0)
- execute("INSERT INTO %s (pk, s1, ck, v) VALUES (?, ?, ?, ?)", pk, "static" + pk, (long) round, "v" + round);
+ execute("INSERT INTO %s (pk, s1, ck, v) VALUES (?, ?, ?, ?)",
+ pk, pk == 0 ? bigStatic + round : "static" + pk, (long) round, "v" + round);
else
execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", pk, (long) round, "v" + round);
}
@@ -108,6 +119,16 @@ public void emptyStaticRows() throws Exception
// static s1 cells.
assertTrue("expected totalColumnsSet=20, got: " + out.sstables.get(0).statsSummary,
out.sstables.get(0).statsSummary.contains("totalColumnsSet=20 "));
+
+ // ABSOLUTE. If pk 0's oversized static counted toward the first index block, its row at
+ // ck 0 would cut that block and the row at ck 1 would leave a tail, giving 2 blocks.
+ assertEquals("the cross-generation rung should leave one cursor-produced output",
+ 1, cfs.getLiveSSTables().size());
+ SSTableReader output = cfs.getLiveSSTables().iterator().next();
+ assertEquals("a static row larger than column_index_size opened an index block: two tiny " +
+ "regular rows cannot reach the threshold on their own, so this partition must " +
+ "not be promoted however large its static row is",
+ 0, blockCount(output, 0L));
}
/**
@@ -517,7 +538,21 @@ public void compositeClustering() throws Exception
assertCursorMatchesIteratorAcrossGenerations(cfs);
}
- /** Wide partition crossing column-index block boundaries (indexed RowIndexEntry path). */
+ /**
+ * Wide partition crossing column-index block boundaries (indexed RowIndexEntry path).
+ *
+ * Round 0's delete opens at {@code ck >= 0}, and the table has no static column, so that
+ * {@code INCL_START_BOUND} sorts ahead of every row and IS the partition's first unfiltered.
+ * At the config's 4 KiB granularity the first block cut falls around row 17, well inside the
+ * range covering rows 0-249, so {@code BtiCursorIndexWriter.blockStartOpenMarker} starts LIVE,
+ * is replaced with the range's deletion at the first cut, and is back to LIVE well before the
+ * last: the whole open-marker-across-a-cut cycle. A writer that carried an open marker across
+ * a partition boundary, or that recorded the marker open at the START of a block rather than
+ * at the end of the previous one, has to produce different index bytes here.
+ *
+ * Cannot see: the recorded marker itself. {@code IndexInfo.openDeletion} is not exposed by any
+ * reader; the byte comparison of the index component is what pins it.
+ */
@Test
public void widePartitionCrossingIndexBlocks() throws Exception
{
@@ -537,6 +572,7 @@ public void widePartitionCrossingIndexBlocks() throws Exception
}
assertCursorMatchesIteratorAcrossGenerations(cfs);
+ assertIndexedCursorOutput(cfs);
}
/**
@@ -572,8 +608,8 @@ public void partitionCrossingOneIndexBlock() throws Exception
// index can be read back directly. The iterator promotes when the total block count INCLUDING
// the tail exceeds one (RowIndexEntry.create); a merge that decides before counting the tail
// leaves a partition crossing the threshold exactly once with no promoted index at all, and no
- // intra-partition seeks. Byte-equality pins this via Index.db only while the reference stays
- // correct, so the promotion is stated here directly.
+ // intra-partition seeks. Byte-equality pins this through the index component only while the
+ // reference stays correct, so the promotion is stated here directly.
assertEquals("the cross-generation rung should leave one cursor-produced output",
1, cfs.getLiveSSTables().size());
SSTableReader output = cfs.getLiveSSTables().iterator().next();
@@ -586,8 +622,8 @@ public void partitionCrossingOneIndexBlock() throws Exception
/** Promoted index block count for {@code pk} in {@code sstable}; 0 when the partition is not indexed. */
private static int blockCount(SSTableReader sstable, long pk)
{
- RowIndexEntry entry = ((BigTableReader) sstable).getRowIndexEntry(sstable.decorateKey(ByteBufferUtil.bytes(pk)),
- SSTableReader.Operator.EQ);
+ AbstractRowIndexEntry entry = sstable.getRowIndexEntry(sstable.decorateKey(ByteBufferUtil.bytes(pk)),
+ SSTableReader.Operator.EQ);
assertNotNull("expected pk " + pk + " to be present in " + sstable.descriptor, entry);
return entry.blockCount();
}
@@ -1883,6 +1919,716 @@ public void frozenCollectionDeleteAndTtl() throws Exception
assertEquals("row column missing at ck " + ck, 1, countOccurrences(json, cellValue("row" + ck)));
}
+ // ------------------------------------------------------------------------------------------
+ // Scenarios below exist for the ROW INDEX, i.e. for partitions above column_index_size.
+ //
+ // Every one of them is here because ClusteringDescriptorPrefixView.parse — the only
+ // cursor-specific input to a BTI row trie — ran in exactly one shape before them: a single
+ // `ck bigint`, fixed width, never null, never empty, never a second component. parse only runs
+ // from BtiCursorIndexWriter.addIndexBlock, which only runs at a block cut, which only happens
+ // above column_index_size (4 KiB in test/conf/cassandra.yaml), so a scenario that stays under
+ // that threshold exercises none of it however exotic its clustering is.
+ //
+ // The oracles are: the harness byte comparison over Rows.db / Partitions.db (or Index.db under
+ // BIG) for the written bytes, assertEveryRowReadableThroughASlice for retrievability, and the
+ // absolute blockCount assertions here for the block arithmetic. A block count is stated
+ // outright wherever the shape makes it computable, because byte equality is blind to a rule
+ // both pipelines get wrong.
+ // ------------------------------------------------------------------------------------------
+
+ /**
+ * Granularity the two one-byte-resolution sweeps below run at. 1 KiB is the smallest
+ * column_index_size the config accepts, and a sweep pays one partition per byte, so the
+ * smallest granularity is the cheapest place to bracket a cut.
+ */
+ private static final int SWEEP_GRANULARITY_KIB = 1;
+ private static final int SWEEP_GRANULARITY = SWEEP_GRANULARITY_KIB * 1024;
+
+ /**
+ * How many one-byte padding steps a block-boundary sweep walks, ending one byte short of
+ * {@link #SWEEP_GRANULARITY}. It has to exceed the per-row serialization overhead — row flags,
+ * clustering, the body and previous-body length vints, the liveness delta and the cell header,
+ * comfortably under 40 bytes today — so the sweep straddles the cut rather than sitting wholly
+ * on one side of it; the marker sweep additionally needs it to exceed that plus one range
+ * tombstone boundary marker. Both sweeps fail with an explicit "widen this" message if the
+ * value ever stops being enough, so a serialization change that outgrows it is a loud failure
+ * and not a silently vacuous test.
+ */
+ private static final int SWEEP_BYTES =
+ CassandraRelevantProperties.TEST_DIFFERENTIAL_BLOCK_BOUNDARY_SWEEP.getInt();
+
+ /** Named in every sweep failure message, so a failure says which knob to turn. */
+ private static final String SWEEP_PROPERTY =
+ CassandraRelevantProperties.TEST_DIFFERENTIAL_BLOCK_BOUNDARY_SWEEP.getKey();
+
+ /**
+ * The single cursor-written sstable the cross-generation rung leaves live, with the absolute
+ * assertion that at least one of its partitions really carried a promoted row index.
+ *
+ * Every scenario in this section is defined by crossing column_index_size. A change to row
+ * sizing, to the config, or to a merge rule that quietly dropped a scenario back under the
+ * threshold would leave it passing while testing an unindexed partition, which is the shape
+ * the rest of this class already covers to death. The non-zero return is what says the index
+ * was exercised at all.
+ */
+ private SSTableReader assertIndexedCursorOutput(ColumnFamilyStore cfs)
+ {
+ assertEquals("the cross-generation rung should leave one cursor-produced output",
+ 1, cfs.getLiveSSTables().size());
+ SSTableReader output = cfs.getLiveSSTables().iterator().next();
+ assertTrue("no partition of the cursor-written output carries a promoted row index: this " +
+ "scenario has stopped crossing column_index_size and now says nothing about the " +
+ "row index at all",
+ assertEveryRowReadableThroughASlice(output) > 0);
+ return output;
+ }
+
+ /**
+ * A single {@code blob} clustering above the block threshold, whose largest value is a run of
+ * {@code 0xFF} bytes and one of whose block-cutting values ends in {@code 0x00}.
+ *
+ * What it covers, stated as what it reaches rather than as what it is named for:
+ *
+ * - The vint length branch of {@code ClusteringDescriptorPrefixView.parse} for a single
+ * variable-width component, which the tree's {@code ck bigint} partitions never touch.
+ * {@code UTF8Type} and {@code BytesType} are both variable width and the type identity does
+ * not change the walk, so a {@code text} clustering of this shape is not written as a
+ * second scenario; {@code partitionEndingOnABlockCutHasNoTailBlock} below runs one anyway,
+ * over a 200-byte shared prefix.
+ * - The {@code 0x00} escape {@code ByteSource.escaped} performs on the way into a trie
+ * separator. A clustering only reaches a separator as a block's FIRST or LAST, the two
+ * positions {@code snapshotOf} is applied to, so the row carrying the trailing
+ * {@code 0x00} clustering takes a value larger than the granularity and cuts a block by
+ * itself, making it both. {@code blob} reaches this and {@code 0xFF} in one table;
+ * {@code text} cannot hold {@code 0xFF} at all.
+ * - Aimed at, but NOT asserted: the {@code 0xFF} arm of {@code RowIndexWriter.nudge}, which
+ * returns the byte unchanged rather than incrementing it. The partition's maximum is
+ * {@code 0xFF} followed by eight more {@code 0xFF} bytes, so the branch fires if the
+ * preceding separator diverges exactly on the order byte, and that position is decided by
+ * where {@code prevSep} left {@code prevMax}.
+ *
+ *
+ * Cannot see: that the nudge branch fired. Nothing counts it, and a nudge that silently
+ * produced a separator no greater than the maximum still yields an ordered trie, so
+ * {@code IncrementalTrieWriterBase.add}'s order assertion stays quiet. The byte comparison pins
+ * that both pipelines nudged alike; the slice read-back pins that the result routes.
+ */
+ @Test
+ public void blobClusteringCrossingIndexBlocks() throws Exception
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck blob, v text, PRIMARY KEY (pk, ck))");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ String padding = "x".repeat(200);
+ for (int round = 0; round < 2; round++)
+ {
+ // orders 0xD8..0xFF, so the partition's maximum clustering — the one nudge() is applied
+ // to — is 0xFF followed by eight more 0xFF bytes
+ for (int order = 0xD8; order <= 0xFF; order++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)",
+ 1L, blobClustering(order, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF),
+ padding + "-" + round);
+ // trailing 0x00 bytes, kept at low order bytes so they cannot become the maximum
+ for (int order = 0xD8; order < 0xDD; order++)
+ {
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)",
+ 1L, blobClustering(order, 0x00, 0x00), padding + "-" + round);
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)",
+ 1L, blobClustering(order, 0x01, 0x00), padding + "-" + round);
+ }
+ // one clustering ending in 0x00 whose value exceeds the 4 KiB granularity, so it cuts
+ // a block on its own and is that block's FIRST and LAST: the escape reaches a trie
+ // separator by design here, not by wherever the cuts happened to fall. Order 0xDD is
+ // below 0xFF, so it cannot become the partition's maximum.
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)",
+ 1L, blobClustering(0xDD, 0x00, 0x00), "e".repeat(5000) + "-" + round);
+ for (int order = 0; order < 3; order++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)",
+ 2L, blobClustering(order, 0x01), "small-" + round);
+ flush();
+ }
+
+ assertCursorMatchesIteratorAcrossGenerations(cfs);
+ SSTableReader output = assertIndexedCursorOutput(cfs);
+ assertEquals("a partition well under column_index_size must not be promoted", 0, blockCount(output, 2L));
+ }
+
+ /** A blob clustering: a 16-byte shared prefix, one ordering byte, then {@code suffix}. */
+ private static ByteBuffer blobClustering(int order, int... suffix)
+ {
+ byte[] bytes = new byte[16 + 1 + suffix.length];
+ for (int i = 0; i < 16; i++)
+ bytes[i] = 0x11;
+ bytes[16] = (byte) order;
+ for (int i = 0; i < suffix.length; i++)
+ bytes[17 + i] = (byte) suffix[i];
+ return ByteBuffer.wrap(bytes);
+ }
+
+ /**
+ * Three clustering columns above the block threshold, variable width then two fixed widths.
+ *
+ * {@code compositeClustering} above builds the same column shape but ~800 bytes per partition,
+ * so it never cuts a block. Here {@code ClusteringDescriptorPrefixView.parse} walks PAST
+ * component 0: the {@code pos += len} advance after a variable-width component, and the
+ * two-bit-per-component header shift, both run only for {@code i > 0}. A single wrong offset
+ * there still yields an ordered separator, so the trie is written and misroutes.
+ *
+ * {@code descendingClusteringCrossingIndexBlocks} below takes the SAME parse branches in the
+ * opposite component order; it is kept for {@code ReversedType}, which is outside parse, not
+ * for this walk. Both are named, deterministic shapes that fail loudly rather than drifting
+ * out of a generator's range, which is what they hold over
+ * {@code RandomDifferentialCompactionTest}'s generated clusterings.
+ *
+ * Cannot see: more than 32 components, where parse reads a second header vint. No table
+ * written by CQL in this class has that many clustering columns;
+ * {@code RandomDifferentialCompactionTest} draws 33-36 on one example in four, and
+ * {@code ClusteringDescriptorPrefixViewTest.parseMatchesTheSerializer} asserts its corpus
+ * reached that branch.
+ */
+ @Test
+ public void compositeClusteringCrossingIndexBlocks() throws Exception
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck1 text, ck2 int, ck3 bigint, v text, " +
+ "PRIMARY KEY (pk, ck1, ck2, ck3))");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ String prefix = "p".repeat(48);
+ String padding = "x".repeat(200);
+ for (int round = 0; round < 2; round++)
+ {
+ for (String ck1 : new String[]{ prefix + "a", prefix + "b" })
+ for (int ck2 = 0; ck2 < 5; ck2++)
+ for (long ck3 = 0; ck3 < 3; ck3++)
+ execute("INSERT INTO %s (pk, ck1, ck2, ck3, v) VALUES (?, ?, ?, ?, ?)",
+ 1L, ck1, ck2, ck3, padding + "-" + round);
+ for (long ck3 = 0; ck3 < 3; ck3++)
+ execute("INSERT INTO %s (pk, ck1, ck2, ck3, v) VALUES (?, ?, ?, ?, ?)",
+ 2L, prefix + "a", 0, ck3, "small-" + round);
+ flush();
+ }
+
+ assertCursorMatchesIteratorAcrossGenerations(cfs);
+ SSTableReader output = assertIndexedCursorOutput(cfs);
+ assertEquals("a partition well under column_index_size must not be promoted", 0, blockCount(output, 2L));
+ }
+
+ /**
+ * A {@code DESC} clustering column above the block threshold, paired with an {@code ASC} one.
+ *
+ * Its parse branch set is the one {@code compositeClusteringCrossingIndexBlocks} above already
+ * walks, in the opposite component order — fixed width then variable rather than the reverse —
+ * and {@code parse} has no branch that separates the two orders. What it adds is outside
+ * {@code parse} entirely:
+ * {@code ClusteringComparator.asByteComparable} emits inverted bytes and
+ * {@code NEXT_COMPONENT_EMPTY_REVERSED} for a {@link org.apache.cassandra.db.marshal.ReversedType}
+ * component, and that is the encoding the row trie's separators are built from. The write side
+ * builds its comparator from {@code SerializationHeader.clusteringTypes()} while the read side
+ * uses {@code metadata.comparator}. No other DETERMINISTIC scenario pairs DESC with an indexed
+ * partition: {@code descendingClustering} and {@code openEndedRangeTombstonesDescending} above
+ * both stay under the threshold and so never reach a trie.
+ * {@code BtiRandomDifferentialCompactionTest} does reach the shape, because its generator wraps
+ * a clustering type in {@code ReversedType} on a coin flip, but only when the draw also lands a
+ * hub partition across the drawn granularity; this scenario is the one that always does.
+ * Mixing DESC with ASC means a comparator that inverted unconditionally fails here too.
+ *
+ * Cannot see: a disagreement between the two comparators that both pipelines share. Both build
+ * the write-side comparator the same way, so the byte comparison is blind to it; the slice
+ * read-back, which goes through {@code metadata.comparator}, is what covers it.
+ */
+ @Test
+ public void descendingClusteringCrossingIndexBlocks() throws Exception
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck1 bigint, ck2 text, v text, " +
+ "PRIMARY KEY (pk, ck1, ck2)) WITH CLUSTERING ORDER BY (ck1 DESC, ck2 ASC)");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ String prefix = "p".repeat(48);
+ String padding = "x".repeat(200);
+ for (int round = 0; round < 2; round++)
+ {
+ for (long ck1 = 0; ck1 < 10; ck1++)
+ for (int ck2 = 0; ck2 < 3; ck2++)
+ execute("INSERT INTO %s (pk, ck1, ck2, v) VALUES (?, ?, ?, ?)",
+ 1L, ck1, prefix + ck2, padding + "-" + round);
+ for (long ck1 = 0; ck1 < 3; ck1++)
+ execute("INSERT INTO %s (pk, ck1, ck2, v) VALUES (?, ?, ?, ?)",
+ 2L, ck1, prefix + "0", "small-" + round);
+ flush();
+ }
+
+ assertCursorMatchesIteratorAcrossGenerations(cfs);
+ SSTableReader output = assertIndexedCursorOutput(cfs);
+ assertEquals("a partition well under column_index_size must not be promoted", 0, blockCount(output, 2L));
+ }
+
+ /**
+ * An EMPTY clustering component inside a multi-block partition, ascending.
+ *
+ * {@code ClusteringDescriptorPrefixView.parse} has a null branch and an empty branch that no
+ * test reaches, because the two scenarios in this class that write an empty clustering
+ * ({@code emptyClusteringValuesAscending} and its DESC twin) build partitions far under
+ * column_index_size. Landing an empty component in a trie separator takes more than writing
+ * one: {@code snapshotOf} is only applied to a block's FIRST and LAST clustering. The
+ * empty-clustering row therefore carries a value larger than the granularity, so it cuts a
+ * block on its own and is necessarily both — under ASC it sorts first, so it is block 1
+ * entire.
+ *
+ * Cannot see: a null (as opposed to empty) clustering component. CQL cannot write one on a
+ * single-column clustering; that branch stays unreached.
+ */
+ @Test
+ public void emptyClusteringComponentCrossingIndexBlocksAscending() throws Exception
+ {
+ emptyClusteringComponentCrossingIndexBlocks(false);
+ }
+
+ /**
+ * DESC twin of the above: the empty component sorts LAST, so it is the last block's last
+ * clustering rather than the first block's first. That is the other of the two positions
+ * {@code snapshotOf} is applied to, and under {@code ReversedType} it is also the value
+ * {@code RowIndexWriter.complete} nudges.
+ */
+ @Test
+ public void emptyClusteringComponentCrossingIndexBlocksDescending() throws Exception
+ {
+ emptyClusteringComponentCrossingIndexBlocks(true);
+ }
+
+ private void emptyClusteringComponentCrossingIndexBlocks(boolean descending) throws Exception
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck text, v text, PRIMARY KEY (pk, ck))" +
+ (descending ? " WITH CLUSTERING ORDER BY (ck DESC)" : ""));
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ String padding = "x".repeat(200);
+ // larger than the 4 KiB column_index_size, so the empty-clustering row cuts a block by itself
+ String bigPadding = "e".repeat(5000);
+ for (int round = 0; round < 2; round++)
+ {
+ for (int i = 0; i < 30; i++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)",
+ 1L, "c" + String.format("%04d", i), padding + "-" + round);
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)",
+ 1L, ByteBufferUtil.EMPTY_BYTE_BUFFER, bigPadding + "-" + round);
+ flush();
+ }
+
+ assertCursorMatchesIteratorAcrossGenerations(cfs);
+ assertIndexedCursorOutput(cfs);
+ }
+
+ /**
+ * The tail-block decision, stated as four absolute block counts, over clusterings that share a
+ * 200-byte prefix.
+ *
+ * {@code BtiCursorIndexWriter.endPartition} cuts a trailing block only when a block is still
+ * open, and BIG reaches the same decision from a tail size of more than the one end-of-partition
+ * marker byte. {@code fixedLengthValuesLargerThanCopyBuffer} and
+ * {@code mapKeysAcrossTheVintLengthBoundary} above already REACH the closed-block arm, because
+ * every row of theirs exceeds the granularity, but neither asserts a block count, so a writer
+ * that cut a tail unconditionally passes both. The pk 1 / pk 2 pair below is what separates the
+ * branches.
+ *
+ * The {@code text} clustering carries a 200-byte shared prefix, so this is also where the
+ * separator chain runs deep: every {@code ByteComparable.separatorGt} result runs the whole
+ * prefix before it diverges, and {@code RowIndexWriter.complete} walks that prefix to find its
+ * nudge point. Neither has a length-dependent branch, so the prefix buys reach into the loops
+ * rather than a new branch.
+ *
+ * Each row here exceeds the 4 KiB granularity on its own, which makes the counts exact without
+ * any arithmetic on the serialized row size:
+ *
+ * - pk 1, two big rows: both cut, nothing is left open, 2 blocks.
+ * - pk 2, two big rows and a small one: the small row leaves a block open, so a tail is cut,
+ * 3 blocks. This is the pair that pins the branch — a writer that never cut a tail gives
+ * pk 2 two blocks, one that always cut gives pk 1 three.
+ * - pk 3, ONE big row: one cut, no tail, and a one-block index is not promoted at all
+ * (BTI's {@code finish} returns a -1 trie root, BIG's {@code totalBlocks <= 1} writes a
+ * plain entry), so 0.
+ * - pk 4, two small rows: never reaches the threshold, 0.
+ *
+ *
+ * Cannot see: a partition that ends EXACTLY on a granularity multiple. Hitting that needs the
+ * serialized size of a row, which no reader exposes; {@code blockCutBracketsTheGranularityCut}
+ * below brackets it to one byte instead of naming it.
+ */
+ @Test
+ public void partitionEndingOnABlockCutHasNoTailBlock() throws Exception
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck text, v text, PRIMARY KEY (pk, ck))");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ // shared by every clustering, so each separator runs 200 bytes deep before it diverges.
+ // The suffix is zero-padded, so lexicographic order is the numeric order the counts assume.
+ String prefix = "p".repeat(200);
+ String big = "x".repeat(4500); // one row > the 4 KiB column_index_size
+ String small = "s".repeat(50);
+ for (int round = 0; round < 2; round++)
+ {
+ for (int ck = 0; ck < 2; ck++)
+ {
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", 1L, clustering(prefix, ck), big + "-" + round);
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", 2L, clustering(prefix, ck), big + "-" + round);
+ }
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", 2L, clustering(prefix, 2), small + "-" + round);
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", 3L, clustering(prefix, 0), big + big + "-" + round);
+ for (int ck = 0; ck < 2; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", 4L, clustering(prefix, ck), small + "-" + round);
+ flush();
+ }
+
+ assertCursorMatchesIteratorAcrossGenerations(cfs);
+ SSTableReader output = assertIndexedCursorOutput(cfs);
+ assertEquals("both rows exceed column_index_size, so the last one ends a block and no tail " +
+ "remains to cut", 2, blockCount(output, 1L));
+ assertEquals("the trailing small row leaves a block open, so endPartition must cut a tail",
+ 3, blockCount(output, 2L));
+ assertEquals("a single row cuts one block and leaves no tail, and a one-block index is never " +
+ "promoted", 0, blockCount(output, 3L));
+ assertEquals("a partition under column_index_size must not be promoted", 0, blockCount(output, 4L));
+ }
+
+ /** A {@code text} clustering: a long shared prefix, then a zero-padded suffix that orders. */
+ private static String clustering(String prefix, int suffix)
+ {
+ return prefix + String.format("%04d", suffix);
+ }
+
+ /**
+ * Brackets the granularity cut to a single byte, by sweeping the row size across it.
+ *
+ * The exact shapes this stands in for — a partition of exactly N granularities, of a
+ * granularity plus one byte, a single row of exactly the block size — need the serialized size
+ * of a row, which is a function of the row flags, the clustering encoding, two length vints, the
+ * liveness delta against the sstable's encoding stats and the cell header. No reader exposes it,
+ * and guessing it would give a scenario that CLAIMS to sit on the cut and does not. This sweeps
+ * instead: one partition per padding length, one byte apart, so the exact boundary is somewhere
+ * inside and the shape of the crossing is asserted rather than its position.
+ *
+ * Every partition holds two rows of the same padding. Row 1 cuts iff its serialized size reaches
+ * the granularity; if it does, row 2 is at least as large and cuts too, giving 2 blocks with no
+ * tail. If row 1 does not cut, the two together do, giving one block — never promoted, reported
+ * as 0. So the count is 0 below the cut and 2 at or above it, and the serialized size is
+ * monotone in the padding, so the sweep must show one step and no other value. A fixed
+ * {@code USING TIMESTAMP} keeps the liveness delta constant, so the row size is a function of
+ * the padding alone.
+ *
+ * Runs at {@link #SWEEP_GRANULARITY_KIB} KiB rather than the config's 4 KiB purely for cost:
+ * the sweep pays one partition per byte either way.
+ *
+ * Cannot see: WHICH padding sits exactly on the cut, only that exactly one does. And it says
+ * nothing about a writer whose cut is off by a constant — the step would simply move, and this
+ * asserts the step's shape, not its position.
+ */
+ @Test
+ public void blockCutBracketsTheGranularityCut() throws Exception
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck bigint, v text, PRIMARY KEY (pk, ck))");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ int previousGranularity = DatabaseDescriptor.getColumnIndexSizeInKiB();
+ // set BEFORE the compaction, not before the schema: BtiCursorIndexWriter reads
+ // column_index_size once, in its constructor
+ DatabaseDescriptor.setColumnIndexSizeInKiB(SWEEP_GRANULARITY_KIB);
+ try
+ {
+ for (int step = 0; step < SWEEP_BYTES; step++)
+ {
+ String padding = "x".repeat(SWEEP_GRANULARITY - SWEEP_BYTES + step);
+ for (long ck = 0; ck < 2; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?) USING TIMESTAMP 1000",
+ (long) step, ck, padding);
+ }
+ flush();
+ // a second input, so this is a merge and not a single-sstable rewrite
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?) USING TIMESTAMP 1000", -1L, 0L, "control");
+ flush();
+
+ assertCursorMatchesIteratorAcrossGenerations(cfs);
+ SSTableReader output = assertIndexedCursorOutput(cfs);
+
+ assertEquals("the sweep starts ABOVE the cut, so it does not bracket it: the shortest " +
+ "padding already cuts a block. Widen " + SWEEP_PROPERTY,
+ 0, blockCount(output, 0));
+ assertEquals("the sweep ends BELOW the cut, so it does not bracket it: even the longest " +
+ "padding never reaches " + SWEEP_GRANULARITY + " serialized bytes. Widen " +
+ SWEEP_PROPERTY,
+ 2, blockCount(output, SWEEP_BYTES - 1));
+
+ int steps = 0;
+ int previousCount = 0;
+ for (int step = 0; step < SWEEP_BYTES; step++)
+ {
+ int count = blockCount(output, step);
+ assertTrue("padding step " + step + " gave " + count + " blocks: two rows can cut at " +
+ "most one block each, and a one-block index is never promoted, so 0 and 2 " +
+ "are the only counts reachable here",
+ count == 0 || count == 2);
+ if (count != previousCount)
+ {
+ assertEquals("the promoted block count FELL as the rows grew, at padding step " +
+ step + ": the serialized row size is monotone in the padding, so the " +
+ "cut cannot un-fire", 2, count);
+ steps++;
+ }
+ previousCount = count;
+ }
+ assertEquals("the block count crossed the cut more than once, so the serialized row size " +
+ "is not monotone in the padding and the byte the cut sits on is not bracketed",
+ 1, steps);
+ }
+ finally
+ {
+ DatabaseDescriptor.setColumnIndexSizeInKiB(previousGranularity);
+ }
+ }
+
+ /**
+ * Puts a range tombstone BOUNDARY marker at the end of an index block, by sweeping the row
+ * before it across the cut.
+ *
+ * {@code SSTableCursorWriter.writeRangeTombstone} sets the open marker to a boundary's
+ * {@code deletionTime2} — the deletion of the range that OPENS there — and when a block is cut
+ * on that marker, that is the value {@code addIndexBlock} carries into the NEXT block's
+ * {@code IndexInfo}. Nothing in the tree lands a boundary marker at a cut by design; the wide
+ * partitions that hold both do it by whatever the layout happened to be.
+ *
+ * Each partition is: the open bound of [0,1), a row of swept padding, the boundary at 1, a row
+ * larger than the granularity, and the close bound of [1,3). The rows are written in one
+ * sstable and the two ranges in another, so the boundary is formed by the MERGE. The block
+ * count is 2 while neither the open bound nor the first row reaches the cut, and 3 once
+ * something does. The first padding at which it becomes 3 is necessarily a partition where the
+ * BOUNDARY MARKER, not the row, ended block 1: the step happens the moment
+ * {@code openBound + row + marker} reaches the granularity, and one padding byte earlier
+ * {@code openBound + row} was already below it by at least a marker's width. So the sweep
+ * contains at least one such partition by construction, not by luck.
+ *
+ * Cannot see: which partition that is, or that the {@code IndexInfo} carried the right
+ * deletion. The byte comparison of Rows.db / Index.db is the oracle for the value; this
+ * scenario's job is only to make the shape occur.
+ */
+ @Test
+ public void blockCutLandsOnARangeTombstoneBoundaryMarker() throws Exception
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck bigint, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH gc_grace_seconds = 864000");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ int previousGranularity = DatabaseDescriptor.getColumnIndexSizeInKiB();
+ DatabaseDescriptor.setColumnIndexSizeInKiB(SWEEP_GRANULARITY_KIB);
+ try
+ {
+ // comfortably over the granularity on its own, so it always cuts, and large enough
+ // that a three-block partition clears the harness's own
+ // "length >= (blocks - 1) * granularity" bound with room
+ String trailing = "y".repeat(SWEEP_GRANULARITY + 512);
+ for (int step = 0; step < SWEEP_BYTES; step++)
+ {
+ String padding = "x".repeat(SWEEP_GRANULARITY - SWEEP_BYTES + step);
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?) USING TIMESTAMP 3000",
+ (long) step, 0L, padding);
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?) USING TIMESTAMP 3000",
+ (long) step, 1L, trailing);
+ }
+ flush();
+
+ // two abutting ranges with DIFFERENT deletion times: equal ones would merge into a
+ // single range and produce no boundary marker at all. Both are older than the rows, so
+ // the rows survive and the markers stay.
+ for (int step = 0; step < SWEEP_BYTES; step++)
+ {
+ execute("DELETE FROM %s USING TIMESTAMP 1000 WHERE pk = ? AND ck >= ? AND ck < ?",
+ (long) step, 0L, 1L);
+ execute("DELETE FROM %s USING TIMESTAMP 2000 WHERE pk = ? AND ck >= ? AND ck < ?",
+ (long) step, 1L, 3L);
+ }
+ flush();
+
+ assertCursorMatchesIteratorAcrossGenerations(cfs);
+ SSTableReader output = assertIndexedCursorOutput(cfs);
+
+ assertEquals("the sweep starts ABOVE the cut: the shortest padding already ends block 1 " +
+ "before the trailing row, so the step this scenario relies on is outside the " +
+ "sweep. Widen " + SWEEP_PROPERTY,
+ 2, blockCount(output, 0));
+ assertEquals("the sweep ends BELOW the cut: even the longest padding leaves block 1 open " +
+ "until the trailing row, so no partition ended a block on the boundary " +
+ "marker. Widen " + SWEEP_PROPERTY,
+ 3, blockCount(output, SWEEP_BYTES - 1));
+
+ int steps = 0;
+ int previousCount = 2;
+ for (int step = 0; step < SWEEP_BYTES; step++)
+ {
+ int count = blockCount(output, step);
+ assertTrue("padding step " + step + " gave " + count + " blocks; the trailing row " +
+ "always cuts and the close bound always leaves a tail, so the only counts " +
+ "reachable here are 2 (nothing cut before the trailing row) and 3",
+ count == 2 || count == 3);
+ if (count != previousCount)
+ {
+ assertEquals("the promoted block count FELL as the first row grew, at padding step " +
+ step, 3, count);
+ steps++;
+ }
+ previousCount = count;
+ }
+ assertEquals("the block count crossed the cut more than once, so the step from 2 to 3 does " +
+ "not identify the partitions whose block 1 ended on the boundary marker",
+ 1, steps);
+ }
+ finally
+ {
+ DatabaseDescriptor.setColumnIndexSizeInKiB(previousGranularity);
+ }
+ }
+
+ /**
+ * An INDEXED partition carrying a non-LIVE partition-level deletion.
+ *
+ * {@code TrieIndexEntry.serialize} writes a partition deletion time into the index entry, and
+ * only for an indexed entry; BIG's promoted entry has the same field. Partition-level deletes
+ * exist elsewhere in the suite but always on small partitions, so neither field has ever been
+ * written non-LIVE, in either format. That also leaves the eager serialization in
+ * {@code BtiTableWriter.IndexWriter.append} unpinned — the caller hands it a REUSED
+ * {@code DeletionTime} instance, and the entry is correct only because it is serialized before
+ * the call returns.
+ *
+ * pk 1 is deleted between two rounds of inserts, so the deletion survives compaction (it is not
+ * purgeable inside gc_grace) while the later rows survive it, leaving a partition that is both
+ * indexed and deleted. pk 2 is the same shape without the delete, so the entry's deletion field
+ * is asserted against both values and cannot pass as a constant.
+ *
+ * The index entry's copy is read back directly here; the harness's
+ * {@code assertPartitionDeletionReadableFromIndexEntry}, which compares it against the data
+ * file's, is dormant until a scenario like this one exists.
+ *
+ * Cannot see: a reused-instance defect that happens to reuse the SAME value. Both partitions'
+ * deletions would have to differ within one sstable for that, which one partition delete
+ * cannot arrange.
+ */
+ @Test
+ public void indexedPartitionCarriesAPartitionDeletion() throws Exception
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck bigint, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH gc_grace_seconds = 864000");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ String padding = "x".repeat(200);
+ for (long pk = 1; pk <= 2; pk++)
+ for (long ck = 0; ck < 30; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?) USING TIMESTAMP 1000", pk, ck, padding + "-0");
+ flush();
+
+ execute("DELETE FROM %s USING TIMESTAMP 2000 WHERE pk = ?", 1L);
+ flush();
+
+ // re-inserted above the deletion, so pk 1 stays wide enough to be indexed
+ for (long pk = 1; pk <= 2; pk++)
+ for (long ck = 0; ck < 30; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?) USING TIMESTAMP 3000", pk, ck, padding + "-1");
+ flush();
+
+ assertCursorMatchesIteratorAcrossGenerations(cfs);
+ SSTableReader output = assertIndexedCursorOutput(cfs);
+
+ assertEquals("pk 1 must still cross column_index_size exactly once after the delete",
+ 2, blockCount(output, 1L));
+ assertEquals("pk 2 is the undeleted control and must be indexed the same way",
+ 2, blockCount(output, 2L));
+
+ AbstractRowIndexEntry deleted = output.getRowIndexEntry(output.decorateKey(ByteBufferUtil.bytes(1L)),
+ SSTableReader.Operator.EQ);
+ assertNotNull("pk 1 lost its index entry", deleted);
+ assertNotNull("an indexed entry must carry a partition deletion time", deleted.deletionTime());
+ assertFalse("the index entry for a deleted partition reports a LIVE deletion: the entry's " +
+ "deletion field is the only copy a read takes when the column filter fetches no " +
+ "statics, so a partition delete lost here is a partition delete lost on read",
+ deleted.deletionTime().isLive());
+ assertEquals("the index entry carries the wrong deletion timestamp",
+ 2000L, deleted.deletionTime().markedForDeleteAt());
+
+ AbstractRowIndexEntry undeleted = output.getRowIndexEntry(output.decorateKey(ByteBufferUtil.bytes(2L)),
+ SSTableReader.Operator.EQ);
+ assertNotNull("pk 2 lost its index entry", undeleted);
+ assertTrue("the undeleted control partition's index entry reports a deletion, so the field is " +
+ "not being read from the partition at all",
+ undeleted.deletionTime().isLive());
+ }
+
+ /**
+ * A DESIGNED partition at BTI's own default granularity, 16 KiB
+ * ({@code BtiFormatPartitionWriter.DEFAULT_GRANULARITY}).
+ *
+ * Every config in the tree sets column_index_size to 4 KiB — test/conf/cassandra.yaml,
+ * test/conf/latest_diff.yaml and InstanceConfig alike. 16 KiB is not unreached, though:
+ * {@code RandomDifferentialCompactionTest} draws its granularity per example from
+ * {@code COLUMN_INDEX_SIZES_KIB}, one of whose eight entries is 16, and applies it immediately
+ * before the compaction. What that soak does not do is state a designed count, so a partition
+ * whose promotion decision moved between 4 KiB and 16 KiB would still leave it green. This
+ * scenario names one. {@code BtiCursorIndexWriter} reads the granularity ONCE, in its
+ * constructor, so it is set after the writes and before the compaction; setting it before
+ * {@code createTable} would change nothing about the compaction under test.
+ *
+ * pk 2 is what proves the setting took effect. It holds 30 padded rows, the shape
+ * {@code partitionCrossingOneIndexBlock} above pins at exactly 2 blocks under the config's
+ * 4 KiB. At 16 KiB it must not be promoted at all. Without that assertion this scenario would
+ * pass identically if the setter were a no-op.
+ *
+ * Cannot see: a granularity read at the wrong TIME. A writer that re-read column_index_size per
+ * partition instead of per writer would behave identically here, because the value does not
+ * change during the compaction.
+ */
+ @Test
+ public void realBtiGranularity() throws Exception
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck bigint, v text, PRIMARY KEY (pk, ck))");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ String padding = "x".repeat(200);
+ for (int round = 0; round < 2; round++)
+ {
+ for (long ck = 0; ck < 400; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", 1L, ck, padding + "-" + round);
+ for (long ck = 0; ck < 30; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", 2L, ck, padding + "-" + round);
+ flush();
+ }
+
+ int previousGranularity = DatabaseDescriptor.getColumnIndexSizeInKiB();
+ DatabaseDescriptor.setColumnIndexSizeInKiB(16);
+ try
+ {
+ assertCursorMatchesIteratorAcrossGenerations(cfs);
+ SSTableReader output = assertIndexedCursorOutput(cfs);
+ // 400 rows carrying a 202-byte value each serialize to at least 82 KiB and at most
+ // ~104 KiB, so the count is between 4 and 8 whatever the exact per-row overhead is. The
+ // bound is deliberately loose: the assertion that matters is pk 2's zero below.
+ int wide = blockCount(output, 1L);
+ assertTrue("pk 1 is over 80 KiB and must cut at least four 16 KiB blocks, got " + wide,
+ wide >= 4);
+ assertTrue("pk 1 is under 110 KiB and cannot cut more than eight 16 KiB blocks, got " + wide,
+ wide <= 8);
+ assertEquals("a 30-row partition crosses 4 KiB but not 16 KiB: a 0 here is what says the " +
+ "granularity change reached the writer at all",
+ 0, blockCount(output, 2L));
+ }
+ finally
+ {
+ DatabaseDescriptor.setColumnIndexSizeInKiB(previousGranularity);
+ }
+ }
+
/** {@code "c".repeat(n)}, spelled out because this suite targets a source level without it. */
private static String repeat(char c, int n)
{
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/HarryDifferentialCompactionTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/HarryDifferentialCompactionTest.java
index 28e1fd9c9fef..2086ae1854eb 100644
--- a/test/unit/org/apache/cassandra/db/compaction/differential/HarryDifferentialCompactionTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/HarryDifferentialCompactionTest.java
@@ -59,7 +59,7 @@ public void harryTombstoneHistories() throws Throwable
{
// withRandom rewraps every Throwable as an AssertionError. An assumption that fails inside
// the callback therefore reaches JUnit as a failure, not as a skip. This check runs first.
- assumeBigFormatSelected();
+ assumeCursorSupportedFormatSelected();
long seed = System.currentTimeMillis();
logger.info("harryTombstoneHistories seed={}", seed);
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/LeveledCompactionDifferentialTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/LeveledCompactionDifferentialTest.java
new file mode 100644
index 000000000000..f3980b5b3d70
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/LeveledCompactionDifferentialTest.java
@@ -0,0 +1,154 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+import java.util.Set;
+
+import org.junit.Test;
+
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.Directories;
+import org.apache.cassandra.db.compaction.LeveledCompactionTask;
+import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
+import org.apache.cassandra.db.compaction.writers.MajorLeveledCompactionWriter;
+import org.apache.cassandra.db.compaction.writers.MaxSSTableSizeWriter;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertTrue;
+
+/**
+ * LCS on the cursor path, through the real {@link LeveledCompactionTask}, which picks the writer and
+ * the output level. {@code MultiOutputDifferentialCompactionTest} builds a
+ * {@link MaxSSTableSizeWriter} by hand at level 0, so neither the task nor level assignment nor
+ * {@link MajorLeveledCompactionWriter} was covered against the cursor writer before this.
+ */
+public class LeveledCompactionDifferentialTest extends DifferentialCompactionTester
+{
+ /** The parameter cannot be named keepOriginals; see {@link TaskFactory}. */
+ private static TaskFactory leveled(int level, long maxSSTableBytes, boolean major, boolean retainOriginals)
+ {
+ return (cfs, txn, gcBefore) -> new LeveledCompactionTask(cfs, txn, level, gcBefore, maxSSTableBytes, major)
+ {
+ @Override
+ public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs,
+ Directories directories,
+ ILifecycleTransaction transaction,
+ Set nonExpiredSSTables)
+ {
+ if (major)
+ return new MajorLeveledCompactionWriter(cfs, directories, transaction, nonExpiredSSTables,
+ maxSSTableBytes, retainOriginals);
+ return new MaxSSTableSizeWriter(cfs, directories, transaction, nonExpiredSSTables,
+ maxSSTableBytes, getLevel(), retainOriginals);
+ }
+ };
+ }
+
+ /**
+ * Deliberately NOT on LeveledCompactionStrategy: the task carries the level and the size cap, and an
+ * LCS manifest would demote the second run's outputs to L0 through
+ * {@code LeveledGenerations.sendToL0} because the first run's outputs still occupy the level. That
+ * rewrites the level in the sstable metadata and shows up as a stats divergence that belongs to the
+ * harness, not to either compaction path.
+ */
+ private ColumnFamilyStore table() throws Throwable
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck bigint, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH compression = {'enabled': 'false'}");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+ return cfs;
+ }
+
+ /** Each partition holds about 1KB, so an 8KB cap splits the output several times. */
+ private void populate(int partitions, int rounds) throws Throwable
+ {
+ String padding = "x".repeat(100);
+ for (int round = 0; round < rounds; round++)
+ {
+ for (long pk = 0; pk < partitions; pk++)
+ for (long ck = 0; ck < 10; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", pk, ck, padding + round + "-" + ck);
+ flush();
+ }
+ }
+
+ @Test
+ public void maxSizeWriterAtLevelTwo() throws Throwable
+ {
+ ColumnFamilyStore cfs = table();
+ populate(40, 2);
+
+ CapturedOutput out = assertCursorMatchesIterator(cfs, cfs.getLiveSSTables(), leveled(2, 8 * 1024, false, true));
+ assertTrue("scenario must produce multiple outputs to test anything, got " + out.sstables.size(),
+ out.sstables.size() >= 2);
+ }
+
+ @Test
+ public void majorLeveledWriter() throws Throwable
+ {
+ ColumnFamilyStore cfs = table();
+ populate(40, 2);
+
+ CapturedOutput out = assertCursorMatchesIterator(cfs, cfs.getLiveSSTables(), leveled(0, 8 * 1024, true, true));
+ assertTrue("scenario must produce multiple outputs to test anything, got " + out.sstables.size(),
+ out.sstables.size() >= 2);
+ }
+
+ /**
+ * Level and size of the committed outputs, which the differential comparison cannot pin: both
+ * paths would have to get them wrong in the same way to still match.
+ */
+ @Test
+ public void committedOutputsCarryTheTaskLevelOnCursorPath() throws Throwable
+ {
+ assertCommittedOutputsCarryTheTaskLevel(true);
+ }
+
+ /** The same expectation on the iterator path, so a failure above is read as a cursor defect. */
+ @Test
+ public void committedOutputsCarryTheTaskLevelOnIteratorPath() throws Throwable
+ {
+ assertCommittedOutputsCarryTheTaskLevel(false);
+ }
+
+ private void assertCommittedOutputsCarryTheTaskLevel(boolean cursor) throws Throwable
+ {
+ ColumnFamilyStore cfs = table();
+ populate(40, 2);
+
+ long maxSSTableBytes = 8 * 1024;
+ commitThroughFactory(cfs, cursor, leveled(3, maxSSTableBytes, false, false));
+
+ Set outputs = cfs.getLiveSSTables();
+ assertTrue("scenario must produce multiple outputs to test anything, got " + outputs.size(),
+ outputs.size() >= 2);
+ long largest = 0;
+ for (SSTableReader sstable : outputs)
+ {
+ assertEquals("output was not written at the task's level", 3, sstable.getSSTableLevel());
+ largest = Math.max(largest, sstable.onDiskLength());
+ }
+ // The cap is honoured to within one partition, exactly as the iterator path overshoots it.
+ assertTrue("an output overshot the cap by more than one partition: " + largest,
+ largest <= maxSSTableBytes * 2);
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/RandomDifferentialCompactionTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/RandomDifferentialCompactionTest.java
index 35904e364966..eae98052efab 100644
--- a/test/unit/org/apache/cassandra/db/compaction/differential/RandomDifferentialCompactionTest.java
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/RandomDifferentialCompactionTest.java
@@ -37,9 +37,11 @@
import org.quicktheories.impl.JavaRandom;
import org.apache.cassandra.config.CassandraRelevantProperties;
+import org.apache.cassandra.config.DatabaseDescriptor;
import org.apache.cassandra.cql3.UntypedResultSet;
import org.apache.cassandra.db.ColumnFamilyStore;
import org.apache.cassandra.db.compaction.CursorCompactor;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
import org.apache.cassandra.schema.ColumnMetadata;
import org.apache.cassandra.schema.TableMetadata;
import org.apache.cassandra.utils.AbstractTypeGenerators;
@@ -70,16 +72,32 @@
* Plus one designated same-timestamp tie per example — one primary key rewritten once per round
* at a single timestamp — so the tie-break is reached without depending on the random draw.
*
+ * Partition shape: most rows go to freshly generated keys, but every example with a clustering also
+ * writes HUB partitions — one or two partition keys that hundreds of rows share, spread over the
+ * rounds — and draws a column_index_size for the compaction. Together those are what make the writer
+ * promote a row index at all; without them a partition holds a row or two, stays far below the
+ * granularity, and the BTI subclass of this test never builds a row trie. The run asserts at the end
+ * that some example actually reached that shape.
+ *
+ * Clustering width: usually 0-3 columns, one draw in four at 33-36, which is the first shape whose
+ * clustering needs a SECOND value header — ClusteringPrefix.Serializer writes headers in batches of
+ * 32.
+ *
* Example count is property-gated: -Dcassandra.test.differential.examples=N (default
- * {@value #DEFAULT_EXAMPLES}; a full validation run uses thousands).
+ * {@value #DEFAULT_EXAMPLES}; a full validation run uses thousands). Hub width is
+ * -Dcassandra.test.differential.hub_rows_per_round=N, the test's main runtime knob.
*
* Reproducing a failure: every failure message is wrapped in a seed; rerun with
* -Dcassandra.test.differential.seed=N (the failing seed becomes example 0), or plug it
* into {@code withFixedSeed} below.
*
* Known coverage gaps (deliberate, covered by the deterministic corpus): EMPTY_BYTES value
- * domain (invalid CQL for some generated multi-cell shapes), collection-element deletes
- * (DELETE m['k'] needs element values of the right type), expired TTLs (timing-dependent).
+ * domain (invalid CQL for some generated multi-cell shapes), collection-element operations
+ * (+=, -=, m[k] = v, DELETE m['k'] all need element values of the right type), expired TTLs
+ * (timing-dependent), a generation-2 differential over cursor-written inputs, and a partition-level
+ * deletion on an INDEXED partition — hub partitions are deliberately excluded from the partition-delete
+ * victim pool, because a partition delete collapses the wide partition and the example stops
+ * exercising the row index.
*/
public class RandomDifferentialCompactionTest extends DifferentialCompactionTester
{
@@ -113,6 +131,41 @@ public class RandomDifferentialCompactionTest extends DifferentialCompactionTest
/** Width of the pool the random workload draws from; the designated tie sits just above it. */
private static final int TIE_POOL_WIDTH = 3;
+ /**
+ * Upper bound of the per-example draw for rows written into a HUB partition per round; the floor is
+ * a quarter of it, so changing the property moves the whole range. This is the test's main runtime
+ * knob: every hub row is written once, dumped once per captured output, and probed by two slice
+ * seeks per captured output.
+ *
+ * The default is sized so that a hub partition crosses the low column_index_size draws below several
+ * times over. ROW COUNT is the only lever available here: the schema's column types are generated,
+ * so nothing in this file controls how many BYTES a row serializes to, and a hub of a designed byte
+ * size cannot be built without a value-size knob the generators do not expose.
+ */
+ private static final int HUB_ROWS_PER_ROUND_MAX =
+ CassandraRelevantProperties.TEST_DIFFERENTIAL_HUB_ROWS_PER_ROUND.getInt();
+ private static final int HUB_ROWS_PER_ROUND_MIN =
+ Math.min(HUB_ROWS_PER_ROUND_MAX, Math.max(2, HUB_ROWS_PER_ROUND_MAX / 4));
+
+ /**
+ * column_index_size values in KiB, one drawn per example, weighted low. Low granularities are what
+ * turn a hub partition of generated rows into several blocks rather than one; the high ones keep the
+ * "a partition just misses the boundary" shape in the space.
+ *
+ * 0 KiB is deliberately absent even though it would cut a block at every row and so guarantee a trie
+ * in every example. At 0 no block ever ACCUMULATES rows, which is the path the writer spends its time
+ * in, and the harness's block-count length bound ({@code partitionLength >= (blocks - 1) * granularity})
+ * degenerates to a tautology.
+ */
+ private static final int[] COLUMN_INDEX_SIZES_KIB = { 1, 1, 1, 2, 2, 4, 8, 16 };
+
+ /** Redraws allowed before a wide clustering that will not fit the key limit is treated as a failure. */
+ private static final int CLUSTERING_REDRAWS = 32;
+
+ /** Run-level index-block coverage; asserted at the end of {@link #randomizedDifferential}. */
+ private int examplesWithPromotedRowIndex;
+ private int promotedRowIndexPartitions;
+
@Test
public void randomizedDifferential() throws Throwable
{
@@ -123,6 +176,54 @@ public void randomizedDifferential() throws Throwable
assertTrue("the explicit-timestamp pool must sit above the wall clock: TIE_POOL_BASE=" +
TIE_POOL_BASE + " nowMicros=" + nowMicros, TIE_POOL_BASE > nowMicros);
new SeedRunner(EXAMPLES).run(this::runOneExample);
+
+ // A generative test that silently stops generating the shape it exists for is the failure mode
+ // this guards. No single example can be made to build a row index: the schema's column types are
+ // generated, so the serialized size of a row is unknown here and the block count is not a designed
+ // number. The RUN-level count is what can be asserted, and it is the only thing standing between
+ // BtiRandomDifferentialCompactionTest and covering nothing it claims to.
+ logger.info("{} of {} examples produced a partition with a promoted row index; {} indexed " +
+ "partitions in total", examplesWithPromotedRowIndex, EXAMPLES, promotedRowIndexPartitions);
+ assertTrue("no example produced a partition with a promoted row index over " + EXAMPLES +
+ " examples: every partition stayed below column_index_size, so BtiCursorIndexWriter " +
+ "took trieRoot -1 everywhere and no row trie was written or read back",
+ examplesWithPromotedRowIndex > 0);
+
+ // WHAT THIS STILL CANNOT SEE, after the widening:
+ // - a trie both paths build wrongly. The differential compares two writers against each other;
+ // only the harness's slice read-back is absolute, and it reads back through the same
+ // deserializer the writer serialized with, so it pins ROUTING, not encoding.
+ // - which block a given row landed in. Nothing here asserts a designed block boundary; the
+ // designed-shape scenarios live in EdgeCaseDifferentialCompactionTest.
+ // - a promoted index on a partition carrying a partition-level deletion (see the class javadoc).
+ // - the granularity the INPUT sstables were flushed at: it is left at the default on purpose, so
+ // the index under test is the one compaction built, never one copied out of an input.
+ }
+
+ /**
+ * A row whose clustering fits the 64KiB key limit ClusteringPrefix.validate enforces. The wide
+ * draw above puts 33-36 generated components in one clustering, and their sizes sum past the
+ * limit often enough to reject an insert; each component is legal, the total is not. Redraw
+ * rather than trim: a trimmed value is not necessarily legal for its type.
+ */
+ private static ByteBuffer[] generateRowWithLegalClustering(Gen dataGen,
+ JavaRandom qtRandom,
+ int partitionColumnCount,
+ int primaryColumnCount)
+ {
+ for (int attempt = 0; attempt < CLUSTERING_REDRAWS; attempt++)
+ {
+ ByteBuffer[] row = dataGen.generate(qtRandom);
+ int sum = 0;
+ for (int i = partitionColumnCount; i < primaryColumnCount; i++)
+ sum += row[i] == null ? 0 : row[i].remaining();
+ // a single component over the limit puts the sum over it too, so one check covers both
+ // of validate's rejections
+ if (sum <= FBUtilities.MAX_UNSIGNED_SHORT)
+ return row;
+ }
+ throw new AssertionError("no generated row in " + CLUSTERING_REDRAWS + " draws had a clustering " +
+ "under " + FBUtilities.MAX_UNSIGNED_SHORT + " bytes");
}
private void runOneExample(long seed) throws Throwable
@@ -130,11 +231,88 @@ private void runOneExample(long seed) throws Throwable
JavaRandom qtRandom = new JavaRandom(seed);
Random workload = new Random(seed);
+ TableMetadata metadata = generateSupportedMetadata(qtRandom, workload);
+
+ maybeCreateUDTs(metadata);
+ String createTableCql = metadata.toCqlString(true, false, false)
+ .replaceAll("org.apache.cassandra.db.marshal.", "");
+ logger.info("randomizedDifferential seed={} schema:\n{}", seed, createTableCql);
+ createTable(KEYSPACE, createTableCql);
+ // the CQL embeds the generator's table name; createTable's returned name is not it
+ ColumnFamilyStore cfs = getColumnFamilyStore(KEYSPACE, metadata.name);
+ cfs.disableAutoCompaction();
+
+ Example example = new Example(metadata, qtRandom, workload);
+ example.writeRounds();
+
+ // column_index_size is read when the compaction writer is CONSTRUCTED — BtiCursorIndexWriter's
+ // rowIndexBlockSize, and BIG's ColumnIndex — so it has to be in effect HERE, not when the schema
+ // was created. Leaving the flushes above on the default is deliberate: the promoted row index
+ // this example exercises is then the one COMPACTION built, never one copied out of an input.
+ // Restored on the failure path too, or one example's granularity would leak into the next.
+ int originalColumnIndexSizeKiB = DatabaseDescriptor.getColumnIndexSizeInKiB();
+ DatabaseDescriptor.setColumnIndexSizeInKiB(example.columnIndexSizeKiB);
+ try
+ {
+ assertCursorMatchesIterator(cfs);
+
+ // The tie candidates must be in DIFFERENT sstables. A memtable reconciles two writes to one key
+ // itself, so the compactor's tie-break is never reached inside one memtable. One tie write per
+ // round and one flush per round makes that structural. This assertion observes the flush half of
+ // it: autocompaction is off and an auto-flush can only ADD sstables, so moving the per-round flush
+ // to after the round loop fails here. Dropping the flush outright fails earlier, in compactPath's
+ // "scenario produced no input sstables". Hoisting the tie write out of the round loop is caught by
+ // the per-column assertion in assertDesignatedTieResolved, not by this one.
+ int inputSSTables = cfs.getLiveSSTables().size();
+ assertTrue("one flush per round must leave one input sstable per round, got " + inputSSTables +
+ " for " + example.rounds + " rounds", inputSSTables >= example.rounds);
+ // The differential restores its own inputs, so commit one real cursor compaction and read the tied
+ // row back out of it: querying before this would merge the inputs at READ time and say nothing
+ // about what compaction wrote.
+ commitCompaction(cfs, cfs.getLiveSSTables(), true, cfs.getDefaultGcBefore(FBUtilities.nowInSeconds()));
+
+ // Assert the soak reached the shape it now claims to reach. capture() already read every
+ // differential output back through slices, but it discards those captures and the count with
+ // them; this pass is over the COMMITTED output, which nothing else reads back. It runs
+ // regardless of cassandra.test.differential.slice_readback: that property lets a local run skip
+ // the harness's read-back, it does not make "the soak built an index at all" optional.
+ int indexedThisExample = 0;
+ for (SSTableReader output : cfs.getLiveSSTables())
+ indexedThisExample += assertEveryRowReadableThroughASlice(output);
+ promotedRowIndexPartitions += indexedThisExample;
+ if (indexedThisExample > 0)
+ examplesWithPromotedRowIndex++;
+ logger.info("seed={} column_index_size={}KiB clustering={} hubs={} hubRows/round={} -> {} indexed partitions",
+ seed, example.columnIndexSizeKiB, example.clusteringColumnCount, example.hubCount,
+ example.hubRowsPerRound, indexedThisExample);
+
+ example.assertDesignatedTieResolved();
+ }
+ finally
+ {
+ DatabaseDescriptor.setColumnIndexSizeInKiB(originalColumnIndexSizeKiB);
+ }
+ }
+
+ /**
+ * A random table restricted to the surface the cursor pipeline supports, redrawn until it passes
+ * the same filter production routes on.
+ */
+ private static TableMetadata generateSupportedMetadata(JavaRandom qtRandom, Random workload)
+ {
Gen udtName = Generators.unique(IDENTIFIER_GEN);
TypeGenBuilder safePrimary = AbstractTypeGenerators.withoutUnsafeEquality().withUDTNames(udtName);
TableMetadata metadata;
do
{
+ // Clustering width. Most examples keep the original 0-3: a wide clustering makes every
+ // statement and every generated row large and crowds the rest of the workload out of the
+ // runtime budget. One draw in four lands at 33-36, because ClusteringPrefix.Serializer
+ // writes its value headers in batches of 32, so 33 components is the first shape with a
+ // SECOND header vint — on the write side and in the cursor path's own read of it.
+ // Redrawn on every retry so the rejection loop below still terminates.
+ int clusteringColumns = workload.nextInt(4) == 0 ? 33 + workload.nextInt(4)
+ : workload.nextInt(4);
metadata = new TableMetadataBuilder()
.withKeyspaceName(KEYSPACE)
.withTableKinds(TableMetadata.Kind.REGULAR)
@@ -147,7 +325,7 @@ private void runOneExample(long seed) throws Throwable
.withUDTNames(udtName))
.withPartitionColumnsBetween(1, 2)
.withPrimaryColumnTypeGen(new TypeGenBuilder(safePrimary).withMaxDepth(1))
- .withClusteringColumnsBetween(0, 3)
+ .withClusteringColumnsCount(clusteringColumns)
.withRegularColumnsBetween(1, 5)
.withStaticColumnsBetween(0, 2)
.build(qtRandom);
@@ -158,258 +336,7 @@ private void runOneExample(long seed) throws Throwable
while (CursorCompactor.unsupportedMetadata(metadata)
|| (metadata.clusteringColumns().isEmpty() && !metadata.staticColumns().isEmpty()));
- maybeCreateUDTs(metadata);
- String createTableCql = metadata.toCqlString(true, false, false)
- .replaceAll("org.apache.cassandra.db.marshal.", "");
- logger.info("randomizedDifferential seed={} schema:\n{}", seed, createTableCql);
- createTable(KEYSPACE, createTableCql);
- // the CQL embeds the generator's table name; createTable's returned name is not it
- ColumnFamilyStore cfs = getColumnFamilyStore(KEYSPACE, metadata.name);
- cfs.disableAutoCompaction();
-
- // ~12% of non-key values are null: cell tombstones on simple columns. The data generator
- // maps null to an empty buffer on clustering columns, because null clustering is invalid and
- // empty is legal. That exercises the empty-vs-valued clustering comparison. The generator
- // never applies the domain to partition keys.
- Gen valueDomains = SourceDSL.integers().between(0, 99)
- .map(i -> i < 12 ? ValueDomain.NULL : ValueDomain.NORMAL);
- Gen dataGen = CassandraGenerators.data(metadata, valueDomains);
-
- int partitionColumnCount = metadata.partitionKeyColumns().size();
- int clusteringColumnCount = metadata.clusteringColumns().size();
- int primaryColumnCount = partitionColumnCount + clusteringColumnCount;
- String insertStmt = insertStmt(metadata);
- String deleteRowStmt = deleteStmt(metadata, primaryColumnCount);
- String deletePartitionStmt = deleteStmt(metadata, partitionColumnCount);
-
- // select-order index of every column, for UPDATE/cell-delete binding
- Map selectOrderIndex = new HashMap<>();
- {
- Iterator it = metadata.allColumnsInSelectOrder();
- for (int i = 0; it.hasNext(); i++)
- selectOrderIndex.put(it.next().name.toString(), i);
- }
- List regularColumns = ImmutableList.copyOf(metadata.regularColumns());
- List staticColumns = ImmutableList.copyOf(metadata.staticColumns());
-
- List rows = new ArrayList<>();
- int rounds = 2 + workload.nextInt(3); // 2-4 sstables
- // which round wins the designated tie; see the arrangement loop below
- int tieWinnerOffset = workload.nextInt(rounds);
-
- // One DETERMINISTIC cross-sstable same-timestamp tie per example: the same primary key written
- // once per round with fresh values at one timestamp. The tie-break is then reached in EVERY
- // example, instead of only when the random draw happens to collide on a key. Its timestamp sits
- // just above the pool the random workload draws from, so a pooled write that lands on the same
- // key loses on timestamp rather than joining the tie; low cardinality key types make such a write
- // likely. The whole pool sits above the wall clock, so no wall-clock write or delete can shadow
- // the tie either. The tie row never enters `rows`, so none of the delete loops below can target it.
- long tieTimestamp = TIE_POOL_BASE + TIE_POOL_WIDTH;
- // no value domain: every tie candidate is a live cell, so the value comparison alone decides
- // the winner, not the tombstone-or-expiring-beats-live branch of
- // CellLivenessInfo.resolveSameTimestampTie on a NULL column
- Gen tieDataGen = CassandraGenerators.data(metadata, null);
- ByteBuffer[] tieKey = tieDataGen.generate(qtRandom);
- List tieWrites = new ArrayList<>();
- for (int round = 0; round < rounds; round++)
- {
- ByteBuffer[] tieRow = tieDataGen.generate(qtRandom);
- System.arraycopy(tieKey, 0, tieRow, 0, primaryColumnCount);
- tieWrites.add(tieRow);
- }
- // Spread the winners ACROSS rounds: column c's greatest bytes are arranged into round
- // (c + tieWinnerOffset) % rounds. Leaving the maxima wherever the generator put them makes the
- // assertion below bite only by luck, because a merge that kept the last writer agrees with the
- // rule about 1/rounds of the time per column. Putting them all in ONE round would instead make a
- // merge that kept that round pass every time. With the spread, no rule of the form "always keep
- // sstable k" can agree with the real rule on a schema with two or more regular columns. The
- // per-example offset means such a rule cannot agree on a single-regular-column schema either
- // without getting lucky in every example. The offset is drawn from `workload`, so a pinned seed
- // still reproduces the arrangement.
- for (int c = 0; c < regularColumns.size(); c++)
- {
- int valueIndex = selectOrderIndex.get(regularColumns.get(c).name.toString());
- int winner = (c + tieWinnerOffset) % rounds;
- for (int round = 0; round < rounds; round++)
- {
- if (round != winner
- && ByteBufferUtil.compareUnsigned(tieWrites.get(round)[valueIndex],
- tieWrites.get(winner)[valueIndex]) > 0)
- {
- ByteBuffer greater = tieWrites.get(round)[valueIndex];
- tieWrites.get(round)[valueIndex] = tieWrites.get(winner)[valueIndex];
- tieWrites.get(winner)[valueIndex] = greater;
- }
- }
- }
-
- for (int round = 0; round < rounds; round++)
- {
- // Watermark taken at ROUND start: the explicit-timestamp branch below draws its overwrite
- // target only from `rows` BELOW this index, that is, from a key written in an EARLIER round.
- // Every iteration appends to `rows`, including earlier iterations of this round. Drawing from
- // all of `rows` therefore lets both writes land in one memtable. The memtable reconciles them
- // and the compactor's tie-break is never reached.
- int rowsBeforeThisRound = rows.size();
- int inserts = 15 + workload.nextInt(26); // 15-40 rows
- for (int i = 0; i < inserts; i++)
- {
- ByteBuffer[] row = dataGen.generate(qtRandom);
- boolean overwrite = !rows.isEmpty() && workload.nextInt(100) < 30;
- if (overwrite)
- {
- // overwrite: keep a previously used primary key, fresh non-key values —
- // this is what makes the merge actually reconcile rather than concatenate
- ByteBuffer[] prev = rows.get(workload.nextInt(rows.size()));
- System.arraycopy(prev, 0, row, 0, primaryColumnCount);
- }
-
- int mode = workload.nextInt(100);
- if (overwrite && rowsBeforeThisRound > 0 && workload.nextInt(100) < 40)
- {
- // explicit-timestamp collision candidate: re-keyed onto a row from an EARLIER round,
- // and stamped from a pool offset DERIVED from the primary key. Two pooled writes to
- // one key therefore land on the SAME timestamp, rather than merely being ordered by
- // it. The pair that ties is two pooled writes. The earlier-round target only
- // guarantees this write lands in a later sstable than the row it re-keys onto, which
- // is what lets a tie form across sstables instead of inside one memtable. Guaranteed
- // coverage comes from the designated tie below, not from this draw.
- ByteBuffer[] prev = rows.get(workload.nextInt(rowsBeforeThisRound));
- System.arraycopy(prev, 0, row, 0, primaryColumnCount);
- long ts = TIE_POOL_BASE + Math.floorMod(primaryKeyHash(row, primaryColumnCount), TIE_POOL_WIDTH);
- execute(insertStmt + " USING TIMESTAMP " + ts, (Object[]) row);
- }
- else if (mode < 15 && !regularColumns.isEmpty())
- {
- // UPDATE: writes cells without primary-key liveness (different row flags)
- execute(updateStmt(metadata, regularColumns),
- updateParams(row, regularColumns, selectOrderIndex, primaryColumnCount));
- }
- else if (mode < 22)
- {
- // primary-key-only INSERT: row liveness with zero cells
- execute(pkOnlyInsertStmt(metadata), (Object[]) Arrays.copyOf(row, primaryColumnCount));
- }
- else if (mode < 30)
- {
- // long TTL: liveness info with ttl + expiration far from the runs
- execute(insertStmt + " USING TTL " + SOAK_TTL_SECONDS, (Object[]) row);
- }
- else
- {
- execute(insertStmt, (Object[]) row);
- }
- rows.add(row);
- }
-
- // row deletes against known keys
- for (int i = 0; i < 3 && !rows.isEmpty(); i++)
- {
- ByteBuffer[] victim = rows.get(workload.nextInt(rows.size()));
- execute(deleteRowStmt, (Object[]) Arrays.copyOf(victim, primaryColumnCount));
- }
-
- // range deletes (clustering tables only): single-sided slices and clustering-prefix
- // deletes against known keys; single-sided bounds cannot produce inverted ranges
- for (int i = 0; i < 2 && clusteringColumnCount > 0 && !rows.isEmpty(); i++)
- {
- ByteBuffer[] victim = rows.get(workload.nextInt(rows.size()));
- if (clusteringColumnCount >= 2 && workload.nextBoolean())
- {
- // prefix delete: equality on a strict prefix of the clustering columns
- int depth = 1 + workload.nextInt(clusteringColumnCount - 1);
- execute(deleteStmt(metadata, partitionColumnCount + depth),
- (Object[]) Arrays.copyOf(victim, partitionColumnCount + depth));
- }
- else
- {
- int eqDepth = workload.nextInt(clusteringColumnCount);
- String op = new String[]{ ">=", ">", "<=", "<" }[workload.nextInt(4)];
- execute(rangeDeleteStmt(metadata, eqDepth, op),
- (Object[]) Arrays.copyOf(victim, partitionColumnCount + eqDepth + 1));
- }
- }
-
- // cell deletes: random subset of regular columns at a known row; occasionally a
- // static cell delete instead
- for (int i = 0; i < 2 && !rows.isEmpty(); i++)
- {
- ByteBuffer[] victim = rows.get(workload.nextInt(rows.size()));
- if (!staticColumns.isEmpty() && workload.nextInt(100) < 30)
- {
- ColumnMetadata col = staticColumns.get(workload.nextInt(staticColumns.size()));
- execute(cellDeleteStmt(metadata, List.of(col), partitionColumnCount),
- (Object[]) Arrays.copyOf(victim, partitionColumnCount));
- }
- else
- {
- List subset = randomSubset(regularColumns, workload);
- execute(cellDeleteStmt(metadata, subset, primaryColumnCount),
- (Object[]) Arrays.copyOf(victim, primaryColumnCount));
- }
- }
-
- // occasional partition delete
- if (workload.nextInt(100) < 40 && !rows.isEmpty())
- {
- ByteBuffer[] victim = rows.get(workload.nextInt(rows.size()));
- execute(deletePartitionStmt, (Object[]) Arrays.copyOf(victim, partitionColumnCount));
- }
-
- // the designated tie: the same primary key, the same timestamp, fresh values, once per round
- execute(insertStmt + " USING TIMESTAMP " + tieTimestamp, (Object[]) tieWrites.get(round));
-
- flush(KEYSPACE, metadata.name);
- }
-
- assertCursorMatchesIterator(cfs);
-
- // The tie candidates must be in DIFFERENT sstables. A memtable reconciles two writes to one key
- // itself, so the compactor's tie-break is never reached inside one memtable. One tie write per
- // round and one flush per round makes that structural. This assertion observes the flush half of
- // it: autocompaction is off and an auto-flush can only ADD sstables, so moving the per-round flush
- // to after the round loop fails here. Dropping the flush outright fails earlier, in compactPath's
- // "scenario produced no input sstables". Hoisting the tie write out of the round loop is caught by
- // the per-column assertion below, not by this one.
- int inputSSTables = cfs.getLiveSSTables().size();
- assertTrue("one flush per round must leave one input sstable per round, got " + inputSSTables +
- " for " + rounds + " rounds", inputSSTables >= rounds);
- // The differential restores its own inputs, so commit one real cursor compaction and read the tied
- // row back out of it: querying before this would merge the inputs at READ time and say nothing
- // about what compaction wrote.
- commitCompaction(cfs, cfs.getLiveSSTables(), true, cfs.getDefaultGcBefore(FBUtilities.nowInSeconds()));
- UntypedResultSet tieResult = execute(selectStmt(metadata, regularColumns),
- (Object[]) Arrays.copyOf(tieKey, primaryColumnCount));
- assertEquals("the designated tie row must survive compaction: it is written above the wall clock, " +
- "so no wall-clock write or delete in this example can shadow it", 1, tieResult.size());
- UntypedResultSet.Row survivor = tieResult.one();
- // For a simple column, the cell with the greater value bytes wins a tie of equal
- // timestamps. This is the last rule of resolveRegular, and it is an unsigned comparison of
- // the whole value.
- //
- // The loop below skips the complex columns. A tie on a complex column resolves separately
- // for each cell path, and the result holds the paths of both writes. It does not compare the
- // whole value of one write against the whole value of the other, which is the only rule this
- // test models. The EdgeCase and Pathological tests cover ties on complex columns.
- //
- // The arrangement loop above put each column's winner in a different round, so a merge that
- // resolved this tie by write order fails on any column whose candidate values are not all
- // equal. A low-cardinality column type can make them equal, which costs an example rather
- // than producing a false failure. Byte equality between the two paths cannot see a rule they
- // both get wrong.
- for (ColumnMetadata col : regularColumns)
- {
- if (col.isComplex())
- continue;
- int valueIndex = selectOrderIndex.get(col.name.toString());
- ByteBuffer expected = tieWrites.get(0)[valueIndex];
- for (ByteBuffer[] candidate : tieWrites)
- if (ByteBufferUtil.compareUnsigned(candidate[valueIndex], expected) > 0)
- expected = candidate[valueIndex];
- assertEquals("the greater raw value bytes must win the same-timestamp tie on " + col.name +
- " (" + tieWrites.size() + " candidates at timestamp " + tieTimestamp + ')',
- expected, survivor.getBytes(col.name.toString()));
- }
+ return metadata;
}
/** Appends {@code columns[i].name}, joined by separator. */
@@ -556,6 +483,413 @@ private static int primaryKeyHash(ByteBuffer[] row, int primaryColumnCount)
return Arrays.hashCode(Arrays.copyOf(row, primaryColumnCount));
}
+ /**
+ * Whether this row's partition key is one of the hub keys, i.e. whether it lands in the wide partition.
+ *
+ * Same stability argument as {@link #primaryKeyHash}: {@code ByteBuffer.equals} is content-based but
+ * position-relative, and holds only because CQLTester pre-converts every bound parameter, so the
+ * buffers here are never consumed.
+ */
+ private static boolean inHubPartition(ByteBuffer[] row, List hubKeys, int partitionColumnCount)
+ {
+ for (ByteBuffer[] hub : hubKeys)
+ {
+ boolean same = true;
+ for (int i = 0; i < partitionColumnCount && same; i++)
+ same = row[i].equals(hub[i]);
+ if (same)
+ return true;
+ }
+ return false;
+ }
+
+ /**
+ * One example's schema-derived statements and generated workload, from the per-example draws made
+ * once here down to the per-round writes.
+ *
+ * Every draw below is taken in the order the single round loop took it. The two generators are
+ * seeded from the example's seed, so reordering a draw within either stream changes every generated
+ * example and a pinned seed stops reproducing.
+ */
+ private final class Example
+ {
+ private final TableMetadata metadata;
+ private final JavaRandom qtRandom;
+ private final Random workload;
+ private final Gen dataGen;
+
+ private final int partitionColumnCount;
+ private final int clusteringColumnCount;
+ private final int primaryColumnCount;
+
+ private final String insertStmt;
+ private final String deleteRowStmt;
+ private final String deletePartitionStmt;
+
+ /** select-order index of every column, for UPDATE/cell-delete binding */
+ private final Map selectOrderIndex = new HashMap<>();
+ private final List regularColumns;
+ private final List staticColumns;
+
+ private final List rows = new ArrayList<>();
+ // Partition-delete victims are drawn from HERE, not from `rows`. A partition delete on a hub
+ // collapses the wide partition to a single tombstone and the example stops exercising the row
+ // index for the rest of the run. Row, range and cell deletes still draw from `rows`, so they do
+ // land inside a hub partition — which is where a range tombstone spanning index blocks, and the
+ // open-marker carried on a block boundary, come from.
+ private final List nonHubRows = new ArrayList<>();
+
+ private final int rounds;
+ /** which round wins the designated tie; see {@link #arrangeTieWinnersAcrossRounds} */
+ private final int tieWinnerOffset;
+ private final long tieTimestamp;
+ private final ByteBuffer[] tieKey;
+ private final List tieWrites = new ArrayList<>();
+
+ private final int hubCount;
+ // a whole generated row, of which only the leading partitionColumnCount components are ever read:
+ // dataGen is the only thing that knows how to produce a legal value per key column type
+ private final List hubKeys = new ArrayList<>();
+ private final int hubRowsPerRound;
+ private final int columnIndexSizeKiB;
+
+ Example(TableMetadata metadata, JavaRandom qtRandom, Random workload)
+ {
+ this.metadata = metadata;
+ this.qtRandom = qtRandom;
+ this.workload = workload;
+
+ // ~12% of non-key values are null: cell tombstones on simple columns. The data generator
+ // maps null to an empty buffer on clustering columns, because null clustering is invalid and
+ // empty is legal. That exercises the empty-vs-valued clustering comparison. The generator
+ // never applies the domain to partition keys.
+ Gen valueDomains = SourceDSL.integers().between(0, 99)
+ .map(i -> i < 12 ? ValueDomain.NULL : ValueDomain.NORMAL);
+ this.dataGen = CassandraGenerators.data(metadata, valueDomains);
+
+ this.partitionColumnCount = metadata.partitionKeyColumns().size();
+ this.clusteringColumnCount = metadata.clusteringColumns().size();
+ this.primaryColumnCount = partitionColumnCount + clusteringColumnCount;
+ this.insertStmt = insertStmt(metadata);
+ this.deleteRowStmt = deleteStmt(metadata, primaryColumnCount);
+ this.deletePartitionStmt = deleteStmt(metadata, partitionColumnCount);
+
+ Iterator it = metadata.allColumnsInSelectOrder();
+ for (int i = 0; it.hasNext(); i++)
+ selectOrderIndex.put(it.next().name.toString(), i);
+ this.regularColumns = ImmutableList.copyOf(metadata.regularColumns());
+ this.staticColumns = ImmutableList.copyOf(metadata.staticColumns());
+
+ this.rounds = 2 + workload.nextInt(3); // 2-4 sstables
+ this.tieWinnerOffset = workload.nextInt(rounds);
+
+ // One DETERMINISTIC cross-sstable same-timestamp tie per example: the same primary key written
+ // once per round with fresh values at one timestamp. The tie-break is then reached in EVERY
+ // example, instead of only when the random draw happens to collide on a key. Its timestamp sits
+ // just above the pool the random workload draws from, so a pooled write that lands on the same
+ // key loses on timestamp rather than joining the tie; low cardinality key types make such a write
+ // likely. The whole pool sits above the wall clock, so no wall-clock write or delete can shadow
+ // the tie either. The tie row never enters `rows`, so none of the delete loops below can target it.
+ this.tieTimestamp = TIE_POOL_BASE + TIE_POOL_WIDTH;
+ // no value domain: every tie candidate is a live cell, so the value comparison alone decides
+ // the winner, not the tombstone-or-expiring-beats-live branch of
+ // CellLivenessInfo.resolveSameTimestampTie on a NULL column
+ Gen tieDataGen = CassandraGenerators.data(metadata, null);
+ this.tieKey = tieDataGen.generate(qtRandom);
+ for (int round = 0; round < rounds; round++)
+ {
+ ByteBuffer[] tieRow = tieDataGen.generate(qtRandom);
+ System.arraycopy(tieKey, 0, tieRow, 0, primaryColumnCount);
+ tieWrites.add(tieRow);
+ }
+ arrangeTieWinnersAcrossRounds();
+
+ // HUB PARTITIONS. Left to itself this workload writes 15-40 rows per round across freshly
+ // generated keys, so a partition holds a row or two, never reaches column_index_size, and no row
+ // index is promoted anywhere: BtiCursorIndexWriter.endPartition takes trieRoot -1 on every
+ // partition and the row trie the BTI subclass exists to cover is never built. A hub is a
+ // partition key that many rows share — the same "copy fewer key columns from an earlier row" move
+ // the overwrite branch below makes, stopping at the PARTITION key so the clustering stays fresh.
+ //
+ // Written in EVERY round, so the wide partition's rows arrive spread across sstables and the block
+ // boundaries in the output are cut by the MERGE rather than copied out of one input.
+ //
+ // Skipped without a clustering: one row per partition then, so no partition is ever indexable.
+ this.hubCount = clusteringColumnCount == 0 ? 0 : 1 + workload.nextInt(2);
+ for (int i = 0; i < hubCount; i++)
+ hubKeys.add(dataGen.generate(qtRandom));
+ this.hubRowsPerRound = hubCount == 0 ? 0
+ : HUB_ROWS_PER_ROUND_MIN
+ + workload.nextInt(HUB_ROWS_PER_ROUND_MAX - HUB_ROWS_PER_ROUND_MIN + 1);
+ this.columnIndexSizeKiB = COLUMN_INDEX_SIZES_KIB[workload.nextInt(COLUMN_INDEX_SIZES_KIB.length)];
+ }
+
+ /**
+ * Spread the winners ACROSS rounds: column c's greatest bytes are arranged into round
+ * (c + tieWinnerOffset) % rounds. Leaving the maxima wherever the generator put them makes the
+ * assertion in {@link #assertDesignatedTieResolved} bite only by luck, because a merge that kept the
+ * last writer agrees with the rule about 1/rounds of the time per column. Putting them all in ONE
+ * round would instead make a merge that kept that round pass every time. With the spread, no rule of
+ * the form "always keep sstable k" can agree with the real rule on a schema with two or more regular
+ * columns. The per-example offset means such a rule cannot agree on a single-regular-column schema
+ * either without getting lucky in every example. The offset is drawn from `workload`, so a pinned
+ * seed still reproduces the arrangement.
+ */
+ private void arrangeTieWinnersAcrossRounds()
+ {
+ for (int c = 0; c < regularColumns.size(); c++)
+ {
+ int valueIndex = selectOrderIndex.get(regularColumns.get(c).name.toString());
+ int winner = (c + tieWinnerOffset) % rounds;
+ for (int round = 0; round < rounds; round++)
+ {
+ if (round != winner
+ && ByteBufferUtil.compareUnsigned(tieWrites.get(round)[valueIndex],
+ tieWrites.get(winner)[valueIndex]) > 0)
+ {
+ ByteBuffer greater = tieWrites.get(round)[valueIndex];
+ tieWrites.get(round)[valueIndex] = tieWrites.get(winner)[valueIndex];
+ tieWrites.get(winner)[valueIndex] = greater;
+ }
+ }
+ }
+ }
+
+ /** The whole workload: one round per input sstable, each round flushed. */
+ void writeRounds() throws Throwable
+ {
+ for (int round = 0; round < rounds; round++)
+ {
+ writeInserts();
+ writeHubRows();
+ writeRowDeletes();
+ writeRangeDeletes();
+ writeCellDeletes();
+ maybeWritePartitionDelete();
+
+ // the designated tie: the same primary key, the same timestamp, fresh values, once per round
+ execute(insertStmt + " USING TIMESTAMP " + tieTimestamp, (Object[]) tieWrites.get(round));
+
+ flush(KEYSPACE, metadata.name);
+ }
+ }
+
+ private void writeInserts() throws Throwable
+ {
+ // Watermark taken at ROUND start: the explicit-timestamp branch below draws its overwrite
+ // target only from `rows` BELOW this index, that is, from a key written in an EARLIER round.
+ // Every iteration appends to `rows`, including earlier iterations of this round. Drawing from
+ // all of `rows` therefore lets both writes land in one memtable. The memtable reconciles them
+ // and the compactor's tie-break is never reached.
+ int rowsBeforeThisRound = rows.size();
+ int inserts = 15 + workload.nextInt(26); // 15-40 rows
+ for (int i = 0; i < inserts; i++)
+ writeOneInsert(rowsBeforeThisRound);
+ }
+
+ private void writeOneInsert(int rowsBeforeThisRound) throws Throwable
+ {
+ ByteBuffer[] row = generateRowWithLegalClustering(dataGen, qtRandom,
+ partitionColumnCount, primaryColumnCount);
+ boolean overwrite = !rows.isEmpty() && workload.nextInt(100) < 30;
+ // A SEPARATE draw, so the full-primary-key overwrite above keeps exactly the rate it had:
+ // that draw is what makes the merge reconcile rather than concatenate, and the
+ // explicit-timestamp collision branch below is gated on it. This one keeps only the
+ // PARTITION key and leaves the generated clustering, putting a second and third row into an
+ // ordinary partition — the shapes either side of the writer's "one block is not an index"
+ // decision. Reaching MANY blocks is the hub loop's job, not this draw's.
+ boolean sharePartition = !overwrite && !rows.isEmpty() && clusteringColumnCount > 0
+ && workload.nextInt(100) < 20;
+ if (overwrite)
+ {
+ // overwrite: keep a previously used primary key, fresh non-key values —
+ // this is what makes the merge actually reconcile rather than concatenate
+ ByteBuffer[] prev = rows.get(workload.nextInt(rows.size()));
+ System.arraycopy(prev, 0, row, 0, primaryColumnCount);
+ }
+ else if (sharePartition)
+ {
+ ByteBuffer[] prev = rows.get(workload.nextInt(rows.size()));
+ System.arraycopy(prev, 0, row, 0, partitionColumnCount);
+ }
+
+ int mode = workload.nextInt(100);
+ writeGeneratedRow(row, mode, overwrite, rowsBeforeThisRound);
+ rows.add(row);
+ if (!inHubPartition(row, hubKeys, partitionColumnCount))
+ nonHubRows.add(row);
+ }
+
+ /**
+ * The statement one generated row is written with: the explicit-timestamp collision, or one of
+ * the four shapes {@code mode} selects between.
+ */
+ private void writeGeneratedRow(ByteBuffer[] row, int mode, boolean overwrite, int rowsBeforeThisRound)
+ throws Throwable
+ {
+ if (overwrite && rowsBeforeThisRound > 0 && workload.nextInt(100) < 40)
+ {
+ // explicit-timestamp collision candidate: re-keyed onto a row from an EARLIER round,
+ // and stamped from a pool offset DERIVED from the primary key. Two pooled writes to
+ // one key therefore land on the SAME timestamp, rather than merely being ordered by
+ // it. The pair that ties is two pooled writes. The earlier-round target only
+ // guarantees this write lands in a later sstable than the row it re-keys onto, which
+ // is what lets a tie form across sstables instead of inside one memtable. Guaranteed
+ // coverage comes from the designated tie below, not from this draw.
+ ByteBuffer[] prev = rows.get(workload.nextInt(rowsBeforeThisRound));
+ System.arraycopy(prev, 0, row, 0, primaryColumnCount);
+ long ts = TIE_POOL_BASE + Math.floorMod(primaryKeyHash(row, primaryColumnCount), TIE_POOL_WIDTH);
+ execute(insertStmt + " USING TIMESTAMP " + ts, (Object[]) row);
+ }
+ else if (mode < 15 && !regularColumns.isEmpty())
+ {
+ // UPDATE: writes cells without primary-key liveness (different row flags)
+ execute(updateStmt(metadata, regularColumns),
+ updateParams(row, regularColumns, selectOrderIndex, primaryColumnCount));
+ }
+ else if (mode < 22)
+ {
+ // primary-key-only INSERT: row liveness with zero cells
+ execute(pkOnlyInsertStmt(metadata), (Object[]) Arrays.copyOf(row, primaryColumnCount));
+ }
+ else if (mode < 30)
+ {
+ // long TTL: liveness info with ttl + expiration far from the runs
+ execute(insertStmt + " USING TTL " + SOAK_TTL_SECONDS, (Object[]) row);
+ }
+ else
+ {
+ execute(insertStmt, (Object[]) row);
+ }
+ }
+
+ /**
+ * The hub rows: one shared partition key, fresh clustering and values. Written before the
+ * delete loops so this round's deletes can already target them.
+ */
+ private void writeHubRows() throws Throwable
+ {
+ for (int i = 0; i < hubRowsPerRound; i++)
+ {
+ ByteBuffer[] row = generateRowWithLegalClustering(dataGen, qtRandom,
+ partitionColumnCount, primaryColumnCount);
+ System.arraycopy(hubKeys.get(workload.nextInt(hubKeys.size())), 0, row, 0, partitionColumnCount);
+ execute(insertStmt, (Object[]) row);
+ rows.add(row);
+ }
+ }
+
+ /** row deletes against known keys */
+ private void writeRowDeletes() throws Throwable
+ {
+ for (int i = 0; i < 3 && !rows.isEmpty(); i++)
+ {
+ ByteBuffer[] victim = rows.get(workload.nextInt(rows.size()));
+ execute(deleteRowStmt, (Object[]) Arrays.copyOf(victim, primaryColumnCount));
+ }
+ }
+
+ /**
+ * range deletes (clustering tables only): single-sided slices and clustering-prefix
+ * deletes against known keys; single-sided bounds cannot produce inverted ranges
+ */
+ private void writeRangeDeletes() throws Throwable
+ {
+ for (int i = 0; i < 2 && clusteringColumnCount > 0 && !rows.isEmpty(); i++)
+ {
+ ByteBuffer[] victim = rows.get(workload.nextInt(rows.size()));
+ if (clusteringColumnCount >= 2 && workload.nextBoolean())
+ {
+ // prefix delete: equality on a strict prefix of the clustering columns
+ int depth = 1 + workload.nextInt(clusteringColumnCount - 1);
+ execute(deleteStmt(metadata, partitionColumnCount + depth),
+ (Object[]) Arrays.copyOf(victim, partitionColumnCount + depth));
+ }
+ else
+ {
+ int eqDepth = workload.nextInt(clusteringColumnCount);
+ String op = new String[]{ ">=", ">", "<=", "<" }[workload.nextInt(4)];
+ execute(rangeDeleteStmt(metadata, eqDepth, op),
+ (Object[]) Arrays.copyOf(victim, partitionColumnCount + eqDepth + 1));
+ }
+ }
+ }
+
+ /**
+ * cell deletes: random subset of regular columns at a known row; occasionally a
+ * static cell delete instead
+ */
+ private void writeCellDeletes() throws Throwable
+ {
+ for (int i = 0; i < 2 && !rows.isEmpty(); i++)
+ {
+ ByteBuffer[] victim = rows.get(workload.nextInt(rows.size()));
+ if (!staticColumns.isEmpty() && workload.nextInt(100) < 30)
+ {
+ ColumnMetadata col = staticColumns.get(workload.nextInt(staticColumns.size()));
+ execute(cellDeleteStmt(metadata, List.of(col), partitionColumnCount),
+ (Object[]) Arrays.copyOf(victim, partitionColumnCount));
+ }
+ else
+ {
+ List subset = randomSubset(regularColumns, workload);
+ execute(cellDeleteStmt(metadata, subset, primaryColumnCount),
+ (Object[]) Arrays.copyOf(victim, primaryColumnCount));
+ }
+ }
+ }
+
+ /** occasional partition delete; hub partitions are excluded, see nonHubRows above */
+ private void maybeWritePartitionDelete() throws Throwable
+ {
+ if (workload.nextInt(100) < 40 && !nonHubRows.isEmpty())
+ {
+ ByteBuffer[] victim = nonHubRows.get(workload.nextInt(nonHubRows.size()));
+ execute(deletePartitionStmt, (Object[]) Arrays.copyOf(victim, partitionColumnCount));
+ }
+ }
+
+ /**
+ * Reads the designated tie row back out of the committed compaction and asserts the greater raw
+ * value bytes won on every simple regular column.
+ */
+ void assertDesignatedTieResolved() throws Throwable
+ {
+ UntypedResultSet tieResult = execute(selectStmt(metadata, regularColumns),
+ (Object[]) Arrays.copyOf(tieKey, primaryColumnCount));
+ assertEquals("the designated tie row must survive compaction: it is written above the wall clock, " +
+ "so no wall-clock write or delete in this example can shadow it", 1, tieResult.size());
+ UntypedResultSet.Row survivor = tieResult.one();
+ // For a simple column, the cell with the greater value bytes wins a tie of equal
+ // timestamps. This is the last rule of resolveRegular, and it is an unsigned comparison of
+ // the whole value.
+ //
+ // The loop below skips the complex columns. A tie on a complex column resolves separately
+ // for each cell path, and the result holds the paths of both writes. It does not compare the
+ // whole value of one write against the whole value of the other, which is the only rule this
+ // test models. The EdgeCase and Pathological tests cover ties on complex columns.
+ //
+ // The arrangement loop above put each column's winner in a different round, so a merge that
+ // resolved this tie by write order fails on any column whose candidate values are not all
+ // equal. A low-cardinality column type can make them equal, which costs an example rather
+ // than producing a false failure. Byte equality between the two paths cannot see a rule they
+ // both get wrong.
+ for (ColumnMetadata col : regularColumns)
+ {
+ if (col.isComplex())
+ continue;
+ int valueIndex = selectOrderIndex.get(col.name.toString());
+ ByteBuffer expected = tieWrites.get(0)[valueIndex];
+ for (ByteBuffer[] candidate : tieWrites)
+ if (ByteBufferUtil.compareUnsigned(candidate[valueIndex], expected) > 0)
+ expected = candidate[valueIndex];
+ assertEquals("the greater raw value bytes must win the same-timestamp tie on " + col.name +
+ " (" + tieWrites.size() + " candidates at timestamp " + tieTimestamp + ')',
+ expected, survivor.getBytes(col.name.toString()));
+ }
+ }
+ }
+
/** Seed chaining copied from RandomSchemaTest so failures reproduce the same way. */
private static final class SeedRunner
{
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/ReversedClusteringDifferentialCompactionTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/ReversedClusteringDifferentialCompactionTest.java
new file mode 100644
index 000000000000..2ee3626e644b
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/ReversedClusteringDifferentialCompactionTest.java
@@ -0,0 +1,143 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+import org.junit.Test;
+
+import org.apache.cassandra.db.ColumnFamilyStore;
+
+/**
+ * Covers the third {@code updateClusteringValues} overload, the one only the cursor path reaches.
+ *
+ * The iterator path feeds {@code MetadataCollector.updateClusteringValues} a {@code Clustering} or a
+ * {@code ClusteringBound}. The cursor path feeds it a {@code ClusteringDescriptor}
+ * ({@code MetadataCollector.updateClusteringValues(ClusteringDescriptor)}), which compares through
+ * {@code ClusteringComparator.compare(ClusteringDescriptor, ClusteringDescriptor)}. Those two
+ * comparison routines are separate code, so agreeing on ordinary rows says nothing about agreeing
+ * where ordering is inverted and where bound kinds decide the result.
+ *
+ * Reversed clustering is where a sign error hides: {@code ReversedType} inverts the component
+ * comparison, so a min and a max that the iterator path assigns one way get swapped, and
+ * {@code coveredClustering} in Statistics.db comes out reversed. Nothing else in the differential
+ * corpus uses {@code CLUSTERING ORDER BY ... DESC} together with range bounds.
+ *
+ * Bound kinds matter for the same reason: an inclusive and an exclusive bound over the same value
+ * differ only in {@code ClusteringPrefix.Kind}, which the descriptor overload compares through its
+ * own {@code clusteringKind()} path rather than the one the iterator path uses.
+ *
+ * The harness byte-compares Statistics.db, so a divergence in either min or max fails here.
+ */
+public class ReversedClusteringDifferentialCompactionTest extends DifferentialCompactionTester
+{
+ /** Rows only, reversed order. Establishes the baseline before bounds enter the picture. */
+ @Test
+ public void reversedClusteringRows() throws Throwable
+ {
+ createTable("CREATE TABLE %s (pk int, ck int, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH CLUSTERING ORDER BY (ck DESC)");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ for (int ck = 0; ck < 40; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", 1, ck, "a" + ck);
+ flush();
+ for (int ck = 20; ck < 60; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", 1, ck, "b" + ck);
+ flush();
+
+ assertCursorMatchesIterator(cfs);
+ }
+
+ /**
+ * Range deletions with every combination of inclusive and exclusive on both sides, under
+ * reversed order. Each one writes a pair of markers whose kinds are what the descriptor overload
+ * must compare correctly.
+ */
+ @Test
+ public void reversedClusteringWithBoundKinds() throws Throwable
+ {
+ createTable("CREATE TABLE %s (pk int, ck int, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH CLUSTERING ORDER BY (ck DESC)");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ for (int ck = 0; ck < 80; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", 1, ck, "a" + ck);
+ flush();
+
+ execute("DELETE FROM %s WHERE pk = ? AND ck >= ? AND ck <= ?", 1, 10, 15);
+ execute("DELETE FROM %s WHERE pk = ? AND ck > ? AND ck < ?", 1, 20, 25);
+ execute("DELETE FROM %s WHERE pk = ? AND ck >= ? AND ck < ?", 1, 30, 35);
+ execute("DELETE FROM %s WHERE pk = ? AND ck > ? AND ck <= ?", 1, 40, 45);
+ // Open-ended on each side, so a marker sits at the very edge of the covered range.
+ execute("DELETE FROM %s WHERE pk = ? AND ck < ?", 1, 3);
+ execute("DELETE FROM %s WHERE pk = ? AND ck > ?", 1, 76);
+ flush();
+
+ assertCursorMatchesIterator(cfs);
+ }
+
+ /**
+ * A static row alongside reversed clustering. STATIC_CLUSTERING sorts ahead of every row and has
+ * its own kind, so it is the case most likely to invert a min.
+ */
+ @Test
+ public void reversedClusteringWithStaticRow() throws Throwable
+ {
+ createTable("CREATE TABLE %s (pk int, ck int, s text static, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH CLUSTERING ORDER BY (ck DESC)");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ execute("INSERT INTO %s (pk, s) VALUES (?, ?)", 1, "static-one");
+ for (int ck = 0; ck < 30; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", 1, ck, "a" + ck);
+ flush();
+
+ execute("INSERT INTO %s (pk, s) VALUES (?, ?)", 1, "static-two");
+ execute("DELETE FROM %s WHERE pk = ? AND ck >= ? AND ck < ?", 1, 5, 12);
+ flush();
+
+ assertCursorMatchesIterator(cfs);
+ }
+
+ /**
+ * Two reversed clustering columns, so the comparison runs past the first component and the
+ * per-component order flags are exercised in combination.
+ */
+ @Test
+ public void twoReversedClusteringColumns() throws Throwable
+ {
+ createTable("CREATE TABLE %s (pk int, c1 int, c2 text, v text, PRIMARY KEY (pk, c1, c2)) " +
+ "WITH CLUSTERING ORDER BY (c1 DESC, c2 ASC)");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ for (int c1 = 0; c1 < 10; c1++)
+ for (int c2 = 0; c2 < 5; c2++)
+ execute("INSERT INTO %s (pk, c1, c2, v) VALUES (?, ?, ?, ?)", 1, c1, "c" + c2, "v" + c1 + c2);
+ flush();
+
+ execute("DELETE FROM %s WHERE pk = ? AND c1 = ? AND c2 >= ?", 1, 4, "c1");
+ execute("DELETE FROM %s WHERE pk = ? AND c1 > ? AND c1 < ?", 1, 6, 9);
+ flush();
+
+ assertCursorMatchesIterator(cfs);
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/TimeWindowCompactionDifferentialTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/TimeWindowCompactionDifferentialTest.java
new file mode 100644
index 000000000000..569339a398bf
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/TimeWindowCompactionDifferentialTest.java
@@ -0,0 +1,132 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+import java.util.List;
+import java.util.Set;
+
+import org.junit.Test;
+
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.Directories;
+import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
+import org.apache.cassandra.db.compaction.writers.DefaultCompactionWriter;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
+import org.apache.cassandra.db.compaction.TimeWindowCompactionTask;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertTrue;
+
+/**
+ * TWCS on the cursor path, through the real {@link TimeWindowCompactionTask}.
+ *
+ * TWCS was the one strategy with no cursor coverage. STCS is covered throughout the corpus, LCS by
+ * {@link LeveledCompactionDifferentialTest} and UCS by {@link UnifiedCompactionDifferentialTest};
+ * nothing drove a {@link TimeWindowCompactionTask} against the cursor writer.
+ *
+ * What TWCS adds over the others is its dependence on cell timestamps rather than on size or level:
+ * it buckets by the sstable's {@code maxTimestamp}, and it sets {@code ignoreOverlaps}, which
+ * changes what the controller will purge. Timestamp metadata is written by the compaction path
+ * itself, so a cursor-side divergence in {@code minTimestamp} or {@code maxTimestamp} would feed
+ * back into the next bucketing decision.
+ */
+public class TimeWindowCompactionDifferentialTest extends DifferentialCompactionTester
+{
+ /** The parameter cannot be named keepOriginals; see {@link TaskFactory}. */
+ private static TaskFactory timeWindow(boolean ignoreOverlaps, boolean retainOriginals)
+ {
+ return (cfs, txn, gcBefore) -> new TimeWindowCompactionTask(cfs, txn, gcBefore, ignoreOverlaps)
+ {
+ @Override
+ public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs,
+ Directories directories,
+ ILifecycleTransaction transaction,
+ Set nonExpiredSSTables)
+ {
+ return new DefaultCompactionWriter(cfs, directories, transaction, nonExpiredSSTables,
+ retainOriginals, 0);
+ }
+ };
+ }
+
+ private ColumnFamilyStore twoWindows() throws Throwable
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck bigint, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH compression = {'enabled': 'false'} " +
+ "AND compaction = {'class': 'TimeWindowCompactionStrategy', " +
+ "'compaction_window_unit': 'MINUTES', 'compaction_window_size': '1'}");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ // Two flushes with explicitly separated timestamps, so the inputs land in different windows
+ // and the merged output's timestamp range spans both.
+ String padding = "x".repeat(200);
+ for (long pk = 0; pk < 300; pk++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?) USING TIMESTAMP 1000000",
+ pk, 0L, padding + "-old");
+ flush();
+ for (long pk = 150; pk < 450; pk++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?) USING TIMESTAMP 90000000",
+ pk, 0L, padding + "-new");
+ flush();
+
+ assertTrue("the fixture needs inputs", cfs.getLiveSSTables().size() >= 2);
+ return cfs;
+ }
+
+ /** Both paths must produce byte-identical output through the TWCS task. */
+ @Test
+ public void timeWindowTaskMatchesIterator() throws Throwable
+ {
+ ColumnFamilyStore cfs = twoWindows();
+ assertCursorMatchesIterator(cfs, cfs.getLiveSSTables(), timeWindow(false, true));
+ }
+
+ /**
+ * ignoreOverlaps changes what the controller may purge, so it is a distinct scenario rather than
+ * a flag on the previous one.
+ */
+ @Test
+ public void timeWindowTaskIgnoringOverlapsMatchesIterator() throws Throwable
+ {
+ ColumnFamilyStore cfs = twoWindows();
+ assertCursorMatchesIterator(cfs, cfs.getLiveSSTables(), timeWindow(true, true));
+ }
+
+ /**
+ * The committed output's timestamp range is what TWCS buckets on next time, so it is asserted
+ * absolutely rather than only compared between the paths.
+ */
+ @Test
+ public void committedOutputCarriesTheSpanningTimestampRange() throws Throwable
+ {
+ ColumnFamilyStore cfs = twoWindows();
+ commitThroughFactory(cfs, true, timeWindow(false, false));
+
+ List outputs = List.copyOf(cfs.getLiveSSTables());
+ assertEquals("expected one compaction output", 1, outputs.size());
+ SSTableReader output = outputs.get(0);
+
+ assertEquals("the output's minTimestamp must be the oldest cell it carries",
+ 1000000L, output.getSSTableMetadata().minTimestamp);
+ assertEquals("the output's maxTimestamp must be the newest cell it carries",
+ 90000000L, output.getSSTableMetadata().maxTimestamp);
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/compaction/differential/UnifiedCompactionDifferentialTest.java b/test/unit/org/apache/cassandra/db/compaction/differential/UnifiedCompactionDifferentialTest.java
new file mode 100644
index 000000000000..cf84a42a572d
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/compaction/differential/UnifiedCompactionDifferentialTest.java
@@ -0,0 +1,178 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.compaction.differential;
+
+import java.util.List;
+import java.util.Set;
+
+import org.junit.Test;
+
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.DecoratedKey;
+import org.apache.cassandra.db.Directories;
+import org.apache.cassandra.db.compaction.AbstractCompactionStrategy;
+import org.apache.cassandra.db.compaction.ShardManager;
+import org.apache.cassandra.db.compaction.ShardManagerNoDisks;
+import org.apache.cassandra.db.compaction.ShardTracker;
+import org.apache.cassandra.db.compaction.UnifiedCompactionStrategy;
+import org.apache.cassandra.db.compaction.unified.ShardedCompactionWriter;
+import org.apache.cassandra.db.compaction.unified.UnifiedCompactionTask;
+import org.apache.cassandra.db.compaction.writers.CompactionAwareWriter;
+import org.apache.cassandra.db.lifecycle.ILifecycleTransaction;
+import org.apache.cassandra.dht.Token;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+
+import static org.apache.cassandra.db.ColumnFamilyStore.RING_VERSION_IRRELEVANT;
+import static org.junit.Assert.assertTrue;
+
+/**
+ * UCS on the cursor path, through the real {@link UnifiedCompactionTask} and
+ * {@link ShardedCompactionWriter}. {@code ShardedCompactionWriterTest} drives the writer through
+ * {@code CompactionIterator} and {@code writer.append}, which is the iterator path, so nothing
+ * covered UCS against the cursor writer before this.
+ *
+ * The differential harness asserts the cursor pipeline really ran, so a silent fallback to the
+ * iterator path fails here rather than passing as a comparison of one path against itself.
+ */
+public class UnifiedCompactionDifferentialTest extends DifferentialCompactionTester
+{
+ /**
+ * The real task, with the writer's keepOriginals forced on: the harness compacts the same
+ * inputs twice and needs them to survive, and {@link UnifiedCompactionTask} has no
+ * keepOriginals constructor. The parameter cannot be named keepOriginals; see {@link TaskFactory}.
+ */
+ private static TaskFactory sharded(ColumnFamilyStore cfs, int numShards, boolean retainOriginals)
+ {
+ UnifiedCompactionStrategy strategy = unifiedStrategy(cfs);
+ ShardManager shardManager = new ShardManagerNoDisks(ColumnFamilyStore.fullWeightedRange(RING_VERSION_IRRELEVANT,
+ cfs.getPartitioner()));
+ return (c, txn, gcBefore) -> new UnifiedCompactionTask(c, strategy, txn, gcBefore, shardManager)
+ {
+ @Override
+ public CompactionAwareWriter getCompactionAwareWriter(ColumnFamilyStore cfs,
+ Directories directories,
+ ILifecycleTransaction transaction,
+ Set nonExpiredSSTables)
+ {
+ // A fresh tracker per writer: it is stateful, and the harness runs the same
+ // factory once per path.
+ return new ShardedCompactionWriter(cfs, directories, transaction, nonExpiredSSTables,
+ retainOriginals, true /* earlyOpenAllowed */,
+ shardManager.boundaries(numShards));
+ }
+ };
+ }
+
+ private static UnifiedCompactionStrategy unifiedStrategy(ColumnFamilyStore cfs)
+ {
+ for (List perRepairState : cfs.getCompactionStrategyManager().getStrategies())
+ for (AbstractCompactionStrategy strategy : perRepairState)
+ if (strategy instanceof UnifiedCompactionStrategy)
+ return (UnifiedCompactionStrategy) strategy;
+ throw new AssertionError("the table is not on UnifiedCompactionStrategy");
+ }
+
+ private ColumnFamilyStore ucsTable() throws Throwable
+ {
+ createTable("CREATE TABLE %s (pk bigint, ck bigint, v text, PRIMARY KEY (pk, ck)) " +
+ "WITH compaction = {'class': 'UnifiedCompactionStrategy'} " +
+ "AND compression = {'enabled': 'false'}");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+ return cfs;
+ }
+
+ @Test
+ public void shardedWriterMatchesIterator() throws Throwable
+ {
+ ColumnFamilyStore cfs = ucsTable();
+
+ String padding = "x".repeat(120);
+ for (int round = 0; round < 2; round++)
+ {
+ for (long pk = 0; pk < 200; pk++)
+ for (long ck = 0; ck < 4; ck++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", pk, ck, padding + round + "-" + ck);
+ flush();
+ }
+
+ CapturedOutput out = assertCursorMatchesIterator(cfs, cfs.getLiveSSTables(), sharded(cfs, 4, true));
+ assertTrue("sharding must produce several outputs to test anything, got " + out.sstables.size(),
+ out.sstables.size() >= 2);
+ }
+
+ /**
+ * Every output must sit inside one shard: the writer switches on a shard boundary, so a key
+ * on the far side of a boundary landing in the same output means the cursor path missed a
+ * switch. The differential comparison alone would not catch that, because both paths would
+ * have to miss it together to still match.
+ */
+ @Test
+ public void everyOutputStaysInsideOneShardOnCursorPath() throws Throwable
+ {
+ assertEveryOutputStaysInsideOneShard(true);
+ }
+
+ /** The same expectation on the iterator path, so a failure above is read as a cursor defect. */
+ @Test
+ public void everyOutputStaysInsideOneShardOnIteratorPath() throws Throwable
+ {
+ assertEveryOutputStaysInsideOneShard(false);
+ }
+
+ private void assertEveryOutputStaysInsideOneShard(boolean cursor) throws Throwable
+ {
+ ColumnFamilyStore cfs = ucsTable();
+
+ String padding = "y".repeat(200);
+ for (long pk = 0; pk < 120; pk++)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", pk, 0L, padding);
+ flush();
+ for (long pk = 0; pk < 120; pk += 2)
+ execute("INSERT INTO %s (pk, ck, v) VALUES (?, ?, ?)", pk, 1L, padding);
+ flush();
+
+ int numShards = 8;
+ ShardManager shardManager = new ShardManagerNoDisks(ColumnFamilyStore.fullWeightedRange(RING_VERSION_IRRELEVANT,
+ cfs.getPartitioner()));
+ // Commit, so the live set really is the sharded output. The harness restores the originals
+ // after a differential run, so it cannot be used here.
+ commitThroughFactory(cfs, cursor, sharded(cfs, numShards, false));
+
+ Set outputs = cfs.getLiveSSTables();
+ assertTrue("sharding must produce several outputs to test anything, got " + outputs.size(),
+ outputs.size() >= 2);
+ for (SSTableReader sstable : outputs)
+ assertInsideOneShard(shardManager.boundaries(numShards), sstable);
+ }
+
+ /**
+ * Walks a fresh tracker to the sstable's first key, then asserts its last key has not crossed
+ * that shard's end. ShardManager offers no listing of its boundaries, only advancement.
+ */
+ private static void assertInsideOneShard(ShardTracker tracker, SSTableReader sstable)
+ {
+ DecoratedKey first = sstable.getFirst();
+ DecoratedKey last = sstable.getLast();
+ tracker.advanceTo(first.getToken());
+ Token shardEnd = tracker.shardEnd();
+ assertTrue(sstable + " spans shard boundary " + shardEnd + " (" + first + " to " + last + ')',
+ shardEnd == null || last.getToken().compareTo(shardEnd) <= 0);
+ }
+}
diff --git a/test/unit/org/apache/cassandra/db/guardrails/CollectionSizeGuardrailCompactionTest.java b/test/unit/org/apache/cassandra/db/guardrails/CollectionSizeGuardrailCompactionTest.java
index 43e7286e5938..678e3a766d76 100644
--- a/test/unit/org/apache/cassandra/db/guardrails/CollectionSizeGuardrailCompactionTest.java
+++ b/test/unit/org/apache/cassandra/db/guardrails/CollectionSizeGuardrailCompactionTest.java
@@ -126,7 +126,7 @@ private List warningsFromOneCompaction(boolean cursor) throws Exception
@Test
public void bothPipelinesReportTheSameOversizedCollection() throws Exception
{
- assumeBigFormatSelected();
+ assumeCursorSupportedFormatSelected();
List iterator = warningsFromOneCompaction(false);
List cursor = warningsFromOneCompaction(true);
diff --git a/test/unit/org/apache/cassandra/db/guardrails/PartitionTombstonesGuardrailCompactionTest.java b/test/unit/org/apache/cassandra/db/guardrails/PartitionTombstonesGuardrailCompactionTest.java
new file mode 100644
index 000000000000..feecb42b9a6d
--- /dev/null
+++ b/test/unit/org/apache/cassandra/db/guardrails/PartitionTombstonesGuardrailCompactionTest.java
@@ -0,0 +1,154 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.db.guardrails;
+
+import java.io.Serializable;
+import java.util.ArrayList;
+import java.util.List;
+import java.util.Map;
+import java.util.Set;
+import java.util.concurrent.CopyOnWriteArrayList;
+import java.util.function.Consumer;
+
+import org.junit.After;
+import org.junit.Before;
+import org.junit.Test;
+
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.db.ColumnFamilyStore;
+import org.apache.cassandra.db.compaction.differential.DifferentialCompactionTester;
+import org.apache.cassandra.db.guardrails.GuardrailEvent.GuardrailEventType;
+import org.apache.cassandra.diag.DiagnosticEventService;
+import org.apache.cassandra.io.sstable.format.SSTableReader;
+
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertFalse;
+
+/**
+ * The {@code partition_tombstones} guardrail must fire at the same count on both compaction pipelines.
+ *
+ * {@code SortedTableWriter} counts the partition-level deletion when the partition starts and checks
+ * the threshold when it ends, so the deletion is inside the count. The cursor writer does both in
+ * {@code addPartitionMetadata}, and counting the deletion after the check would leave the partition
+ * one tombstone short. This fixture sits exactly on that boundary: a partition deletion plus enough
+ * row tombstones that the total crosses the threshold only if the partition deletion counts.
+ */
+public class PartitionTombstonesGuardrailCompactionTest extends DifferentialCompactionTester
+{
+ /** Row tombstones in the fixture. With the partition deletion the total is one more. */
+ private static final int ROW_TOMBSTONES = 5;
+ /** The total is ROW_TOMBSTONES + 1 and the guardrail fires above the threshold, not at it. */
+ private static final long WARN_THRESHOLD = ROW_TOMBSTONES;
+ private static final long FAIL_THRESHOLD = 1000;
+
+ private final WarningCollector collector = new WarningCollector();
+ private long originalWarn;
+ private long originalFail;
+ private boolean originalDiagnostics;
+
+ @Before
+ public void armGuardrails()
+ {
+ originalWarn = Guardrails.instance.getPartitionTombstonesWarnThreshold();
+ originalFail = Guardrails.instance.getPartitionTombstonesFailThreshold();
+ originalDiagnostics = DatabaseDescriptor.diagnosticEventsEnabled();
+
+ Guardrails.instance.setPartitionTombstonesThreshold(WARN_THRESHOLD, FAIL_THRESHOLD);
+ DatabaseDescriptor.setDiagnosticEventsEnabled(true);
+ DiagnosticEventService.instance().subscribe(GuardrailEvent.class, collector);
+ }
+
+ @After
+ public void disarmGuardrails()
+ {
+ DiagnosticEventService.instance().unsubscribe(collector);
+ DatabaseDescriptor.setDiagnosticEventsEnabled(originalDiagnostics);
+ Guardrails.instance.setPartitionTombstonesThreshold(originalWarn, originalFail);
+ }
+
+ /**
+ * One partition holding a partition deletion and, above it in timestamp, {@link #ROW_TOMBSTONES}
+ * row tombstones. The row deletions are newer, so the compaction keeps all of them, and the two
+ * kinds arrive in separate sstables so neither flush can warn on its own.
+ */
+ private ColumnFamilyStore partitionDeletionOverRowTombstones()
+ {
+ createTable("CREATE TABLE %s (k int, c int, v int, PRIMARY KEY (k, c)) " +
+ "WITH gc_grace_seconds = 864000");
+ ColumnFamilyStore cfs = getCurrentColumnFamilyStore();
+ cfs.disableAutoCompaction();
+
+ execute("DELETE FROM %s USING TIMESTAMP 1 WHERE k = 1");
+ flush();
+ for (int c = 0; c < ROW_TOMBSTONES; c++)
+ execute("DELETE FROM %s USING TIMESTAMP ? WHERE k = 1 AND c = ?", 10L + c, c);
+ flush();
+
+ assertEquals("the fixture needs two sstables to merge", 2, cfs.getLiveSSTables().size());
+ collector.drain();
+ return cfs;
+ }
+
+ /** Compacts a fresh fixture down one pipeline and returns the warnings that compaction emitted. */
+ private List warningsFromOneCompaction(boolean cursor) throws Exception
+ {
+ ColumnFamilyStore cfs = partitionDeletionOverRowTombstones();
+ Set inputs = cfs.getLiveSSTables();
+ // gcBefore 0 keeps every tombstone: a purged one would never reach the guardrail.
+ commitCompaction(cfs, inputs, cursor, 0);
+ return collector.drain();
+ }
+
+ @Test
+ public void bothPipelinesWarnAtTheSameTombstoneCount() throws Exception
+ {
+ assumeCursorSupportedFormatSelected();
+
+ List iterator = warningsFromOneCompaction(false);
+ List cursor = warningsFromOneCompaction(true);
+
+ assertFalse("the iterator path must warn, or this says nothing about the cursor path",
+ iterator.isEmpty());
+ assertEquals("the cursor path must count the same tombstones as the iterator path",
+ iterator, cursor);
+ }
+
+ /** Records the redacted text of each partition_tombstones warning, in arrival order. */
+ private static final class WarningCollector implements Consumer
+ {
+ private final List warnings = new CopyOnWriteArrayList<>();
+
+ @Override
+ public void accept(GuardrailEvent event)
+ {
+ if (event.getType() != GuardrailEventType.WARNED)
+ return;
+ Map map = event.toMap();
+ if (Guardrails.partitionTombstones.name.equals(map.get("name")))
+ warnings.add(String.valueOf(map.get("message")));
+ }
+
+ List drain()
+ {
+ List drained = new ArrayList<>(warnings);
+ warnings.clear();
+ return drained;
+ }
+ }
+}
diff --git a/test/unit/org/apache/cassandra/dht/ReusableDecoratedKeyTest.java b/test/unit/org/apache/cassandra/dht/ReusableDecoratedKeyTest.java
new file mode 100644
index 000000000000..8126ed50f6fc
--- /dev/null
+++ b/test/unit/org/apache/cassandra/dht/ReusableDecoratedKeyTest.java
@@ -0,0 +1,83 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.dht;
+
+import java.nio.ByteBuffer;
+
+import org.junit.BeforeClass;
+import org.junit.Test;
+
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.db.DecoratedKey;
+import org.apache.cassandra.db.marshal.Int32Type;
+import org.apache.cassandra.utils.ByteBufferUtil;
+
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertNotSame;
+
+/**
+ * A reusable key's bytes and token move with every {@code copyKey}, so {@code retainable()} must
+ * hand back a key that stays put after the next copy.
+ */
+public class ReusableDecoratedKeyTest
+{
+ @BeforeClass
+ public static void setUp()
+ {
+ DatabaseDescriptor.daemonInitialization();
+ }
+
+ @Test
+ public void retainableSurvivesTheNextKeyOnMurmur3()
+ {
+ assertRetainableSurvivesTheNextKey(Murmur3Partitioner.instance);
+ }
+
+ @Test
+ public void retainableSurvivesTheNextKeyOnLocalPartitioner()
+ {
+ assertRetainableSurvivesTheNextKey(new LocalPartitioner(Int32Type.instance));
+ }
+
+ private static void assertRetainableSurvivesTheNextKey(IPartitioner partitioner)
+ {
+ byte[] first = ByteBufferUtil.bytes(1).array();
+ byte[] second = ByteBufferUtil.bytes(2).array();
+ DecoratedKey expectedFirst = partitioner.decorateKey(ByteBuffer.wrap(first));
+ DecoratedKey expectedSecond = partitioner.decorateKey(ByteBuffer.wrap(second));
+
+ ReusableDecoratedKey reusable = partitioner.createReusableKey(0);
+ reusable.copyKey(first, first.length);
+ assertEquals(expectedFirst, reusable);
+ // By order, not equals: the reusable key's token is a subclass, and LongToken.equals is
+ // class-strict. The detached copy below is checked with equals.
+ assertEquals(0, expectedFirst.getToken().compareTo(reusable.getToken()));
+
+ DecoratedKey retained = reusable.retainable();
+ assertNotSame(reusable, retained);
+ reusable.copyKey(second, second.length);
+
+ assertEquals(expectedFirst, retained);
+ assertEquals(expectedFirst.getToken(), retained.getToken());
+ assertEquals(expectedFirst.getKey(), retained.getKey());
+ assertEquals(expectedSecond, reusable);
+ // By order for the same reason as above
+ assertEquals(0, expectedSecond.getToken().compareTo(reusable.getToken()));
+ }
+}
diff --git a/test/unit/org/apache/cassandra/io/sstable/ClusteringDescriptorPrefixViewTest.java b/test/unit/org/apache/cassandra/io/sstable/ClusteringDescriptorPrefixViewTest.java
new file mode 100644
index 000000000000..672bf46a118b
--- /dev/null
+++ b/test/unit/org/apache/cassandra/io/sstable/ClusteringDescriptorPrefixViewTest.java
@@ -0,0 +1,674 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.cassandra.io.sstable;
+
+import java.io.ByteArrayOutputStream;
+import java.io.IOException;
+import java.nio.ByteBuffer;
+import java.util.ArrayList;
+import java.util.Arrays;
+import java.util.List;
+import java.util.UUID;
+
+import org.junit.BeforeClass;
+import org.junit.Test;
+import org.quicktheories.core.Gen;
+import org.quicktheories.core.RandomnessSource;
+import org.quicktheories.generators.SourceDSL;
+
+import org.apache.cassandra.config.CassandraRelevantProperties;
+import org.apache.cassandra.config.DatabaseDescriptor;
+import org.apache.cassandra.db.Clustering;
+import org.apache.cassandra.db.ClusteringComparator;
+import org.apache.cassandra.db.ClusteringPrefix;
+import org.apache.cassandra.db.marshal.AbstractType;
+import org.apache.cassandra.db.marshal.ByteBufferAccessor;
+import org.apache.cassandra.db.marshal.BytesType;
+import org.apache.cassandra.db.marshal.Int32Type;
+import org.apache.cassandra.db.marshal.LongType;
+import org.apache.cassandra.db.marshal.ReversedType;
+import org.apache.cassandra.db.marshal.UTF8Type;
+import org.apache.cassandra.db.marshal.UUIDType;
+import org.apache.cassandra.io.tries.Walker;
+import org.apache.cassandra.io.util.DataOutputBuffer;
+import org.apache.cassandra.utils.AbstractTypeGenerators;
+import org.apache.cassandra.utils.AbstractTypeGenerators.TypeKind;
+import org.apache.cassandra.utils.AbstractTypeGenerators.ValueDomain;
+import org.apache.cassandra.utils.ByteBufferUtil;
+import org.apache.cassandra.utils.bytecomparable.ByteComparable;
+import org.apache.cassandra.utils.bytecomparable.ByteSource;
+
+import static org.junit.Assert.assertArrayEquals;
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertNotNull;
+import static org.junit.Assert.assertNotSame;
+import static org.junit.Assert.assertNull;
+import static org.junit.Assert.assertSame;
+import static org.junit.Assert.assertTrue;
+import static org.junit.Assert.fail;
+import static org.quicktheories.QuickTheory.qt;
+
+/**
+ * Pins {@link ClusteringDescriptorPrefixView} to the wire format it re-implements.
+ *
+ * The view is the only cursor-specific input to the BTI row trie: {@code BtiCursorIndexWriter}
+ * snapshots one per index block and hands it to {@code RowIndexWriter}, which turns it into a
+ * {@link ByteComparable} through {@link ClusteringComparator#asByteComparable}. Its
+ * {@code parse} method is a hand-written decoder for the same bytes
+ * {@link ClusteringPrefix.Serializer#deserializeValuesWithoutSize} reads, so a divergence there
+ * writes a trie that indexes the wrong rows without corrupting a single byte of the data file.
+ *
+ * Oracle
+ *
+ * Reference implementation, already in the tree:
+ * {@link ClusteringDescriptor#toClusteringPrefix(List)} decodes the descriptor's bytes through
+ * {@code Clustering.serializer} / {@code ClusteringPrefix.serializer}. The bytes themselves are
+ * produced by the production encoder, {@code Clustering.serializer.serialize}, which is what
+ * {@code SSTableCursorReader.readUnfilteredClustering} copies verbatim off disk into the
+ * descriptor. So encoder and reference decoder are both production code; only the view is new.
+ *
+ * Three things are compared per example: kind and size, every component (null, empty and valued
+ * distinguished), and the full byte-comparable encoding compared byte for byte rather than through
+ * {@code ByteComparable.compare}, so a prefix relationship fails instead of passing.
+ *
+ * What this test cannot see
+ *
+ *
+ * - Everything the two sides share. The reference and the view both call the same
+ * {@code AbstractType.asComparableBytes}, the same {@code isValueLengthFixed}, and the same
+ * {@code ClusteringComparator}; a defect in any of those is invisible here. Only the decode
+ * of the header bits and the component walk is genuinely differential.
+ * - Whether the bytes in a real descriptor match what this test writes. The test encodes with
+ * {@code Clustering.serializer.serialize}; production fills the descriptor through
+ * {@code SSTableCursorReader.readUnfilteredClustering}. Those two agreeing is asserted
+ * elsewhere (the differential compaction suite), not here.
+ * - Concurrency and reuse across threads. The view is single-threaded by construction.
+ * - Whether a live (non-snapshot) view left pointing at a resized descriptor still reads
+ * sensibly. It does not, by design; {@code retainable()} exists for that, and that is what is
+ * asserted below.
+ * - The trie itself. A view that byte-compares identically to the reference can still be
+ * mis-used by the caller; that is {@code BtiCursorIndexWriter}'s coverage, not this class's.
+ *
+ */
+public class ClusteringDescriptorPrefixViewTest
+{
+ /** The version the row trie is built and read at; see {@code RowIndexWriter}. */
+ private static final ByteComparable.Version BYTE_COMPARABLE_VERSION = Walker.BYTE_COMPARABLE_VERSION;
+
+ /**
+ * Passed to the serializer and to {@link ClusteringDescriptor#toClusteringPrefix(List)}, which
+ * hardcodes 0. Neither the header nor the component encoding depends on it.
+ */
+ private static final int SERIALIZATION_VERSION = 0;
+
+ /**
+ * Past two header blocks, so {@code i == 32} and {@code i == 33} are reached by the generator
+ * and not only by {@link #headerBlockBoundaries()}.
+ */
+ private static final int MAX_CLUSTERING_COLUMNS = 40;
+
+ /** Every kind that can reach the row trie. STATIC_CLUSTERING carries no bytes and no components. */
+ private static final List KINDS =
+ Arrays.asList(ClusteringPrefix.Kind.CLUSTERING,
+ ClusteringPrefix.Kind.INCL_START_BOUND,
+ ClusteringPrefix.Kind.EXCL_START_BOUND,
+ ClusteringPrefix.Kind.INCL_END_BOUND,
+ ClusteringPrefix.Kind.EXCL_END_BOUND,
+ ClusteringPrefix.Kind.INCL_END_EXCL_START_BOUNDARY,
+ ClusteringPrefix.Kind.EXCL_END_INCL_START_BOUNDARY);
+
+ /**
+ * Weighted towards NORMAL. An unweighted pick over 40 components leaves too few present values
+ * to exercise the fixed-length and vint length walks; {@link #assertCorpusReachedEveryBranch}
+ * is the check that this weighting actually paid off.
+ */
+ private static final List VALUE_DOMAINS =
+ Arrays.asList(ValueDomain.NORMAL, ValueDomain.NORMAL, ValueDomain.NORMAL,
+ ValueDomain.NORMAL, ValueDomain.NORMAL, ValueDomain.NORMAL,
+ ValueDomain.NULL, ValueDomain.EMPTY_BYTES);
+
+ private static final int SEEN_FIXED_PRESENT = 0;
+ private static final int SEEN_VARIABLE_PRESENT = 1;
+ private static final int SEEN_NULL = 2;
+ private static final int SEEN_EMPTY = 3;
+ private static final int SEEN_SECOND_HEADER_BLOCK = 4;
+ private static final int SEEN_PRESENT_NOT_LAST = 5;
+ private static final String[] BRANCH_LABELS = { "a present fixed-length component (parse: valueLengthIfFixed)",
+ "a present variable-length component (parse: the length vint)",
+ "a null component (parse: the 2i+1 bit)",
+ "an empty component (parse: the 2i bit)",
+ "a prefix longer than 32 components (parse: the second header vint)",
+ "a present component followed by another (parse: pos += len)" };
+
+ @BeforeClass
+ public static void beforeClass()
+ {
+ DatabaseDescriptor.daemonInitialization();
+ }
+
+ /**
+ * The property. Every generated prefix must decode through the view exactly as it decodes
+ * through {@link ClusteringDescriptor#toClusteringPrefix(List)}, and must produce the same
+ * byte-comparable bytes.
+ *
+ * Shrinking is deliberately left on: this is a pure function, so quicktheories can shrink a
+ * failure to a minimal type list and value set. A failure prints "Seed was N" and the shrunk
+ * example in full, including its serialized bytes; replay it with {@code -DQT_SEED=N}.
+ */
+ @Test
+ public void parseMatchesTheSerializer()
+ {
+ int[] counters = new int[BRANCH_LABELS.length];
+ qt().withExamples(CassandraRelevantProperties.TEST_CLUSTERING_PREFIX_VIEW_EXAMPLES.getInt())
+ .forAll(exampleGen())
+ .checkAssert(example -> {
+ TestDescriptor descriptor = example.load(new TestDescriptor(example.types));
+ ClusteringComparator comparator = new ClusteringComparator(example.types);
+ ClusteringPrefix reference = referenceOf(descriptor, example.types);
+
+ assertSamePrefix("reset", reference,
+ new ClusteringDescriptorPrefixView(example.types).reset(descriptor), comparator);
+ // The snapshot copies the bytes and re-parses them, so it is a second, independent
+ // trip through parse over the same input.
+ assertSamePrefix("snapshotOf", reference,
+ ClusteringDescriptorPrefixView.snapshotOf(descriptor, example.types), comparator);
+
+ recordBranches(counters, reference, example.types);
+ });
+ assertCorpusReachedEveryBranch(counters);
+ }
+
+ /**
+ * Fixed sizes either side of every header-block boundary, with the null and empty positions
+ * asserted absolutely rather than against the reference. 33 is the first size that needs a
+ * second header vint; 65 the first that needs a third.
+ */
+ @Test
+ public void headerBlockBoundaries()
+ {
+ for (int count : new int[]{ 1, 2, 31, 32, 33, 40, 64, 65 })
+ {
+ AbstractType>[] types = deterministicTypes(count);
+ ByteBuffer[] values = new ByteBuffer[count];
+ for (int i = 0; i < count; i++)
+ values[i] = deterministicValue(types[i], i);
+
+ TestDescriptor descriptor = new TestDescriptor(types);
+ descriptor.load(ClusteringPrefix.Kind.CLUSTERING, count, serialize(types, values));
+
+ ClusteringDescriptorPrefixView view = new ClusteringDescriptorPrefixView(types).reset(descriptor);
+ assertSamePrefix("size " + count, referenceOf(descriptor, types), view, new ClusteringComparator(types));
+
+ // Absolute: the answer is stated here, not read off either implementation.
+ for (int i = 0; i < count; i++)
+ {
+ ByteBuffer component = view.get(i);
+ if (i % 5 == 3)
+ assertNull("size " + count + " component " + i + " must be null", component);
+ else if (i % 5 == 4)
+ assertEquals("size " + count + " component " + i + " must be empty",
+ 0, component.remaining());
+ else
+ assertTrue("size " + count + " component " + i + " must carry bytes",
+ component.remaining() > 0);
+ }
+ }
+ }
+
+ /**
+ * A bound with no components. {@code resetMaxStart} and {@code resetMinEnd} are the production
+ * calls that produce it, and the view must carry only the kind.
+ */
+ @Test
+ public void emptyBoundsCarryOnlyTheirKind()
+ {
+ AbstractType>[] types = { Int32Type.instance, UTF8Type.instance };
+ ClusteringComparator comparator = new ClusteringComparator(types);
+ TestDescriptor descriptor = new TestDescriptor(types);
+
+ descriptor.resetMaxStart();
+ assertSamePrefix("max start", referenceOf(descriptor, types),
+ ClusteringDescriptorPrefixView.snapshotOf(descriptor, types), comparator);
+
+ descriptor.resetMinEnd();
+ assertSamePrefix("min end", referenceOf(descriptor, types),
+ ClusteringDescriptorPrefixView.snapshotOf(descriptor, types), comparator);
+ }
+
+ /**
+ * Covers both sides of the {@code backing != bytes} identity check in
+ * {@link ClusteringDescriptorPrefixView#reset}: a reload that keeps the descriptor's array must
+ * re-parse without re-wrapping, and a different descriptor must re-wrap. The array identity is
+ * asserted rather than assumed, so the branch claim is checkable.
+ */
+ @Test
+ public void resetRewrapsOnlyOnANewBackingArray()
+ {
+ AbstractType>[] types = { Int32Type.instance, UTF8Type.instance, LongType.instance };
+ ClusteringComparator comparator = new ClusteringComparator(types);
+
+ TestDescriptor first = new TestDescriptor(types);
+ first.load(ClusteringPrefix.Kind.CLUSTERING, 3,
+ serialize(types, values(Int32Type.instance.decompose(1),
+ UTF8Type.instance.decompose("a"),
+ LongType.instance.decompose(2L))));
+ byte[] backing = first.clusteringBytes();
+
+ ClusteringDescriptorPrefixView view = new ClusteringDescriptorPrefixView(types);
+ view.reset(first);
+ assertSamePrefix("first parse", referenceOf(first, types), view, comparator);
+
+ // Same array, different content and a different length: the identity check must take the
+ // "already wrapped" path and parse must still honour the new limit.
+ first.load(ClusteringPrefix.Kind.INCL_END_BOUND, 2,
+ serialize(types, values(null, UTF8Type.instance.decompose("bbbbbbbb"))));
+ assertSame("the reload must not have resized the descriptor, or the backing == bytes branch " +
+ "is not the one being covered",
+ backing, first.clusteringBytes());
+ view.reset(first);
+ assertSamePrefix("same backing array", referenceOf(first, types), view, comparator);
+
+ // A different descriptor owns a different array, so the view must re-wrap.
+ TestDescriptor second = new TestDescriptor(types);
+ second.load(ClusteringPrefix.Kind.CLUSTERING, 3,
+ serialize(types, values(ByteBufferUtil.EMPTY_BYTE_BUFFER,
+ UTF8Type.instance.decompose("c"),
+ LongType.instance.decompose(-9L))));
+ assertNotSame(first.clusteringBytes(), second.clusteringBytes());
+ view.reset(second);
+ assertSamePrefix("new backing array", referenceOf(second, types), view, comparator);
+ }
+
+ /**
+ * A snapshot, and a {@code retainable()} taken from a live view, must both survive the source
+ * descriptor being overwritten.
+ *
+ * Two overwrites, because only the first one catches a snapshot that aliased the descriptor's
+ * array instead of copying it: the second is long enough to force a resize, which leaves an
+ * aliasing snapshot pointing at the old array and still reading the right answer by accident.
+ */
+ @Test
+ public void retainedViewsSurviveTheDescriptorBeingOverwritten()
+ {
+ AbstractType>[] types = { Int32Type.instance, UTF8Type.instance };
+ ClusteringComparator comparator = new ClusteringComparator(types);
+
+ TestDescriptor descriptor = new TestDescriptor(types);
+ descriptor.load(ClusteringPrefix.Kind.CLUSTERING, 2,
+ serialize(types, values(Int32Type.instance.decompose(7),
+ UTF8Type.instance.decompose("before"))));
+ ClusteringPrefix expected = referenceOf(descriptor, types);
+ byte[] backing = descriptor.clusteringBytes();
+
+ ClusteringDescriptorPrefixView snapshot = ClusteringDescriptorPrefixView.snapshotOf(descriptor, types);
+ ClusteringDescriptorPrefixView live = new ClusteringDescriptorPrefixView(types).reset(descriptor);
+ ClusteringPrefix> retained = live.retainable();
+ assertNotSame("retainable() on a live view must copy", live, retained);
+
+ // Same serialized length, so the descriptor is rewritten in place, in the array the
+ // snapshot would be aliasing if it had not copied.
+ descriptor.load(ClusteringPrefix.Kind.CLUSTERING, 2,
+ serialize(types, values(Int32Type.instance.decompose(-7),
+ UTF8Type.instance.decompose("AFTER!"))));
+ assertSame("the overwrite must have stayed in the same array, or an aliasing snapshot " +
+ "would pass this test by accident",
+ backing, descriptor.clusteringBytes());
+ assertSamePrefix("snapshot after an in-place overwrite", expected, snapshot, comparator);
+ assertSamePrefix("retainable after an in-place overwrite", expected,
+ (ClusteringDescriptorPrefixView) retained, comparator);
+
+ // And one that replaces the array outright.
+ descriptor.load(ClusteringPrefix.Kind.CLUSTERING, 2,
+ serialize(types, values(Int32Type.instance.decompose(11),
+ UTF8Type.instance.decompose(repeat('x', 300)))));
+ assertNotSame(backing, descriptor.clusteringBytes());
+ assertSamePrefix("snapshot after a resizing overwrite", expected, snapshot, comparator);
+ assertSamePrefix("retainable after a resizing overwrite", expected,
+ (ClusteringDescriptorPrefixView) retained, comparator);
+ }
+
+ /** A view that already owns its bytes returns itself and refuses to be re-pointed. */
+ @Test
+ public void anOwnedViewIsItsOwnRetainableAndRejectsReset()
+ {
+ AbstractType>[] types = { Int32Type.instance };
+ TestDescriptor descriptor = new TestDescriptor(types);
+ descriptor.load(ClusteringPrefix.Kind.CLUSTERING, 1,
+ serialize(types, values(Int32Type.instance.decompose(3))));
+
+ ClusteringDescriptorPrefixView snapshot = ClusteringDescriptorPrefixView.snapshotOf(descriptor, types);
+ assertSame(snapshot, snapshot.retainable());
+ try
+ {
+ snapshot.reset(descriptor);
+ fail("a snapshot owns its bytes and must refuse reset");
+ }
+ catch (IllegalStateException expected)
+ {
+ // the contract in the javadoc of reset
+ }
+ }
+
+ // ------------------------------------------------------------------------------------------
+ // oracle
+
+ private static void assertSamePrefix(String context,
+ ClusteringPrefix reference,
+ ClusteringPrefix view,
+ ClusteringComparator comparator)
+ {
+ assertEquals(context + ": kind", reference.kind(), view.kind());
+ assertEquals(context + ": size", reference.size(), view.size());
+
+ // get(i) hands back one shared, repositioned window, so each component is consumed before
+ // the next is asked for.
+ for (int i = 0; i < reference.size(); i++)
+ {
+ byte[] expected = reference.get(i);
+ ByteBuffer actual = view.get(i);
+ if (expected == null)
+ {
+ assertNull(context + ": component " + i + " must be null", actual);
+ }
+ else
+ {
+ assertNotNull(context + ": component " + i + " must not be null", actual);
+ assertArrayEquals(context + ": component " + i, expected, ByteBufferUtil.getArray(actual));
+ }
+ }
+
+ // Byte for byte, not ByteComparable.compare: a prefix relationship must fail here.
+ assertArrayEquals(context + ": byte-comparable encoding",
+ drain(comparator.asByteComparable(reference)),
+ drain(comparator.asByteComparable(view)));
+ }
+
+ private static byte[] drain(ByteComparable comparable)
+ {
+ ByteSource source = comparable.asComparableBytes(BYTE_COMPARABLE_VERSION);
+ ByteArrayOutputStream out = new ByteArrayOutputStream();
+ for (int b = source.next(); b != ByteSource.END_OF_STREAM; b = source.next())
+ out.write(b);
+ return out.toByteArray();
+ }
+
+ @SuppressWarnings("unchecked")
+ private static ClusteringPrefix referenceOf(ClusteringDescriptor descriptor, AbstractType>[] types)
+ {
+ return (ClusteringPrefix) descriptor.toClusteringPrefix(Arrays.asList(types));
+ }
+
+ // ------------------------------------------------------------------------------------------
+ // input
+
+ /**
+ * Fills a descriptor the way {@code SSTableCursorReader.readUnfilteredClustering} does, without
+ * a data file. The subclass exists only to reach {@code ResizableByteBuffer.overwrite}; no
+ * production accessor was added for the test.
+ */
+ private static class TestDescriptor extends ClusteringDescriptor
+ {
+ TestDescriptor(AbstractType>[] types)
+ {
+ super(types);
+ }
+
+ void load(ClusteringPrefix.Kind kind, int bound, byte[] serialized)
+ {
+ clusteringKind(kind);
+ clusteringColumnsBound = bound;
+ overwrite(serialized, serialized.length);
+ }
+ }
+
+ private static class Example
+ {
+ final AbstractType>[] types;
+ final ClusteringPrefix.Kind kind;
+ final ByteBuffer[] values;
+ final byte[] serialized;
+
+ Example(AbstractType>[] types, ClusteringPrefix.Kind kind, ByteBuffer[] values)
+ {
+ this.types = types;
+ this.kind = kind;
+ this.values = values;
+ this.serialized = serialize(types, values);
+ }
+
+ TestDescriptor load(TestDescriptor descriptor)
+ {
+ descriptor.load(kind, values.length, serialized);
+ return descriptor;
+ }
+
+ /** Enough to rebuild the example by hand from a failure message. */
+ @Override
+ public String toString()
+ {
+ StringBuilder sb = new StringBuilder(kind.toString()).append(" over ").append(types.length)
+ .append(" types, ").append(values.length)
+ .append(" components:");
+ for (int i = 0; i < types.length; i++)
+ {
+ sb.append("\n [").append(i).append("] ").append(types[i].asCQL3Type())
+ .append(types[i].isValueLengthFixed() ? " fixed(" + types[i].valueLengthIfFixed() + ")" : " variable")
+ .append(" = ");
+ if (i >= values.length)
+ sb.append("");
+ else if (values[i] == null)
+ sb.append("null");
+ else if (!values[i].hasRemaining())
+ sb.append("empty");
+ else
+ sb.append("0x").append(ByteBufferUtil.bytesToHex(values[i]));
+ }
+ return sb.append("\n bytes = 0x").append(ByteBufferUtil.bytesToHex(ByteBuffer.wrap(serialized))).toString();
+ }
+ }
+
+ /**
+ * Primitives supply both halves of the branch {@code parse} actually turns on, fixed-length and
+ * variable-length. Vectors over fixed-length primitives are mixed in for a fixed width wider
+ * than any primitive's; their elements are restricted to fixed-length types because
+ * {@code AbstractTypeGenerators} can otherwise generate a vector holding an empty element,
+ * which {@code VectorType.unpack} rejects on read. Frozen collections and tuples are left out
+ * deliberately: they are all variable-length, so they add no branch here, and the differential
+ * compaction suite covers them end to end.
+ */
+ private static Gen> componentTypeGen()
+ {
+ Gen> primitiveGen = AbstractTypeGenerators.builder()
+ .withoutUnsafeEquality()
+ .withTypeKinds(TypeKind.PRIMITIVE)
+ .withMaxDepth(0)
+ .build();
+ Gen> fixedPrimitiveGen = AbstractTypeGenerators.builder()
+ .withoutUnsafeEquality()
+ .withTypeKinds(TypeKind.PRIMITIVE)
+ .withMaxDepth(0)
+ .withTypeFilter(AbstractType::isValueLengthFixed)
+ .build();
+ Gen> vectorGen = AbstractTypeGenerators.vectorTypeGen(fixedPrimitiveGen,
+ SourceDSL.integers().between(1, 3))
+ .map(vector -> (AbstractType>) vector);
+ Gen vectorChance = SourceDSL.integers().between(0, 4).map(i -> i == 0);
+ return AbstractTypeGenerators.allowReversed(
+ rnd -> vectorChance.generate(rnd) ? vectorGen.generate(rnd) : primitiveGen.generate(rnd));
+ }
+
+ private static Gen exampleGen()
+ {
+ Gen> typeGen = componentTypeGen();
+ Gen countGen = SourceDSL.integers().between(1, MAX_CLUSTERING_COLUMNS);
+ Gen kindGen = SourceDSL.arbitrary().pick(KINDS);
+ Gen domainGen = SourceDSL.arbitrary().pick(VALUE_DOMAINS);
+
+ return rnd -> {
+ int count = countGen.generate(rnd);
+ AbstractType>[] types = new AbstractType>[count];
+ for (int i = 0; i < count; i++)
+ types[i] = typeGen.generate(rnd);
+
+ ClusteringPrefix.Kind kind = kindGen.generate(rnd);
+ // A bound or boundary may stop short of the full clustering key; a Clustering may not.
+ int bound = kind == ClusteringPrefix.Kind.CLUSTERING
+ ? count
+ : SourceDSL.integers().between(1, count).generate(rnd);
+
+ ByteBuffer[] values = new ByteBuffer[bound];
+ for (int i = 0; i < bound; i++)
+ values[i] = value(types[i], domainGen.generate(rnd), rnd);
+ return new Example(types, kind, values);
+ };
+ }
+
+ /**
+ * A null component never reaches its type: the header carries it and
+ * {@link ClusteringComparator#asByteComparable} emits NEXT_COMPONENT_NULL, so NULL is legal for
+ * every type. An empty component is only offered to types that accept empty bytes;
+ * {@code VectorType.asComparableBytes} throws on one, which says nothing about {@code parse}
+ * and would only make the oracle unusable. Enough fixed-length primitives allow empty
+ * (int, bigint, boolean, uuid, timestamp, ...) to keep the empty-on-fixed-length case, which is
+ * the one that matters here: {@code parse} must not advance by valueLengthIfFixed for it.
+ */
+ private static ByteBuffer value(AbstractType> type, ValueDomain domain, RandomnessSource rnd)
+ {
+ if (domain == ValueDomain.NULL)
+ return null;
+ if (domain == ValueDomain.EMPTY_BYTES && type.unwrap().allowsEmpty())
+ return ByteBufferUtil.EMPTY_BYTE_BUFFER;
+ return AbstractTypeGenerators.getTypeSupport(type).bytesGen().generate(rnd);
+ }
+
+ /**
+ * Encodes with the production encoder. The bytes are a function of the values and the types
+ * only, so serialising the present components as a {@code Clustering} over the matching prefix
+ * of the type list produces exactly what a bound of that length holds on disk.
+ */
+ private static byte[] serialize(AbstractType>[] types, ByteBuffer[] values)
+ {
+ Clustering clustering = values.length == 0
+ ? ByteBufferAccessor.instance.factory().clustering()
+ : ByteBufferAccessor.instance.factory().clustering(values);
+ List> present = Arrays.asList(types).subList(0, values.length);
+ try (DataOutputBuffer out = new DataOutputBuffer())
+ {
+ Clustering.serializer.serialize(clustering, out, SERIALIZATION_VERSION, present);
+ return out.toByteArray();
+ }
+ catch (IOException e)
+ {
+ throw new AssertionError("writing to an in-memory buffer must not fail", e);
+ }
+ }
+
+ private static ByteBuffer[] values(ByteBuffer... values)
+ {
+ return values;
+ }
+
+ /** Cycles fixed and variable, plain and reversed, so no size lands on a single shape. */
+ private static AbstractType>[] deterministicTypes(int count)
+ {
+ AbstractType>[] pattern = { Int32Type.instance,
+ UTF8Type.instance,
+ LongType.instance,
+ ReversedType.getInstance(Int32Type.instance),
+ BytesType.instance,
+ UUIDType.instance,
+ ReversedType.getInstance(UTF8Type.instance) };
+ AbstractType>[] types = new AbstractType>[count];
+ for (int i = 0; i < count; i++)
+ types[i] = pattern[i % pattern.length];
+ return types;
+ }
+
+ /** null at {@code i % 5 == 3}, empty at {@code i % 5 == 4}, a distinct value elsewhere. */
+ private static ByteBuffer deterministicValue(AbstractType> type, int i)
+ {
+ if (i % 5 == 3)
+ return null;
+ if (i % 5 == 4)
+ return ByteBufferUtil.EMPTY_BYTE_BUFFER;
+
+ AbstractType> base = type.unwrap();
+ if (base == Int32Type.instance)
+ return Int32Type.instance.decompose(i);
+ if (base == LongType.instance)
+ return LongType.instance.decompose(i * 1_000_003L);
+ if (base == UUIDType.instance)
+ return UUIDType.instance.decompose(new UUID(i, ~i));
+ if (base == UTF8Type.instance)
+ return UTF8Type.instance.decompose("component-" + i);
+ if (base == BytesType.instance)
+ return ByteBuffer.wrap(new byte[]{ (byte) i, (byte) (i >>> 8), 0x7f });
+ throw new AssertionError("no deterministic value defined for " + type);
+ }
+
+ private static String repeat(char c, int length)
+ {
+ char[] chars = new char[length];
+ Arrays.fill(chars, c);
+ return new String(chars);
+ }
+
+ // ------------------------------------------------------------------------------------------
+ // corpus gate: prove the generated input reached the branches this test exists for
+
+ private static void recordBranches(int[] counters, ClusteringPrefix reference, AbstractType>[] types)
+ {
+ int size = reference.size();
+ if (size > 32)
+ counters[SEEN_SECOND_HEADER_BLOCK]++;
+ for (int i = 0; i < size; i++)
+ {
+ byte[] component = reference.get(i);
+ if (component == null)
+ {
+ counters[SEEN_NULL]++;
+ }
+ else if (component.length == 0)
+ {
+ counters[SEEN_EMPTY]++;
+ }
+ else
+ {
+ counters[types[i].isValueLengthFixed() ? SEEN_FIXED_PRESENT : SEEN_VARIABLE_PRESENT]++;
+ if (i < size - 1)
+ counters[SEEN_PRESENT_NOT_LAST]++;
+ }
+ }
+ }
+
+ private static void assertCorpusReachedEveryBranch(int[] counters)
+ {
+ List missed = new ArrayList<>();
+ for (int i = 0; i < counters.length; i++)
+ {
+ if (counters[i] == 0)
+ missed.add(BRANCH_LABELS[i]);
+ }
+ if (!missed.isEmpty())
+ fail("the generated corpus never reached: " + String.join("; ", missed) +
+ ". The test passed without exercising them, so raise " +
+ CassandraRelevantProperties.TEST_CLUSTERING_PREFIX_VIEW_EXAMPLES.getKey() +
+ " or fix the generator; do not treat this run as coverage.");
+ }
+}
diff --git a/test/unit/org/apache/cassandra/io/sstable/CursorIndexWriterOffsetWidthTest.java b/test/unit/org/apache/cassandra/io/sstable/CursorIndexWriterOffsetWidthTest.java
index cb4522df54b8..859b3abfbb39 100644
--- a/test/unit/org/apache/cassandra/io/sstable/CursorIndexWriterOffsetWidthTest.java
+++ b/test/unit/org/apache/cassandra/io/sstable/CursorIndexWriterOffsetWidthTest.java
@@ -20,6 +20,7 @@
import org.junit.Test;
+import org.apache.cassandra.db.DecoratedKey;
import org.apache.cassandra.db.DeletionTime;
import static org.junit.Assert.assertEquals;
@@ -98,7 +99,7 @@ public void rowWritten(UnfilteredDescriptor descriptor, long rowStart, long rowE
}
@Override
- public void endPartition(byte[] key, int keyLength, int headerLength,
+ public void endPartition(DecoratedKey key, byte[] keyBytes, int keyLength, int headerLength,
DeletionTime partitionDeletionTime, long partitionEnd,
ClusteringDescriptor lastName)
{