Skip to content

Commit bce9a67

Browse files
Merge pull request #367 from MPCoreDeveloper/perf/tombstone-delete
perf: durable Columnar DELETE in-place via tombstone markers (no flush rewrite)
2 parents fcafdfa + 0d85ac9 commit bce9a67

9 files changed

Lines changed: 235 additions & 21 deletions

File tree

‎docs/CHANGELOG.md‎

Lines changed: 11 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -42,6 +42,17 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
4242
overflow arena is reclaimed on the next explicit VACUUM/compaction. Regression: delete half the
4343
rows, `Flush`, reopen ÔÇö exactly the remaining rows come back. Measured DELETE in the `--pk`
4444
harness now includes this durability rewrite (~18.6K ops/s when deleting 10K of 100K rows).
45+
- **Durable DELETE is now in-place via tombstones (no rewrite)** ÔÇö non-transactional Columnar
46+
deletes write a tombstone marker (the record's 4-byte length prefix is replaced by the NEGATIVE
47+
slot size) instead of queueing a flush-time rewrite, and every raw record enumerator/compactor
48+
skips the slot, so DELETE survives a reopen in O(delete). Flush/dispose compaction now only
49+
covers transactional deletes (a marker written before commit would survive a rollback that
50+
restores the row in the PK index; those deletes are compacted against the current PK after commit
51+
ÔÇö rollback-safe). Tombstoned space is reclaimed by the tombstone-aware `CompactTable`, including
52+
the ULID-migration compaction (which previously produced an empty file when tombstones were
53+
present because `CompactTable` broke on a negative prefix). Measured `--pk` DELETE (10K of 100K
54+
rows): ~0.54s/18.6K ops/s (flush rewrite)  **~0.16s/~64K ops/s (legacy)** and
55+
**~0.12s/~81K ops/s (fixed-width)** ÔÇö DELETE is back on par with UPDATE.
4556
- **Dedicated SQL batch-INSERT fast path (WP14)** ÔÇö `ExecuteBatchSQL` INSERTs no longer build a
4657
per-row `Dictionary<string, object>`; VALUES clauses are parsed directly into column-ordered
4758
`object[]` rows (`PreparedInsertStatement.ParseValuesToArray`) and inserted via the new

‎src/SharpCoreDB/DataStructures/Table.CRUD.cs‎

Lines changed: 43 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -1364,7 +1364,20 @@ private List<Dictionary<string, object>> ScanRowsWithSimdAndFilterStale(byte[] d
13641364
dataSpan.Slice(filePosition, 4));
13651365

13661366
const int MaxRecordSize = 1_000_000_000;
1367-
if (recordLength < 0 || recordLength > MaxRecordSize)
1367+
if (recordLength < 0)
1368+
{
1369+
// Tombstoned (deleted) record: the prefix stores the negative slot size to skip.
1370+
int slotSize = -recordLength;
1371+
if (slotSize < 4)
1372+
{
1373+
break;
1374+
}
1375+
1376+
filePosition += slotSize;
1377+
continue;
1378+
}
1379+
1380+
if (recordLength > MaxRecordSize)
13681381
{
13691382
break;
13701383
}
@@ -2586,9 +2599,10 @@ private bool HasExplicitNamedIndex(string column)
25862599
foreach (var (storagePosition, _) in recordsToDelete)
25872600
engine.Delete(Name, storagePosition);
25882601
}
2589-
else
2602+
else if (StorageMode != StorageMode.Columnar)
25902603
{
2591-
// Track Columnar logical deletes so flush-time compaction makes them durable.
2604+
// Legacy logical-delete accounting (Columnar deletes are tracked in the branch below,
2605+
// which decides between an immediate durable tombstone and deferred flush compaction).
25922606
Interlocked.Add(ref _pendingLogicalDeletes, recordsToDelete.Count);
25932607
}
25942608

@@ -2639,6 +2653,20 @@ private bool HasExplicitNamedIndex(string column)
26392653
}
26402654
}
26412655

2656+
if (this.storage is { IsInTransaction: true })
2657+
{
2658+
// Transactional delete: writing the tombstone now would survive a rollback that
2659+
// restores the row in the PK index, so defer the physical removal to the post-commit
2660+
// flush compaction (rollback-safe: that rewrite is driven by the CURRENT PK, which
2661+
// still contains the row after a rollback).
2662+
Interlocked.Add(ref _pendingLogicalDeletes, recordsToDelete.Count);
2663+
}
2664+
else
2665+
{
2666+
// Durable DELETE: physically mark the removed records so a reopen skips them.
2667+
TombstoneDeletedPositions(positions);
2668+
}
2669+
26422670
TryAutoCompact();
26432671
}
26442672

@@ -3184,8 +3212,19 @@ this.storage is null ||
31843212
hashIdx.RemoveBatchKeys(decoded, positions);
31853213
}
31863214

3215+
if (this.storage is { IsInTransaction: true })
3216+
{
3217+
// Transactional delete: defer physical removal to the post-commit flush compaction
3218+
// (see DeleteRecordsCore — a tombstone would survive a rollback that restores the row).
3219+
Interlocked.Add(ref _pendingLogicalDeletes, count);
3220+
}
3221+
else
3222+
{
3223+
// Durable DELETE: physically mark the removed records so a reopen skips them.
3224+
TombstoneDeletedPositions(positions);
3225+
}
3226+
31873227
Interlocked.Add(ref _cachedRowCount, -count);
3188-
Interlocked.Add(ref _pendingLogicalDeletes, count);
31893228
Interlocked.Increment(ref _bulkContiguousDeleteBatches);
31903229
return true;
31913230
}

‎src/SharpCoreDB/DataStructures/Table.Compaction.cs‎

Lines changed: 33 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -8,6 +8,7 @@ namespace SharpCoreDB.DataStructures;
88
using SharpCoreDB.Storage.Engines;
99
using System;
1010
using System.Collections.Generic;
11+
using System.IO;
1112
using System.Linq;
1213
using System.Threading.Tasks;
1314

@@ -52,12 +53,14 @@ public void TryAutoCompact()
5253
}
5354

5455
/// <summary>
55-
/// Physically removes rows that were logically deleted since the last flush (Columnar tables
56-
/// with a primary key) so DELETE survives a reopen — the on-load PK-index rebuild would otherwise
57-
/// resurrect them from the untouched <c>.dat</c>. Runs synchronously at flush/dispose, outside a
58-
/// transaction, when any logical deletes are pending. Data-file only: the overflow arena is left
59-
/// untouched (its space is reclaimed by a later explicit compaction/VACUUM) so the flush stays
60-
/// proportional to the rewritten live rows.
56+
/// Makes TRANSACTIONAL deletes durable (Columnar tables with a primary key). Deletes issued
57+
/// inside a transaction are intentionally not tombstoned at delete time (a marker would survive a
58+
/// rollback that restores the row in the PK index), so they are physically removed here by
59+
/// compacting against the CURRENT PK once the owning transaction has committed. Runs at
60+
/// flush/dispose, outside a transaction, when any transactional deletes are pending. No-op when
61+
/// none are pending — non-transactional DELETE is already durable via the per-row tombstone and
62+
/// must not pay a rewrite. Data-file only: the overflow arena is left untouched (its space is
63+
/// reclaimed by a later explicit compaction/VACUUM).
6164
/// </summary>
6265
public void CompactPendingDeletes()
6366
{
@@ -105,6 +108,30 @@ public void CompactPendingDeletes()
105108
}
106109
}
107110

111+
/// <summary>
112+
/// Writes the deleted-record marker over the length prefix of each position that is physically
113+
/// present in the data file, making the Columnar logical delete durable across reopen (readers
114+
/// skip the marker) without rewriting the remaining rows. Rows appended inside an uncommitted
115+
/// transaction (positions beyond the current file length) are skipped — their delete commits
116+
/// with the rest of the transaction.
117+
/// </summary>
118+
private void TombstoneDeletedPositions(long[] positions)
119+
{
120+
if (this.storage is null || positions.Length == 0)
121+
{
122+
return;
123+
}
124+
125+
long fileLength = File.Exists(DataFile) ? new FileInfo(DataFile).Length : 0;
126+
foreach (var position in positions)
127+
{
128+
if (position >= 0 && position + 4 <= fileLength)
129+
{
130+
this.storage.TombstoneRecord(DataFile, position);
131+
}
132+
}
133+
}
134+
108135
/// <summary>
109136
/// Compacts the table storage by removing deleted and stale rows.
110137
/// Only applicable for columnar (append-only) storage mode.

‎src/SharpCoreDB/DataStructures/Table.ParallelScan.cs‎

Lines changed: 24 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -173,6 +173,18 @@ private List<Dictionary<string, object>> ScanRowsParallel(byte[] data, string? w
173173
int recordLength = System.Buffers.Binary.BinaryPrimitives.ReadInt32LittleEndian(
174174
data.AsSpan(filePosition, 4));
175175

176+
if (recordLength < 0)
177+
{
178+
int slotSize = -recordLength; // tombstoned slot: skip the full slot
179+
if (slotSize < 4)
180+
{
181+
break;
182+
}
183+
184+
filePosition += slotSize;
185+
continue;
186+
}
187+
176188
if (recordLength <= 0 || recordLength > 1_000_000_000) break;
177189

178190
currentRecordCount++;
@@ -207,6 +219,18 @@ private List<Dictionary<string, object>> ScanRowsParallel(byte[] data, string? w
207219
int recordLength = System.Buffers.Binary.BinaryPrimitives.ReadInt32LittleEndian(
208220
partitionData.Slice(localFilePosition, 4));
209221

222+
if (recordLength < 0)
223+
{
224+
int slotSize = -recordLength; // tombstoned slot: skip the full slot
225+
if (slotSize < 4)
226+
{
227+
break;
228+
}
229+
230+
localFilePosition += slotSize;
231+
continue;
232+
}
233+
210234
if (recordLength <= 0 || recordLength > 1_000_000_000) break;
211235

212236
if (localFilePosition + 4 + recordLength > partitionData.Length) break;

‎src/SharpCoreDB/DataStructures/Table.Scanning.cs‎

Lines changed: 15 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -65,10 +65,23 @@ private List<Dictionary<string, object>> ScanRowsWithSimd(byte[] data, string? w
6565
// ✅ C# 14: Range operator - extract length prefix span first
6666
var lengthSpan = dataSpan[filePosition..(filePosition + 4)];
6767
int recordLength = System.Buffers.Binary.BinaryPrimitives.ReadInt32LittleEndian(lengthSpan);
68-
68+
69+
if (recordLength < 0)
70+
{
71+
// Tombstoned (deleted) record: the prefix stores the negative slot size to skip.
72+
int slotSize = -recordLength;
73+
if (slotSize < 4)
74+
{
75+
break;
76+
}
77+
78+
filePosition += slotSize;
79+
continue;
80+
}
81+
6982
// Sanity check: record length must be reasonable
7083
const int MaxRecordSize = 1_000_000_000; // 1 GB max per record
71-
if (recordLength < 0 || recordLength > MaxRecordSize)
84+
if (recordLength > MaxRecordSize)
7285
{
7386
break;
7487
}

‎src/SharpCoreDB/DataStructures/Table.cs‎

Lines changed: 6 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -733,9 +733,10 @@ public void Flush()
733733
}
734734
}
735735

736-
// Durable DELETE across reopen: physically compact rows that were logically deleted
737-
// since the last flush (Columnar + PK, outside a transaction). Runs after the engine
738-
// and any transaction buffer have been flushed.
736+
// Transactional deletes (which are intentionally NOT tombstoned at delete time so a
737+
// rollback stays correct) become durable by compacting against the current PK once the
738+
// owning transaction has committed. No-op when there are no deferred transactional
739+
// deletes, so non-transactional DELETE keeps the O(delete) tombstone path.
739740
CompactPendingDeletes();
740741

741742
// Flush indexes
@@ -783,7 +784,8 @@ protected virtual void Dispose(bool disposing)
783784
{
784785
if (disposing)
785786
{
786-
// Durable DELETE across reopen for flows that dispose without an explicit flush.
787+
// Transactional deletes deferred past their commit flush (e.g. dispose-without-flush
788+
// flows) are compacted here; no-op when none are pending.
787789
CompactPendingDeletes();
788790

789791
// Dispose storage engine first

‎src/SharpCoreDB/Interfaces/IStorage.cs‎

Lines changed: 9 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -170,6 +170,15 @@ bool OverwriteRecordAtSameLength(string path, long offset, byte[] data) =>
170170
/// </summary>
171171
bool AreRecordsEncrypted(string path) => false;
172172

173+
/// <summary>
174+
/// Marks the record whose 4-byte length prefix sits at <paramref name="offset"/> as deleted by
175+
/// replacing the prefix with the NEGATIVE slot size (4-byte prefix + payload). Every record
176+
/// enumerator treats a negative prefix as a deleted record and skips |value| bytes, so the
177+
/// delete survives a reopen without rewriting the file. The default returns false (unsupported
178+
/// layout / mock storage).
179+
/// </summary>
180+
bool TombstoneRecord(string path, long offset) => false;
181+
173182
/// <summary>
174183
/// Enumerates every record in a table data file, yielding the literal file offset of the
175184
/// 4-byte length prefix (the offset returned by <see cref="AppendBytes"/>) together with the

‎src/SharpCoreDB/Services/Storage.Append.cs‎

Lines changed: 63 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -563,6 +563,56 @@ public bool HasBufferedOverwrite(string path) =>
563563
/// <inheritdoc />
564564
public bool AreRecordsEncrypted(string path) => UseRecordEncryption && FileHasEncryptedHeader(path);
565565

566+
/// <inheritdoc />
567+
public bool TombstoneRecord(string path, long offset)
568+
{
569+
// Read the current slot size so the marker can encode the exact number of bytes to skip
570+
// (4-byte prefix + payload), keeping every record enumerator aligned.
571+
int slotSize;
572+
try
573+
{
574+
SafeFileHandle readHandle = GetOrOpenReadHandle(path);
575+
Span<byte> lengthBuffer = stackalloc byte[4];
576+
if (RandomAccess.Read(readHandle, lengthBuffer, offset) != 4)
577+
{
578+
return false;
579+
}
580+
581+
int currentLength = BinaryPrimitives.ReadInt32LittleEndian(lengthBuffer);
582+
if (currentLength <= 0)
583+
{
584+
return false; // already tombstoned or invalid
585+
}
586+
587+
slotSize = 4 + currentLength;
588+
}
589+
catch (IOException)
590+
{
591+
return false;
592+
}
593+
594+
Span<byte> marker = stackalloc byte[4];
595+
BinaryPrimitives.WriteInt32LittleEndian(marker, -slotSize);
596+
597+
try
598+
{
599+
WriteRecordInPlace(path, offset, marker, ReadOnlySpan<byte>.Empty);
600+
}
601+
catch (IOException)
602+
{
603+
return false;
604+
}
605+
606+
// Invalidate the app-level page cache (mirrors the other in-place writers).
607+
if (this.pageCache != null)
608+
{
609+
int pageId = ComputePageId(path, offset);
610+
this.pageCache.EvictPage(pageId);
611+
}
612+
613+
return true;
614+
}
615+
566616
/// <inheritdoc />
567617
[MethodImpl(MethodImplOptions.AggressiveOptimization)]
568618
public long[] AppendBytesMultiple(string path, List<byte[]> dataBlocks)
@@ -924,6 +974,19 @@ private void FlushBufferedOverwrites()
924974
}
925975

926976
int length = BinaryPrimitives.ReadInt32LittleEndian(lengthBuffer);
977+
if (length < 0)
978+
{
979+
// Tombstoned (deleted) record: the prefix stores the negative slot size to skip.
980+
int slotSize = -length;
981+
if (slotSize < 4)
982+
{
983+
yield break;
984+
}
985+
986+
position += slotSize;
987+
continue;
988+
}
989+
927990
if (length <= 0 || length > MaxRecordSize || position + 4 + length > fileLength)
928991
{
929992
yield break; // Invalid or incomplete record tail

‎src/SharpCoreDB/Storage/Engines/AppendOnlyEngine.cs‎

Lines changed: 31 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -214,8 +214,21 @@ public void Delete(string tableName, long storageReference)
214214
// Read record length (4 bytes, little-endian)
215215
int recordLength = System.Buffers.Binary.BinaryPrimitives.ReadInt32LittleEndian(
216216
allData.AsSpan((int)position, 4));
217-
218-
if (recordLength <= 0 || position + 4 + recordLength > allData.Length)
217+
218+
if (recordLength < 0)
219+
{
220+
// Tombstoned (deleted) record: the prefix stores the negative slot size to skip.
221+
int slotSize = -recordLength;
222+
if (slotSize < 4)
223+
{
224+
break;
225+
}
226+
227+
position += slotSize;
228+
continue;
229+
}
230+
231+
if (recordLength == 0 || position + 4 + recordLength > allData.Length)
219232
{
220233
break; // Invalid or incomplete record
221234
}
@@ -354,10 +367,23 @@ public long CompactTable(string tableName, List<long> activePositions)
354367

355368
int recordLength = System.Buffers.Binary.BinaryPrimitives.ReadInt32LittleEndian(
356369
allData.AsSpan((int)position, 4));
357-
358-
if (recordLength <= 0 || position + 4 + recordLength > allData.Length)
370+
371+
if (recordLength < 0)
372+
{
373+
// Tombstoned (deleted) record: the prefix stores the negative slot size to skip.
374+
int slotSize = -recordLength;
375+
if (slotSize < 4)
376+
{
377+
break;
378+
}
379+
380+
position += slotSize;
381+
continue;
382+
}
383+
384+
if (recordLength == 0 || position + 4 + recordLength > allData.Length)
359385
break;
360-
386+
361387
// Check if this position is active
362388
if (activeSet.Contains(position))
363389
{

0 commit comments

Comments
 (0)