diff --git a/docs/turso-benchmark-portability.md b/docs/turso-benchmark-portability.md index d6f3a75..acfd935 100644 --- a/docs/turso-benchmark-portability.md +++ b/docs/turso-benchmark-portability.md @@ -17,6 +17,7 @@ corresponding Rust internal API. | `core/benches/fts_benchmark.rs` | Cold/warm search, selectivity, ingest, commit/merge churn | Direct Ahtola index-method port | | `core/benches/fts_comparison_benchmark.rs` | Ahtola-style FTS versus SQLite FTS5 | Direct SQL port with storage-model labels | | (Ahtola workload) `TursoFtsWorkloadBenchmarks` | Zipf 20,000-term corpus: rare/mid/common terms, AND, phrase, prefix, ranked top-10, write-then-search cycle, versus SQLite FTS5 | Managed workload complementing the upstream ports | +| (Ahtola workload) `DotNetHotspotBenchmarks` | .NET application access patterns through the `Microsoft.Data.Sqlite`-compatible facade on a file-backed WAL database: pooled connection-per-operation, PK/unique-key lookups (new and reused commands), COUNT probes, typed/`GetValue`/`GetFieldValue` materialization, paging, keyset, `IN` lists, LIKE, JOIN + GROUP BY, multiple result sets, autocommit and batched INSERT, UPDATE, UPSERT, EF-style unit of work with RETURNING, BLOB round trip | Managed workload versus `Microsoft.Data.Sqlite` on the same file | | `core/benches/graph_queries_benchmark.rs` | Analyzed/unanalyzed graph queries | Managed deterministic-fixture adaptation | | `core/benches/tpc_h_benchmark.rs` | Supported TPC-H query execution | Managed asset and supported-query adaptation | | `core/benches/json_benchmark.rs` | JSONB conversion and JSON Patch | Direct SQL port | diff --git a/src/Ahtola.Core/CowChunkedList.cs b/src/Ahtola.Core/CowChunkedList.cs index bac155a..b1a8ac8 100644 --- a/src/Ahtola.Core/CowChunkedList.cs +++ b/src/Ahtola.Core/CowChunkedList.cs @@ -29,10 +29,37 @@ internal sealed class CowChunkedList : IList, IReadOnlyList private object _owner = new(); private int _version; + // Process-unique identity of this list's current contents; 0 while unassigned (see + // ContentStamp). Every mutation clears it and ShareFrom copies it, so two lists report the + // same stamp only while they hold exactly the same elements in the same order. + private long _contentStamp; + private static long s_contentStampSequence; + public int Count => _count; public bool IsReadOnly => false; + /// + /// A process-unique token for this list's exact current contents. It changes on every + /// mutation and is carried by , so equal stamps on any two lists prove + /// identical contents; a working copy that diverges and is discarded can never alias another + /// copy's state, which a per-list counter such as a revision cannot guarantee. Assigned + /// lazily so mutations only clear a field. + /// + public long ContentStamp + { + get + { + var stamp = Volatile.Read(ref _contentStamp); + if (stamp != 0) + return stamp; + + var fresh = Interlocked.Increment(ref s_contentStampSequence); + var existing = Interlocked.CompareExchange(ref _contentStamp, fresh, 0); + return existing == 0 ? fresh : existing; + } + } + public T this[int index] { get @@ -46,10 +73,21 @@ public T this[int index] if ((uint)index >= (uint)_count) ThrowIndexOutOfRange(); WritableChunk(index >> Shift)[index & Mask] = value; - _version++; + Mutated(); } } + /// + /// Whether chunk (elements chunkIndex * ChunkSize onward) + /// is the same physical chunk in both lists. A chunk is copied before any write, so a shared + /// chunk holds identical elements in both; when the lists also have equal counts, every + /// element of that chunk below the count is equal without comparing them. + /// + internal bool SharesChunkWith(CowChunkedList other, int chunkIndex) + => chunkIndex < _chunkCount + && chunkIndex < other._chunkCount + && ReferenceEquals(_chunks[chunkIndex], other._chunks[chunkIndex]); + /// Makes this list an O(chunks) copy-on-write clone of . public void ShareFrom(CowChunkedList source) { @@ -62,6 +100,7 @@ public void ShareFrom(CowChunkedList source) _owner = new object(); source._owner = new object(); _version++; + _contentStamp = source.ContentStamp; } public void Add(T item) @@ -72,7 +111,7 @@ public void Add(T item) : WritableChunk(_count >> Shift); items[offset] = item; _count++; - _version++; + Mutated(); } public void AddRange(IEnumerable items) @@ -88,7 +127,7 @@ public void Clear() _chunks = []; _chunkCount = 0; _count = 0; - _version++; + Mutated(); } public bool Contains(T item) => IndexOf(item) >= 0; @@ -163,7 +202,7 @@ public void Insert(int index, T item) } WritableChunk(firstChunk)[index & Mask] = item; - _version++; + Mutated(); } public bool Remove(T item) @@ -199,7 +238,7 @@ public void RemoveAt(int index) _count--; if ((_count & Mask) == 0 && _chunkCount > (_count >> Shift)) _chunks[--_chunkCount] = null!; - _version++; + Mutated(); } public void Sort() @@ -208,7 +247,7 @@ public void Sort() Array.Sort(items); for (var index = 0; index < items.Length; index++) WritableChunk(index >> Shift)[index & Mask] = items[index]; - _version++; + Mutated(); } public List GetRange(int index, int count) @@ -238,6 +277,12 @@ public IEnumerator GetEnumerator() IEnumerator IEnumerable.GetEnumerator() => GetEnumerator(); + private void Mutated() + { + _version++; + _contentStamp = 0; + } + private T[] WritableChunk(int chunk) { var current = _chunks[chunk]; diff --git a/src/Ahtola.Core/EmbeddedDatabase.cs b/src/Ahtola.Core/EmbeddedDatabase.cs index 944f8f1..03d8c16 100644 --- a/src/Ahtola.Core/EmbeddedDatabase.cs +++ b/src/Ahtola.Core/EmbeddedDatabase.cs @@ -291,6 +291,9 @@ public EmbeddedPostCommitMaintenanceException(Exception maintenanceFailure) public sealed partial class EmbeddedDatabase : IDisposable { private const int MaximumTriggerDepth = 1_000; + // A join whose left side has at most this many rows (or at most an eighth of the right + // table) probes the right base table per left row instead of hashing all of it. + private const int SmallJoinProbeRows = 64; private const int MaximumForeignKeyActionDepth = 1_000; private const int RecursiveTriggerStackSize = 32 * 1024 * 1024; @@ -382,6 +385,9 @@ public sealed partial class EmbeddedDatabase : IDisposable private readonly bool _readOnly; private readonly bool _foreignReadOnly; private SqlitePagerViewToken _foreignViewToken; + + // When _foreignViewToken was captured; see IsForeignViewTokenRacy. + private DateTimeOffset _foreignViewTokenCapturedAt; private SqlitePagerViewToken _ownedViewToken; private FileCatalogVersion _fileCatalogVersion; private long _ownedCommittedGeneration; @@ -438,6 +444,7 @@ private EmbeddedDatabase( _fileCatalogWriteLock = fileCatalogWriteLock; _readOnly = readOnly; _foreignReadOnly = foreignReadOnly; + _foreignViewTokenCapturedAt = DateTimeOffset.UtcNow; _foreignViewToken = foreignReadOnly ? fileStore.CaptureCommittedViewToken() : default; @@ -1334,12 +1341,6 @@ internal sealed class StatementExecutionState(long lastInsertRowId) /// conn_txn_id for an autocommit statement; -1 until first set. public long ConnTxnId { get; set; } = -1; - // Automatic (transient) equality indexes built on demand for scans whose predicates - // equate a scan column to a scan-constant expression - the shape correlated - // subqueries and pushed-down join side predicates produce once per outer row. - // Entries validate against RowStore.Revision, so any table mutation rebuilds them. - public Dictionary TransientLookups { get; } = []; - // Collation-resolution memo. Without caching, every projection of a query over a // derived or compound source re-derives the full collation scope of the source // below it, multiplying cost by the column count at each nesting level - EF's @@ -1409,25 +1410,17 @@ public int GetHashCode(CollationMemoKey key) key.LexicalCteCount); } + // Key of a shared equality lookup (EmbeddedTable.GetOrCreateDerived). // The comparison affinity SQLite applies to the *scanned column* is part of the bucket's // identity: `t.text_col = ` buckets the column's stored text under its // numeric value, while `t.text_col = ` buckets it verbatim. Two probes // that disagree on that conversion must not share a cached index. internal readonly record struct TransientLookupKey( - EmbeddedTable Table, int ColumnOrdinal, string Collation, bool ColumnConvertsTextToNumeric, bool ColumnConvertsNumericToText); - internal sealed class TransientLookup - { - public long Revision { get; set; } = -1; - - // Canonical key -> table row positions, appended in ascending scan order. - public Dictionary> Buckets { get; } = new(StringComparer.Ordinal); - } - internal sealed class ForeignKeyStatementState { @@ -5048,6 +5041,9 @@ private void PersistFileCatalog( EnsureFileCatalogVersionCurrent(busyTimeout); using var writeRegistration = RegisterCatalogWrite(_databasePath); _fileStore.DeferredCheckpointFrameThreshold = _mvStore is null ? _deferredCheckpointFrameThreshold : 0; + // Without MVCC every commit goes through this store, so the committed catalog is + // exactly the file's content. + var unchangedRowsAreDurable = _mvStore is null; try { var committedVersion = checkpointAfterCommit @@ -5060,7 +5056,8 @@ private void PersistFileCatalog( forceFullRewrite, previousTables: _tables, targetedIndexRebuild: targetedIndexRebuild, - maximumPageCount: maxPageCount) + maximumPageCount: maxPageCount, + unchangedRowsAreDurable: unchangedRowsAreDurable) : _fileStore.PersistForMvccCheckpoint( catalog.Tables, catalog.Views, @@ -5305,7 +5302,7 @@ private void RefreshForeignCatalogIfChangedLocked() return; var token = _fileStore.CaptureCommittedViewToken(); - if (token == _foreignViewToken) + if (token == _foreignViewToken && !IsForeignViewTokenRacy(token)) return; var replacement = EmbeddedFileStore.Open( @@ -5328,6 +5325,7 @@ private void RefreshForeignCatalogIfChangedLocked() RefreshCollationResolverBinding(); _fileCatalogVersion = ReadFileCatalogVersion(_fileSystem, _databasePath, foreignReadOnly: true); PublishCatalog(replacementCatalog); + _foreignViewTokenCapturedAt = DateTimeOffset.UtcNow; _foreignViewToken = _fileStore.CaptureCommittedViewToken(); previous.Dispose(); } @@ -5357,6 +5355,19 @@ internal void RefreshForeignCatalogForStatementIfNeeded() } } + // File timestamps only advance with the system clock tick (15.6 ms on many Windows hosts, + // whole seconds on some file systems), and a peer that checkpoints a WAL-mode database and + // deletes its WAL changes nothing else a foreign reader can observe: the main file keeps its + // size and its change counter. A write landing in the same tick the token was captured in + // therefore leaves the token equal. Like git's "racily clean" index entries, a token whose + // database stamp is that close to its capture time does not prove the file unchanged, so the + // caller re-reads; once the file has been quiet for the window, the token is trusted again. + private static readonly TimeSpan ForeignStampRacyWindow = TimeSpan.FromSeconds(2); + + private bool IsForeignViewTokenRacy(SqlitePagerViewToken token) + => token.DatabaseStamp is { } stamp + && stamp.LastWriteTimeUtc >= _foreignViewTokenCapturedAt - ForeignStampRacyWindow; + /// /// Statement-boundary adoption for owned file-backed connections in autocommit. /// SQLite serves every autocommit statement from the latest committed view, so @@ -5726,12 +5737,7 @@ private static string GetFileCatalogLockPath(IFileSystem fileSystem, string path if (AhtolaEncryptionFileSystem.Unwrap(fileSystem) is not PhysicalFileSystem) return null; - var lockPath = Path.GetFullPath(path); - if (OperatingSystem.IsWindows()) - lockPath = lockPath.ToUpperInvariant(); - var name = "Ahtola.ManagedCatalog." - + Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(lockPath))); - var mutex = new Mutex(initiallyOwned: false, name); + var mutex = new Mutex(initiallyOwned: false, GetFileCatalogMutexName(path)); try { try @@ -5754,6 +5760,30 @@ private static string GetFileCatalogLockPath(IFileSystem fileSystem, string path } } + // Every commit enters the catalog mutex; its name is a pure function of the path, and + // hashing the normalized path each time showed up on autocommit writes. + private static readonly System.Collections.Concurrent.ConcurrentDictionary s_fileCatalogMutexNames = + new(StringComparer.Ordinal); + + private static string GetFileCatalogMutexName(string path) + { + if (s_fileCatalogMutexNames.TryGetValue(path, out var cached)) + return cached; + + var lockPath = Path.GetFullPath(path); + if (OperatingSystem.IsWindows()) + lockPath = lockPath.ToUpperInvariant(); + var name = "Ahtola.ManagedCatalog." + + Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(lockPath))); + // A relative path resolves against the current directory, which can change. + if (!Path.IsPathFullyQualified(path)) + return name; + if (s_fileCatalogMutexNames.Count >= 1024) + s_fileCatalogMutexNames.Clear(); + s_fileCatalogMutexNames[path] = name; + return name; + } + private sealed class FileCatalogWriteLockScope { private readonly Dictionary _locks = new(StringComparer.OrdinalIgnoreCase); @@ -14586,8 +14616,8 @@ private ExecutionResult PerformInsertEvaluated(InsertStatement statement, SqlVal ? null : ExecuteQuery(statement.Source, parameters, context, outerRow: null).Rows; var rowCount = sourceRows?.Count ?? statement.Rows.Count; - var originalRows = statement.Returning is null ? null : table.Rows.ToArray(); - var originalRowIds = statement.Returning is null ? null : table.RowIds.ToArray(); + var originalRows = ReturningReadsTableState(statement.Returning) ? table.Rows.ToArray() : null; + var originalRowIds = ReturningReadsTableState(statement.Returning) ? table.RowIds.ToArray() : null; var rowsToInsert = new List(rowCount); var insertedRowIds = new List(rowCount); if (sourceRows is not null) @@ -14630,11 +14660,13 @@ private ExecutionResult PerformInsertEvaluated(InsertStatement statement, SqlVal table.HasRowid ? insertedRowIds.Cast().ToArray() : null, - CaptureIncrementalInsertReturningSnapshots( - originalRows!, - originalRowIds!, - rowsToInsert, - insertedRowIds)); + originalRows is null + ? null + : CaptureIncrementalInsertReturningSnapshots( + originalRows, + originalRowIds!, + rowsToInsert, + insertedRowIds)); } return new ExecutionResult([], [], rowsToInsert.Count, rowsToInsert.Count > 0) @@ -14654,8 +14686,8 @@ private ExecutionResult PerformAutoIncrementInsertEvaluated( ? null : ExecuteQuery(statement.Source, parameters, context, outerRow: null).Rows; var rowCount = sourceRows?.Count ?? statement.Rows.Count; - var originalRows = statement.Returning is null ? null : table.Rows.ToArray(); - var originalRowIds = statement.Returning is null ? null : table.RowIds.ToArray(); + var originalRows = ReturningReadsTableState(statement.Returning) ? table.Rows.ToArray() : null; + var originalRowIds = ReturningReadsTableState(statement.Returning) ? table.RowIds.ToArray() : null; var rowsToInsert = new List(rowCount); var insertedRowIds = new List(rowCount); var backup = CloneTablesShallow(context.Tables); @@ -14719,11 +14751,13 @@ private ExecutionResult PerformAutoIncrementInsertEvaluated( table.HasRowid ? insertedRowIds.Cast().ToArray() : null, - CaptureIncrementalInsertReturningSnapshots( - originalRows!, - originalRowIds!, - rowsToInsert, - insertedRowIds)); + originalRows is null + ? null + : CaptureIncrementalInsertReturningSnapshots( + originalRows, + originalRowIds!, + rowsToInsert, + insertedRowIds)); } return new ExecutionResult([], [], rowsToInsert.Count, rowsToInsert.Count > 0) @@ -15463,7 +15497,7 @@ private ExecutionResult ExecuteUpdateWithRowTriggers( updatedRows.Add(updated); updatedRowIds.Add(newRowId); - if (statement.Returning is not null) + if (ReturningReadsTableState(statement.Returning)) { // RETURNING sees the row after the write but before the AFTER // trigger fires (returning.sqltest update-returning-after-trigger @@ -15518,7 +15552,9 @@ private ExecutionResult ExecuteUpdateWithRowTriggers( updatedRows.Count > 0, parameters, context, - returningTableSnapshots: returningSnapshots); + returningTableSnapshots: ReturningReadsTableState(statement.Returning) + ? returningSnapshots + : null); } catch { @@ -15675,8 +15711,8 @@ private ExecutionResult PerformUpdateEvaluated( var rowIds = table.RowIds.Count == table.Rows.Count ? table.RowIds.ToList() : Enumerable.Range(1, table.Rows.Count).Select(position => (long)position).ToList(); - var originalRows = statement.Returning is null ? null : table.Rows.ToArray(); - var originalRowIds = statement.Returning is null ? null : rowIds.ToArray(); + var originalRows = ReturningReadsTableState(statement.Returning) ? table.Rows.ToArray() : null; + var originalRowIds = ReturningReadsTableState(statement.Returning) ? rowIds.ToArray() : null; var updatedRows = statement.Returning is null ? null : new List(); var updatedRowIds = statement.Returning is null ? null : new List(); var updatedPositions = new List(); @@ -15788,10 +15824,10 @@ private ExecutionResult PerformUpdateEvaluated( rowsAffected++; } - var returningSnapshots = statement.Returning is null + var returningSnapshots = originalRows is null ? null : CaptureIncrementalUpdateReturningSnapshots( - originalRows!, + originalRows, originalRowIds!, rows, rowIds, @@ -16283,12 +16319,30 @@ private EmbeddedSqlException CreateConflict(EmbeddedIndex index) EmbeddedTable table, SqlValue[] parameters, QueryContext context) + => TryGetDmlEqualityCandidatePositions( + statement.TableName, + statement.Alias, + statement.Where, + table, + parameters, + context); + + // Positions of the rows a DML WHERE's leading scan-constant equality can match: a superset of + // the rows the full WHERE selects, so evaluating only these rows is exact (the same pruning + // the SELECT transient lookup applies). Null when the WHERE has no such equality. + private HashSet? TryGetDmlEqualityCandidatePositions( + string tableName, + string? alias, + Expression? where, + EmbeddedTable table, + SqlValue[] parameters, + QueryContext context) { - if (statement.Where is null + if (where is null || !TryCreateTransientEqualityLookup( - new NamedTableSource(statement.TableName, statement.Alias), + new NamedTableSource(tableName, alias), table, - statement.Where, + where, context, outerRow: null, out var lookup, @@ -16310,11 +16364,60 @@ private EmbeddedSqlException CreateConflict(EmbeddedIndex index) return []; var candidates = new HashSet(); - for (var position = 0; position < table.Rows.Count; position++) + var rows = table.Rows; + // The rowid alias column holds each row's rowid, so a numeric probe that is exactly an + // integer (doubles are exact up to 2^53, where the scan below compares) can match only the + // row with that rowid; a non-integral probe matches none. + if (lookup.ColumnOrdinal == table.RowidAliasColumnIndex + && !lookup.ColumnConvertsTextToNumeric + && !lookup.ColumnConvertsNumericToText + && EquiJoinHashIndex.TryGetNumericKeyBits(key, out var rowidProbeBits)) + { + var number = BitConverter.Int64BitsToDouble(rowidProbeBits); + if (double.IsFinite(number) && Math.Abs(number) <= 9007199254740992d) + { + if (Math.Floor(number) != number) + return candidates; + + var position = table.TryGetRowIdPosition((long)number, out _); + if (position >= 0 + && rows[position][lookup.ColumnOrdinal] is { Kind: SqlValueKind.Integer } cell + && EquiJoinHashIndex.GetNumericKeyBits(cell.AsInteger()) == rowidProbeBits) + { + candidates.Add(position); + return candidates; + } + } + } + + // A numeric probe against a column compared without conversion matches exactly the + // numeric cells whose canonical bits are equal (CanonicalizeJoinKeyValue's "N" key); + // comparing the bits directly avoids formatting a key string for every row, which made + // UPDATE/DELETE ... WHERE id = ? cost microseconds per table row. + if (!lookup.ColumnConvertsTextToNumeric + && !lookup.ColumnConvertsNumericToText + && EquiJoinHashIndex.TryGetNumericKeyBits(key, out var probeBits)) + { + for (var position = 0; position < rows.Count; position++) + { + context.CheckInterrupt(); + var cell = rows[position][lookup.ColumnOrdinal]; + if (cell.Kind is SqlValueKind.Integer or SqlValueKind.Real + && EquiJoinHashIndex.GetNumericKeyBits( + cell.Kind == SqlValueKind.Integer ? cell.AsInteger() : cell.AsReal()) == probeBits) + { + candidates.Add(position); + } + } + + return candidates; + } + + for (var position = 0; position < rows.Count; position++) { context.CheckInterrupt(); var segment = EquiJoinHashIndex.CanonicalizeJoinKeyValue( - table.Rows[position][lookup.ColumnOrdinal], + rows[position][lookup.ColumnOrdinal], lookup.ColumnConvertsTextToNumeric, lookup.ColumnConvertsNumericToText, lookup.Collation); @@ -17831,6 +17934,23 @@ private bool ParentContains( if (selfReferential || parentRows is not RowStore parentRowStore) return ParentContainsLinear(parent, parentRows, childValues, childRow, selfReferential); + // A rowid-alias parent key is the rowid itself, so the parent lookup is a rowid seek, as + // SQLite's OP_NotExists: probe the table's cached rowid set, which is current for exactly + // this row-store state and extended in place by appends, instead of hashing every parent + // row per statement. Only an INTEGER child value (after the parent column's affinity) can + // equal a rowid; anything else takes the general path below. + if (ParentUsesRowidAlias(parent) + && ReferenceEquals(parentRowStore, parent.Table.Rows) + && parent.Table.RowIds.Count == parentRowStore.Count) + { + var parentColumn = parent.ColumnIndices[0]; + var coercedChild = parent.Table.CoerceColumnAffinity( + parent.Table.ColumnDefinitions[parentColumn], + childValues[0]); + if (coercedChild.Kind == SqlValueKind.Integer) + return parent.Table.GetRowIdSet().Ids.Contains(coercedChild.AsInteger()); + } + // Fast path: hash the parent's FK key columns once and probe per child row, turning // N×parentRows comparisons into parentRows + N×O(1). Bucket hits are confirmed with // the exact ValuesMatchParent comparison, so the probe never yields a false positive. @@ -17890,7 +18010,7 @@ private bool TryGetForeignKeyParentProbe( RowStore parentRowStore, [System.Diagnostics.CodeAnalysis.NotNullWhen(true)] out ForeignKeyParentProbe? probe) { - var probeKey = new ForeignKeyParentProbeKey(parent.Table, BuildForeignKeyColumnSignature(parent)); + var probeKey = new ForeignKeyParentProbeKey(parent.TableName, BuildForeignKeyColumnSignature(parent)); if (!_foreignKeyParentProbes.TryGetValue(probeKey, out probe)) { probe = new ForeignKeyParentProbe(); @@ -17903,11 +18023,12 @@ private bool TryGetForeignKeyParentProbe( return true; } - if (probe.Revision == parentRowStore.Revision) + if (ReferenceEquals(probe.Source, parentRowStore) && probe.Revision == parentRowStore.Revision) return true; // Parent mutated since the probe was built (e.g. a trigger inserted into the - // parent mid-statement): rebuild against the current rows. + // parent mid-statement), or this is another statement's working copy of the parent: + // rebuild against the current rows. probe.Buckets.Clear(); if (!BuildForeignKeyParentProbe(parent, parentRowStore, probe)) { @@ -17923,6 +18044,7 @@ private bool BuildForeignKeyParentProbe( RowStore parentRowStore, ForeignKeyParentProbe probe) { + probe.Source = parentRowStore; probe.Revision = parentRowStore.Revision; for (var position = 0; position < parent.ColumnIndices.Count; position++) { @@ -17995,19 +18117,27 @@ private static string BuildForeignKeyColumnSignature(ForeignKeyParent parent) private sealed class ForeignKeyParentProbe { + public RowStore? Source { get; set; } + public long Revision { get; set; } = -1; public Dictionary> Buckets { get; } = new(StringComparer.Ordinal); } - private readonly record struct ForeignKeyParentProbeKey(EmbeddedTable ParentTable, string ColumnSignature) + // Keyed by parent table name, not EmbeddedTable identity: every statement works on its own + // clone of the catalog, so an identity key added one never-reused probe per statement (an + // unbounded leak of whole-table hash maps). The probe's Source/Revision pair is what proves + // it current for the row store being probed. + private readonly record struct ForeignKeyParentProbeKey(string ParentTableName, string ColumnSignature) { public bool Equals(ForeignKeyParentProbeKey other) - => ReferenceEquals(ParentTable, other.ParentTable) + => string.Equals(ParentTableName, other.ParentTableName, StringComparison.OrdinalIgnoreCase) && string.Equals(ColumnSignature, other.ColumnSignature, StringComparison.Ordinal); public override int GetHashCode() - => HashCode.Combine(RuntimeHelpers.GetHashCode(ParentTable), StringComparer.Ordinal.GetHashCode(ColumnSignature)); + => HashCode.Combine( + StringComparer.OrdinalIgnoreCase.GetHashCode(ParentTableName), + StringComparer.Ordinal.GetHashCode(ColumnSignature)); } private bool ValuesMatchParent( @@ -18959,33 +19089,46 @@ private ExecutionResult PerformDeleteEvaluated( statement.Returning, parameters, context); - var rows = new List(table.Rows.Count); - var rowIds = new List(table.Rows.Count); var originalRows = table.Rows.ToArray(); var originalRowIds = table.RowIds.Count == table.Rows.Count ? table.RowIds.ToArray() : Enumerable.Range(1, table.Rows.Count).Select(position => (long)position).ToArray(); + // A leading equality narrows the rows whose WHERE must be evaluated, as for UPDATE. + var candidatePositions = selectedPositions is null + ? TryGetDmlEqualityCandidatePositions( + statement.TableName, + statement.Alias, + statement.Where, + table, + parameters, + context) + : null; var deletedRows = new List(); var deletedRowIds = new List(); + var deletedPositions = new List(); var rowsAffected = 0; - for (var position = 0; position < table.Rows.Count; position++) + for (var position = 0; position < originalRows.Length; position++) { - var row = table.Rows[position]; - var rowid = position < table.RowIds.Count ? table.RowIds[position] : position + 1; - var source = CreateDmlTargetRow(table, statement.TargetQualifier, row, rowid); + if (candidatePositions is not null && !candidatePositions.Contains(position)) + continue; + + var row = originalRows[position]; + var rowid = originalRowIds[position]; var shouldDelete = selectedPositions is not null ? selectedPositions.Contains(position) - : statement.Where is null || IsTrue(Evaluate(statement.Where, parameters, source, context)); + : statement.Where is null + || IsTrue(Evaluate( + statement.Where, + parameters, + CreateDmlTargetRow(table, statement.TargetQualifier, row, rowid), + context)); if (shouldDelete) { rowsAffected++; deletedRows.Add(row); deletedRowIds.Add(rowid); - continue; + deletedPositions.Add(position); } - - rows.Add(row); - rowIds.Add(rowid); } var returningResult = statement.Returning is null @@ -19000,16 +19143,46 @@ private ExecutionResult PerformDeleteEvaluated( rowsAffected > 0, parameters, context, - returningTableSnapshots: CaptureIncrementalDeleteReturningSnapshots( + returningTableSnapshots: !ReturningReadsTableState(statement.Returning) + ? null + : CaptureIncrementalDeleteReturningSnapshots( originalRows, originalRowIds, deletedRows, deletedRowIds)); var revisionBeforeSwap = table.Rows.Revision; - table.Rows.Clear(); - table.Rows.AddRange(rows); - table.RowIds.Clear(); - table.RowIds.AddRange(rowIds); + if (table.RowIds.Count == table.Rows.Count && deletedPositions.Count * 8 <= originalRows.Length) + { + // A few deletions: remove them in place (highest position first) instead of + // re-adding every surviving row, which rewrote the whole table per DELETE. + var rowIdsBefore = table.HasRowid ? table.TryGetCurrentRowIdSet() : null; + for (var index = deletedPositions.Count - 1; index >= 0; index--) + { + table.Rows.RemoveAt(deletedPositions[index]); + table.RowIds.RemoveAt(deletedPositions[index]); + } + + if (deletedPositions.Count > 0) + table.RecordRemovedRowIds(rowIdsBefore, deletedRowIds); + } + else if (deletedPositions.Count > 0 || table.RowIds.Count != table.Rows.Count) + { + var deleted = deletedPositions.ToHashSet(); + var rows = new List(originalRows.Length - deletedPositions.Count); + var rowIds = new List(originalRows.Length - deletedPositions.Count); + for (var position = 0; position < originalRows.Length; position++) + { + if (deleted.Contains(position)) + continue; + rows.Add(originalRows[position]); + rowIds.Add(originalRowIds[position]); + } + + table.Rows.Clear(); + table.Rows.AddRange(rows); + table.RowIds.Clear(); + table.RowIds.AddRange(rowIds); + } // The swap re-adds every kept row; only the deleted rowids changed. table.RecordMethodIndexBulkMutation(deletedRowIds, revisionBeforeSwap); @@ -19166,6 +19339,14 @@ rowFailureLastInsertRowIds is not null }; } + // RETURNING evaluates each projection against the affected row's own values; only a subquery + // reads the table, and only then does evaluation need the per-row table image. Capturing that + // image copied the whole table once per affected row (and swapped it in and out), which made + // INSERT ... RETURNING id, the shape EF Core issues for every insert, linear in table size. + private static bool ReturningReadsTableState(IReadOnlyList? returning) + => returning is not null + && returning.Any(projection => ContainsSubqueryExpression(projection.Expression)); + private static IReadOnlyList CaptureIncrementalInsertReturningSnapshots( IReadOnlyList originalRows, IReadOnlyList originalRowIds, @@ -20310,7 +20491,8 @@ ScanTarget CreateTarget( context, outerRow, predicate: select.Where, - parameters: parameters); + parameters: parameters, + covering: CanSeekIndexCovering(select, plan)); rows = indexed.Rows.Select(row => row.Values).ToArray(); rowIds = plan.Table.HasRowid ? indexed.Rows.Select(row => @@ -33254,6 +33436,190 @@ private static string FormatManagedIndexExplainDetail( : detail; } + /// + /// Whether is exactly ascending rowid order over a plain rowid base + /// table, which scans in that order. + /// + private static bool IsAscendingRowidOrderScan( + SelectStatement statement, + IReadOnlyList orderBy, + QueryContext context) + { + if (orderBy.Count != 1 + || orderBy[0] is not { Descending: false, NullPlacement: NullPlacement.Default } term + || term.Expression is not ColumnExpression { BooleanKeyword: null } column + || statement.Source is not NamedTableSource source + || source.IndexDirective is not null + || IsSchemaTable(source.Name) + || IsCommonTableExpression(source, context) + || context.Views?.ContainsKey(source.Name) == true + || TryGetVirtualTable(context, source, out _) + || !context.Tables.TryGetValue(source.Name, out var table) + || !table.HasRowid + || table.HasMethodIndexes) + { + return false; + } + + var name = column.UnqualifiedName ?? column.Name; + var separator = name.LastIndexOf('.'); + if (separator >= 0) + { + if (!string.Equals(name[..separator], source.Alias ?? source.Name, StringComparison.OrdinalIgnoreCase)) + return false; + name = name[(separator + 1)..]; + } + + return table.TryGetColumnIndex(name, out var columnIndex) + ? columnIndex == table.RowidAliasColumnIndex + : EmbeddedTable.IsRowidAliasName(name); + } + + /// + /// Whether scanning 's index (in its planned direction) already yields + /// rows in order: each ORDER BY term is the matching leading index + /// column of a single-table SELECT, in the scan's effective direction, under the index + /// column's own built-in collation and default NULLS placement. Rows that tie on every term + /// keep index order, which is also what the evaluator's stable sort would have produced. + /// + private bool IndexPlanSatisfiesOrderBy( + SelectStatement statement, + ManagedIndexScanPlan plan, + IReadOnlyList orderBy) + { + if (statement.Source is not NamedTableSource source + || !ReferenceEquals(source, plan.Source) + || orderBy.Count == 0 + || orderBy.Count > plan.Index.Columns.Count + || plan.Index.IsMethodIndex) + { + return false; + } + + var table = plan.Table; + var qualifier = source.Alias ?? source.Name; + for (var position = 0; position < orderBy.Count; position++) + { + var term = orderBy[position]; + var indexTerm = plan.Index.Columns[position]; + if (term.NullPlacement != NullPlacement.Default + || indexTerm.NullPlacement != NullPlacement.Default + || indexTerm.IsExpression + || indexTerm.ColumnIndex < 0 + || term.Descending != (indexTerm.Descending != plan.Reverse)) + { + return false; + } + + var expression = term.Expression; + string? explicitCollation = null; + if (expression is CollationExpression collated) + { + explicitCollation = collated.Name; + expression = collated.Expression; + } + + if (expression is not ColumnExpression { BooleanKeyword: null } column) + return false; + + var name = column.UnqualifiedName ?? column.Name; + var separator = name.LastIndexOf('.'); + if (separator >= 0) + { + if (!string.Equals(name[..separator], qualifier, StringComparison.OrdinalIgnoreCase)) + return false; + name = name[(separator + 1)..]; + } + + if (!table.TryGetColumnIndex(name, out var columnIndex) || columnIndex != indexTerm.ColumnIndex) + return false; + + var termCollation = explicitCollation + ?? NormalizeDeclaredCollation(table.ColumnDefinitions[columnIndex].Collation) + ?? "BINARY"; + var indexCollation = IndexExpressionSemantics.GetCollationName(table, indexTerm) ?? "BINARY"; + if (!string.Equals(termCollation, indexCollation, StringComparison.OrdinalIgnoreCase) + || !IsBuiltInCollation(indexCollation) + || IsUnsafeCompiledCollation(indexCollation)) + { + return false; + } + } + + return true; + } + + /// + /// Whether an index equality seek may skip the table-row fetch: rows then carry only the + /// index's columns (and rowid), so this admits only statements that read nothing else. + /// Deliberately narrow (constant or COUNT(*) projections, a WHERE over plain index columns, + /// a non-partial index), which covers count and existence probes, the shapes ORMs issue most. + /// + private static bool CanSeekIndexCovering(SelectStatement select, ManagedIndexScanPlan plan) + { + if (select.GroupBy.Count != 0 + || select.Having is not null + || select.OrderBy.Count != 0 + || select.Distinct + || select.Where is null + || plan.Index.Where is not null + || plan.Index.Columns.Any(term => term.IsExpression || term.ColumnIndex < 0) + || !IndexCoversSelect(select, plan.Table, plan.Index)) + { + return false; + } + + foreach (var projection in select.Projections) + { + switch (projection.Expression) + { + case LiteralExpression: + case ParameterExpression: + case FunctionExpression { CountStar: true, Distinct: false, Filter: null, Window: null }: + continue; + default: + return false; + } + } + + var covered = plan.Index.Columns.Select(term => term.ColumnIndex).ToHashSet(); + return ReadsOnlyIndexedColumns(select.Where, plan.Table, covered); + } + + // Stricter than ExpressionCoveredByIndex: covering seek rows hold only the index's own + // column values, so a rowid or rowid-alias reference is not satisfiable here either. + private static bool ReadsOnlyIndexedColumns(Expression expression, EmbeddedTable table, HashSet covered) + { + switch (expression) + { + case LiteralExpression: + case ParameterExpression: + return true; + case ColumnExpression column: + var bare = column.UnqualifiedName ?? column.Name; + var name = bare.IndexOf('.') >= 0 ? bare[(bare.IndexOf('.') + 1)..] : bare; + return table.TryGetColumnIndex(name, out var columnIndex) + && covered.Contains(columnIndex) + && !(table.HasRowid && columnIndex == table.RowidAliasColumnIndex); + case BinaryExpression binary: + return ReadsOnlyIndexedColumns(binary.Left, table, covered) + && ReadsOnlyIndexedColumns(binary.Right, table, covered); + case UnaryExpression unary: + return ReadsOnlyIndexedColumns(unary.Operand, table, covered); + case CollationExpression collation: + return ReadsOnlyIndexedColumns(collation.Expression, table, covered); + case InExpression inList: + return ReadsOnlyIndexedColumns(inList.Value, table, covered) + && inList.Values.All(value => ReadsOnlyIndexedColumns(value, table, covered)); + case BetweenExpression between: + return ReadsOnlyIndexedColumns(between.Value, table, covered) + && ReadsOnlyIndexedColumns(between.Lower, table, covered) + && ReadsOnlyIndexedColumns(between.Upper, table, covered); + default: + return false; + } + } + /// /// True when every column the needs is available from /// keys (plus the rowid for rowid tables). Used to emit @@ -33650,7 +34016,9 @@ private SourceData GetManagedIndexRows( SourceRow? outerRow, long? maximumRows = null, Expression? predicate = null, - SqlValue[]? parameters = null) + SqlValue[]? parameters = null, + bool covering = false, + bool lazy = false) { var table = plan.Table; if (context.ConcurrentMvStore is { } store @@ -33667,6 +34035,49 @@ private SourceData GetManagedIndexRows( var qualifier = plan.Source.Alias ?? plan.Source.Name; var qualifiedColumns = BuildQualifiedColumns(qualifier, table.Columns); + + // A sorted order shared across statements (EmbeddedTable.GetOrCreateDerived) serves the + // scan from the loaded rows: equality and leading-column ranges binary-search it, which + // beats a durable seek per matching row once the table is in memory. A cold order is + // still built below, after the durable seek has had its chance. + var orderSignature = TryGetManagedIndexOrderSignature(table, plan.Index); + ManagedIndexOrder? warmOrder = null; + if (orderSignature is not null + && !table.HasPendingRowLoad + && table.RowIds.Count == table.Rows.Count) + { + table.TryGetDerived(new ManagedIndexOrderKey(orderSignature), out warmOrder); + + // A durable seek walks and re-validates b-tree pages on every statement, so a point + // lookup never built the order above. Once the unchanged table has served enough seeks + // on this index, build it: the count lives with the table's content, so any write + // restarts it and write-interleaved workloads keep seeking instead of rebuilding. Only + // the crossing attempts it, so a table whose visible rows cannot use the order (see + // TryGetManagedIndexOrder) is not re-checked on every later seek. + if (warmOrder is null + && CountDurableIndexSeek(table, orderSignature) == DurableSeeksBeforeWarmIndexOrder) + { + var loadedRows = GetNamedTableRows(plan.Source, context, maximumRows: null, outerRow).Rows; + warmOrder = TryGetManagedIndexOrder(table, plan.Index, orderSignature, loadedRows, context); + } + } + + if (warmOrder is not null) + { + context.RegisterMethodIndexSource(qualifier, ResolveMethodIndexSourceName(plan.Source.Name), table); + return GetCachedManagedIndexRows( + plan, + warmOrder, + context, + outerRow, + maximumRows, + predicate, + parameters, + lazy, + qualifier, + qualifiedColumns); + } + if (TrySeekManagedIndexEquality( plan, context, @@ -33675,6 +34086,7 @@ private SourceData GetManagedIndexRows( parameters, qualifier, qualifiedColumns, + covering, out var seekRows)) { IEnumerable sought = seekRows; @@ -33690,6 +34102,22 @@ private SourceData GetManagedIndexRows( context, maximumRows: null, outerRow).Rows; + if (orderSignature is not null + && TryGetManagedIndexOrder(table, plan.Index, orderSignature, visibleRows, context) is { } builtOrder) + { + return GetCachedManagedIndexRows( + plan, + builtOrder, + context, + outerRow, + maximumRows, + predicate, + parameters, + lazy, + qualifier, + qualifiedColumns); + } + var entries = GetManagedIndexEntries(table, plan.Index, visibleRows, context); // An outer-row equality is scan-constant for a correlated subquery. Narrow the // declared index traversal before aggregate evaluation instead of merely reporting a @@ -33724,16 +34152,44 @@ private SourceData GetManagedIndexRows( } } - var projected = candidates.Select(entry => entry.Row with + var qualifiedColumnDefinitions = BuildQualifiedColumnDefinitions( + qualifier, + table.ColumnDefinitions); + SourceRow Project((SourceRow Row, SqlValue[] Key) entry) => entry.Row with { QualifiedColumns = qualifiedColumns, Parent = outerRow, RowIdQualifier = qualifier, ColumnDefinitions = table.ColumnDefinitions, - QualifiedColumnDefinitions = BuildQualifiedColumnDefinitions( - qualifier, - table.ColumnDefinitions), - }); + QualifiedColumnDefinitions = qualifiedColumnDefinitions, + }; + + if (lazy) + { + // The caller consumes rows in order and may stop early (ORDER BY ... LIMIT over this + // index), so project on demand instead of materializing every entry first. + var ordered = candidates as IReadOnlyList<(SourceRow Row, SqlValue[] Key)> ?? candidates.ToList(); + IEnumerable EnumerateOrdered() + { + if (plan.Reverse) + { + for (var index = ordered.Count - 1; index >= 0; index--) + yield return Project(ordered[index]); + } + else + { + for (var index = 0; index < ordered.Count; index++) + yield return Project(ordered[index]); + } + } + + IEnumerable lazyRows = EnumerateOrdered(); + if (maximumRows is { } lazyMaximum) + lazyRows = lazyRows.Take(checked((int)Math.Min(lazyMaximum, int.MaxValue))); + return new SourceData(table.Columns, new LazyReadOnlyList(lazyRows)); + } + + var projected = candidates.Select(Project); if (plan.Reverse) projected = projected.Reverse(); if (maximumRows is { } maximum) @@ -33743,6 +34199,279 @@ private SourceData GetManagedIndexRows( return new SourceData(table.Columns, rows); } + /// + /// Serves an index scan from the shared sorted : the same rows, in the + /// same order, as projecting and filtering every index entry, but narrowed by binary search + /// to the run a leading-column equality or range can match. Only rows outside that run are + /// skipped, and only when the comparison is conversion-free under the index collation; the + /// caller still applies the whole WHERE. + /// + private SourceData GetCachedManagedIndexRows( + ManagedIndexScanPlan plan, + ManagedIndexOrder order, + QueryContext context, + SourceRow? outerRow, + long? maximumRows, + Expression? predicate, + SqlValue[]? parameters, + bool lazy, + string qualifier, + IReadOnlyDictionary qualifiedColumns) + { + var table = plan.Table; + var keys = order.Keys; + var lower = 0; + var upper = keys.Length; + var leading = plan.Index.Columns[0]; + var leadingCollation = IndexExpressionSemantics.GetCollationName(table, leading) ?? "BINARY"; + // Ascending with default NULL placement: NULLs first, then values by Compare. + var ascending = !leading.Descending && leading.NullPlacement == NullPlacement.Default; + SqlValue? equalityFilter = null; + + int FirstNonNull(int from, int to) + { + while (from < to) + { + var middle = (from + to) >>> 1; + if (keys[middle][0].Kind == SqlValueKind.Null) + from = middle + 1; + else + to = middle; + } + + return from; + } + + // First position in [from, to) whose key is above the probe (strict) or at least it. + int LowerBound(int from, int to, SqlValue probe, bool strict) + { + while (from < to) + { + var middle = (from + to) >>> 1; + var comparison = Compare(keys[middle][0], probe, leadingCollation); + if (strict ? comparison <= 0 : comparison < 0) + from = middle + 1; + else + to = middle; + } + + return from; + } + + if (plan.Search + && predicate is not null + && parameters is not null + && TryCreateTransientEqualityLookup( + plan.Source, + table, + predicate, + context, + outerRow, + out var lookup) + && leading.ColumnIndex == lookup.ColumnOrdinal + && string.Equals(leadingCollation, lookup.Collation, StringComparison.OrdinalIgnoreCase)) + { + var probeValue = ConvertTransientProbe( + lookup, + Evaluate(lookup.ValueExpression, parameters, outerRow, context)); + if (probeValue.Kind is SqlValueKind.Null) + { + upper = lower; + } + else if (!lookup.ColumnConvertsTextToNumeric && !lookup.ColumnConvertsNumericToText) + { + if (ascending) + { + lower = LowerBound(FirstNonNull(lower, upper), upper, probeValue, strict: false); + upper = LowerBound(lower, upper, probeValue, strict: true); + } + else + { + equalityFilter = probeValue; + } + } + } + + if (ascending && predicate is not null && parameters is not null && lower < upper) + { + var outputColumns = GetOutputColumns(plan.Source, context); + foreach (var (op, bound) in CollectLeadingIndexRangeBounds( + plan, table, leadingCollation, predicate, parameters, outerRow, context, outputColumns)) + { + if (bound.Kind == SqlValueKind.Null) + { + // A comparison with NULL is never true. + upper = lower; + break; + } + + lower = Math.Max(lower, FirstNonNull(lower, upper)); + switch (op) + { + case BinaryOperator.GreaterThan: + lower = LowerBound(lower, upper, bound, strict: true); + break; + case BinaryOperator.GreaterThanOrEqual: + lower = LowerBound(lower, upper, bound, strict: false); + break; + case BinaryOperator.LessThan: + upper = LowerBound(lower, upper, bound, strict: false); + break; + default: + upper = LowerBound(lower, upper, bound, strict: true); + break; + } + + if (lower >= upper) + break; + } + } + + var scanOrder = table.GetRowidScanOrder(); + var qualifiedColumnDefinitions = BuildQualifiedColumnDefinitions(qualifier, table.ColumnDefinitions); + SourceRow Project(int index) + { + var position = scanOrder[order.Permutation[index]]; + return new SourceRow( + table.Columns, + table.Rows[position], + qualifiedColumns, + outerRow, + RowId: table.RowIds[position], + RowIdQualifier: qualifier, + ColumnDefinitions: table.ColumnDefinitions, + QualifiedColumnDefinitions: qualifiedColumnDefinitions); + } + + var first = lower; + var last = upper; + IEnumerable Enumerate() + { + if (plan.Reverse) + { + for (var index = last - 1; index >= first; index--) + { + if (equalityFilter is not { } probe || Compare(keys[index][0], probe, leadingCollation) == 0) + yield return Project(index); + } + } + else + { + for (var index = first; index < last; index++) + { + if (equalityFilter is not { } probe || Compare(keys[index][0], probe, leadingCollation) == 0) + yield return Project(index); + } + } + } + + IEnumerable rows = Enumerate(); + if (maximumRows is { } maximum) + rows = rows.Take(checked((int)Math.Min(maximum, int.MaxValue))); + return lazy + ? new SourceData(table.Columns, new LazyReadOnlyList(rows)) + : new SourceData(table.Columns, rows.ToArray()); + } + + // The comparison affinity SQLite applies to the value side of a transient lookup, applied to an + // evaluated probe (the same conversion TrySeekManagedIndexEquality makes before seeking). + private static SqlValue ConvertTransientProbe(TransientEqualityLookup lookup, SqlValue probe) + { + if (lookup.ValueConvertsTextToNumeric) + return ApplyComparisonNumericAffinity(probe); + if (lookup.ValueConvertsNumericToText && probe.Kind is SqlValueKind.Integer or SqlValueKind.Real) + return SqlValue.Text(ToSqlText(probe)); + return probe; + } + + // Bounds a WHERE puts on the index's leading column: `col op value` (either operand order) and + // `col BETWEEN low AND high`, with a literal or parameter value whose comparison needs no + // affinity conversion and uses the index collation, so the bound is exact against the keys. + private IEnumerable<(BinaryOperator Operator, SqlValue Bound)> CollectLeadingIndexRangeBounds( + ManagedIndexScanPlan plan, + EmbeddedTable table, + string leadingCollation, + Expression predicate, + SqlValue[] parameters, + SourceRow? outerRow, + QueryContext context, + IReadOnlyList outputColumns) + { + var leadingColumn = plan.Index.Columns[0].ColumnIndex; + bool TryBound(Expression column, Expression value, bool columnOnLeft, out SqlValue bound) + { + bound = SqlValue.Null; + var operand = value is UnaryExpression { Operator: UnaryOperator.Negate or UnaryOperator.Plus } unary + ? unary.Operand + : value; + if (operand is not (LiteralExpression or ParameterExpression) + || !TryMatchTransientEquality( + column, + value, + columnOnLeft, + plan.Source, + table, + outputColumns, + outerRow, + out var lookup) + || lookup.ColumnOrdinal != leadingColumn + || lookup.ColumnConvertsTextToNumeric + || lookup.ColumnConvertsNumericToText + || !string.Equals(lookup.Collation, leadingCollation, StringComparison.OrdinalIgnoreCase)) + { + return false; + } + + // Only the value side converts, so the converted bound compares against the stored keys + // exactly as the evaluator's comparison would. + bound = ConvertTransientProbe(lookup, Evaluate(value, parameters, outerRow, context)); + return true; + } + + var pending = new Stack(); + pending.Push(predicate); + while (pending.Count > 0) + { + var conjunct = pending.Pop(); + if (conjunct is BinaryExpression { Operator: BinaryOperator.And } and) + { + pending.Push(and.Right); + pending.Push(and.Left); + continue; + } + + if (conjunct is BinaryExpression + { + Operator: BinaryOperator.LessThan or BinaryOperator.LessThanOrEqual + or BinaryOperator.GreaterThan or BinaryOperator.GreaterThanOrEqual, + } comparison) + { + if (TryBound(comparison.Left, comparison.Right, columnOnLeft: true, out var bound)) + { + yield return (comparison.Operator, bound); + } + else if (TryBound(comparison.Right, comparison.Left, columnOnLeft: false, out bound)) + { + // `value op col` bounds the column from the other side. + yield return (comparison.Operator switch + { + BinaryOperator.LessThan => BinaryOperator.GreaterThan, + BinaryOperator.LessThanOrEqual => BinaryOperator.GreaterThanOrEqual, + BinaryOperator.GreaterThan => BinaryOperator.LessThan, + _ => BinaryOperator.LessThanOrEqual, + }, bound); + } + } + else if (conjunct is BetweenExpression { Negated: false } between + && TryBound(between.Value, between.Lower, columnOnLeft: true, out var low) + && TryBound(between.Value, between.Upper, columnOnLeft: true, out var high)) + { + yield return (BinaryOperator.GreaterThanOrEqual, low); + yield return (BinaryOperator.LessThanOrEqual, high); + } + } + } + /// /// Serves a SEARCH whose leading index column is bound by equality to a scan-constant value /// (a literal, a parameter, or an outer-row column) from the durable index b-tree, instead of @@ -33764,6 +34493,7 @@ private bool TrySeekManagedIndexEquality( SqlValue[]? parameters, string qualifier, IReadOnlyDictionary qualifiedColumns, + bool covering, out SourceRow[] rows) { rows = []; @@ -33813,7 +34543,7 @@ private bool TrySeekManagedIndexEquality( table, plan.Index, prefixLength: 1, - covering: false, + covering, context, out var transactionAccessor)) { @@ -33824,7 +34554,7 @@ private bool TrySeekManagedIndexEquality( table, plan.Index, prefixLength: 1, - covering: false, + covering, sharedSnapshot: null, out var committedAccessor)) { @@ -34102,6 +34832,135 @@ private int CompareMvccIndexRows( EmbeddedIndex index, IReadOnlyList visibleRows, QueryContext context) + { + if (TryGetCachedManagedIndexEntries(table, index, visibleRows, context) is { } cachedEntries) + return cachedEntries; + + return BuildManagedIndexEntries(table, index, visibleRows, context); + } + + // The sorted index order over a table's full rowid-ordered scan is a pure function of its rows + // for a plain index under built-in collations, so it is shared across statements (and clones) + // through EmbeddedTable.GetOrCreateDerived instead of re-projecting and re-sorting every entry + // on each execution. The cached permutation indexes the scan order, and is applied only after + // proving the caller's visible rows are exactly that scan. + private List<(SourceRow Row, SqlValue[] Key)>? TryGetCachedManagedIndexEntries( + EmbeddedTable table, + EmbeddedIndex index, + IReadOnlyList visibleRows, + QueryContext context) + { + if (TryGetManagedIndexOrderSignature(table, index) is not { } signature + || TryGetManagedIndexOrder(table, index, signature, visibleRows, context) is not { } order) + { + return null; + } + + var entries = new List<(SourceRow Row, SqlValue[] Key)>(order.Permutation.Length); + for (var position = 0; position < order.Permutation.Length; position++) + entries.Add((visibleRows[order.Permutation[position]], order.Keys[position])); + return entries; + } + + // Null when the index order cannot be shared: partial, method or expression indexes, custom or + // overridden collations, and WITHOUT ROWID tables (whose scan is not the rowid order). + private string? TryGetManagedIndexOrderSignature(EmbeddedTable table, EmbeddedIndex index) + { + if (index.IsPartial + || index.IsMethodIndex + || !table.HasRowid + || table.HasMethodIndexes + || index.Columns.Count == 0) + { + return null; + } + + var signature = new StringBuilder(index.Name).Append('|'); + foreach (var term in index.Columns) + { + var collation = IndexExpressionSemantics.GetCollationName(table, term); + if (term.IsExpression + || term.ColumnIndex < 0 + || !IsBuiltInCollation(collation) + || IsUnsafeCompiledCollation(collation)) + { + return null; + } + + signature.Append(term.ColumnIndex).Append(',') + .Append(collation?.ToUpperInvariant()).Append(',') + .Append(term.Descending ? 'D' : 'A').Append(',') + .Append((int)term.NullPlacement).Append(';'); + } + + return signature.ToString(); + } + + // The shared sorted order, built from the caller's visible rows only after proving they are + // exactly the table's rowid scan order (the permutation indexes that order). + private ManagedIndexOrder? TryGetManagedIndexOrder( + EmbeddedTable table, + EmbeddedIndex index, + string signature, + IReadOnlyList visibleRows, + QueryContext context) + { + if (visibleRows.Count != table.Rows.Count || table.RowIds.Count != table.Rows.Count) + return null; + + var scanOrder = table.GetRowidScanOrder(); + for (var position = 0; position < visibleRows.Count; position++) + { + if (!ReferenceEquals(visibleRows[position].Values, table.Rows[scanOrder[position]])) + return null; + } + + return table.GetOrCreateDerived( + new ManagedIndexOrderKey(signature), + () => + { + var built = BuildManagedIndexEntries(table, index, visibleRows, context); + var positions = new Dictionary(visibleRows.Count, ReferenceEqualityComparer.Instance); + for (var position = 0; position < visibleRows.Count; position++) + positions[visibleRows[position]] = position; + var permutation = new int[built.Count]; + var keys = new SqlValue[built.Count][]; + for (var position = 0; position < built.Count; position++) + { + permutation[position] = positions[built[position].Row]; + keys[position] = built[position].Key; + } + + return new ManagedIndexOrder(permutation, keys); + }); + } + + private readonly record struct ManagedIndexOrderKey(string Signature); + + private const int DurableSeeksBeforeWarmIndexOrder = 32; + + private readonly record struct DurableIndexSeekCountKey(string Signature); + + private sealed class DurableIndexSeekCount + { + public int Value; + } + + private static int CountDurableIndexSeek(EmbeddedTable table, string orderSignature) + { + var count = table.GetOrCreateDerived( + new DurableIndexSeekCountKey(orderSignature), + static () => new DurableIndexSeekCount()); + return Interlocked.Increment(ref count.Value); + } + + private sealed record ManagedIndexOrder(int[] Permutation, SqlValue[][] Keys); + + private List<(SourceRow Row, SqlValue[] Key)> BuildManagedIndexEntries( + EmbeddedTable table, + EmbeddedIndex index, + IReadOnlyList visibleRows, + QueryContext context) { var entries = new List<(SourceRow Row, SqlValue[] Key)>(visibleRows.Count); foreach (var row in visibleRows) @@ -34613,6 +35472,29 @@ private ExecutionResult ExecuteSelect( && !(context.ConcurrentMvStore is not null && indexPlan is not null && limit is >= 0); + // An index scan whose key order is the ORDER BY needs no sort, and with a LIMIT the scan + // can stop at the last row it keeps (SQLite's ORDER BY ... LIMIT over an index). + var indexPlanOrdersRows = indexPlan is not null + && statement.OrderBy.Count > 0 + && context.ConcurrentMvStore is null + && IndexPlanSatisfiesOrderBy(statement, indexPlan, resolvedOrderBy); + // Likewise a plain unfiltered scan already walks the table in ascending rowid order, so + // ORDER BY rowid (the INTEGER PRIMARY KEY paging shape) reads only offset + limit rows. + var rowidScanOrdersRows = intersectionPlan is null + && indexPlan is null + && orUnionPlan is null + && statement.Where is null + && context.ConcurrentMvStore is null + && IsAscendingRowidOrderScan(statement, resolvedOrderBy, context); + if (rowidScanOrdersRows + && !hasAggregate + && !hasWindow + && statement.GroupBy.Count == 0 + && !statement.Distinct + && limit is >= 0) + { + sourceLimit = limit.Value > long.MaxValue - offset ? null : offset + limit.Value; + } // A managed index plan wins over the transient probe: the index path defines // row order (SQLite parity, INDEXED BY), while the probe only prunes a plain scan. // Top-level OR equality branches can each SEARCH a different index and union positions. @@ -34629,7 +35511,10 @@ private ExecutionResult ExecuteSelect( outerRow, sourceLimit, statement.Where, - parameters) + parameters, + lazy: indexPlanOrdersRows && limit is >= 0) is var indexed && indexPlanOrdersRows + ? indexed with { OrderByConsumed = true } + : indexed : orUnionPlan is not null ? GetManagedOrIndexUnionRows(orUnionPlan, parameters, context, outerRow) : TryGetTransientLookupRows( @@ -34663,19 +35548,38 @@ private ExecutionResult ExecuteSelect( // can prove a method may return just the rows its pushed-down LIMIT keeps. AllowsMethodIndexRowTruncation(statement), statement.Projections.Select(static projection => projection.Expression).ToArray())); + if (rowidScanOrdersRows) + source = source with { OrderByConsumed = true }; + + // The rowid-ordered scan is already the output order and has no WHERE, so the OFFSET is + // just a starting position: start there rather than build and discard the skipped rows. + IEnumerable scannedRows = source.Rows; + if (rowidScanOrdersRows + && offset > 0 + && !hasAggregate + && !hasWindow + && statement.GroupBy.Count == 0 + && !statement.Distinct + && limit is >= 0 + && source.Rows is IndexedSourceRowList indexedRows) + { + scannedRows = indexedRows.EnumerateFrom((int)Math.Min(offset, indexedRows.Count)); + offset = 0; + } + var selectedRows = new List(); var selectedRowLimit = !streamProjectionRows && !hasAggregate && !hasWindow && statement.GroupBy.Count == 0 - && statement.OrderBy.Count == 0 + && (statement.OrderBy.Count == 0 || source.OrderByConsumed) && !statement.Distinct && limit is >= 0 ? limit.Value > long.MaxValue - offset ? long.MaxValue : offset + limit.Value : (long?)null; - foreach (var row in source.Rows) + foreach (var row in scannedRows) { context.CheckInterrupt(); if (streamProjectionRows @@ -39168,82 +40072,119 @@ private SourceData GetSourceRowsWithTransientIndex( { if (source is not NamedTableSource named || predicate is null - || context.StatementState is not { } statementState + || context.StatementState is null || IsSchemaTable(named.Name) || IsCommonTableExpression(named, context) || context.Views?.ContainsKey(named.Name) == true - || !context.Tables.TryGetValue(named.Name, out var table) - || !TryCreateTransientEqualityLookup( + || !context.Tables.TryGetValue(named.Name, out var table)) + { + return null; + } + + IReadOnlyList? inListValues = null; + if (!TryCreateTransientEqualityLookup( named, table, predicate, context, outerRow, out var lookup, + preserveErrors) + && !TryCreateTransientInListLookup( + named, + table, + predicate, + context, + outerRow, + out lookup, + out inListValues, preserveErrors)) { - return null; - } - - var probeValue = Evaluate(lookup.ValueExpression, parameters, outerRow, context); - if (probeValue.Kind is SqlValueKind.Null) - { - // SQL equality against NULL never holds, so no row can match. - return new SourceData(table.Columns, []); + return TryGetRowidRangeRows(named, table, predicate, parameters, context, outerRow, preserveErrors); } // Both sides are canonicalized with exactly the conversion SQLite's comparison affinity // rules apply to that side, so the bucket a value lands in is the bucket the evaluator's // `=` would agree with. Applying the scanned column's affinity to the probe instead // would silently answer "no row" for `INTEGER 7 = TEXT '007'`. - var key = EquiJoinHashIndex.CanonicalizeJoinKeyValue( - probeValue, - lookup.ValueConvertsTextToNumeric, - lookup.ValueConvertsNumericToText, - lookup.Collation); - if (key is null) - return new SourceData(table.Columns, []); - - var lookups = statementState.TransientLookups; - var cacheKey = new TransientLookupKey( - table, - lookup.ColumnOrdinal, - lookup.Collation, - lookup.ColumnConvertsTextToNumeric, - lookup.ColumnConvertsNumericToText); - if (!lookups.TryGetValue(cacheKey, out var transient)) + string? CanonicalizeProbe(Expression valueExpression) { - transient = new TransientLookup(); - lookups[cacheKey] = transient; + var probeValue = Evaluate(valueExpression, parameters, outerRow, context); + // SQL equality against NULL never holds, so no row can match. + return probeValue.Kind is SqlValueKind.Null + ? null + : EquiJoinHashIndex.CanonicalizeJoinKeyValue( + probeValue, + lookup.ValueConvertsTextToNumeric, + lookup.ValueConvertsNumericToText, + lookup.Collation); } - if (transient.Revision != table.Rows.Revision) - { - transient.Buckets.Clear(); - for (var position = 0; position < table.Rows.Count; position++) + var key = inListValues is null ? CanonicalizeProbe(lookup.ValueExpression) : null; + if (inListValues is null && key is null) + return new SourceData(table.Columns, []); + + // The bucket map is derived from the table's rows alone, so it is shared by every clone + // holding the same rows (TableDerivedCache): a parameterized point query no longer + // re-hashes the whole table on every statement. + var columnOrdinal = lookup.ColumnOrdinal; + var columnConvertsTextToNumeric = lookup.ColumnConvertsTextToNumeric; + var columnConvertsNumericToText = lookup.ColumnConvertsNumericToText; + var collation = lookup.Collation; + var buckets = table.GetOrCreateDerived( + new TransientLookupKey( + columnOrdinal, + collation, + columnConvertsTextToNumeric, + columnConvertsNumericToText), + () => { - context.CheckInterrupt(); - var segment = EquiJoinHashIndex.CanonicalizeJoinKeyValue( - table.Rows[position][lookup.ColumnOrdinal], - lookup.ColumnConvertsTextToNumeric, - lookup.ColumnConvertsNumericToText, - lookup.Collation); - if (segment is null) - continue; - if (!transient.Buckets.TryGetValue(segment, out var bucket)) + var built = new Dictionary>(StringComparer.Ordinal); + var rows = table.Rows; + for (var position = 0; position < rows.Count; position++) { - bucket = []; - transient.Buckets[segment] = bucket; + context.CheckInterrupt(); + var segment = EquiJoinHashIndex.CanonicalizeJoinKeyValue( + rows[position][columnOrdinal], + columnConvertsTextToNumeric, + columnConvertsNumericToText, + collation); + if (segment is null) + continue; + if (!built.TryGetValue(segment, out var bucket)) + { + bucket = []; + built[segment] = bucket; + } + + bucket.Add(position); } - bucket.Add(position); - } + return built; + }); - transient.Revision = table.Rows.Revision; + List? positions; + if (inListValues is null) + { + if (!buckets.TryGetValue(key!, out positions) || positions.Count == 0) + return new SourceData(table.Columns, []); } + else + { + // `col IN (v1, v2, ...)` is `col = v1 OR col = v2 ...`: the union of the buckets, + // in table order like a single probe, with duplicate values contributing once. + var union = new HashSet(); + foreach (var value in inListValues) + { + if (CanonicalizeProbe(value) is { } valueKey && buckets.TryGetValue(valueKey, out var bucket)) + union.UnionWith(bucket); + } - if (!transient.Buckets.TryGetValue(key, out var positions) || positions.Count == 0) - return new SourceData(table.Columns, []); + positions = [.. union]; + positions.Sort(); + if (positions.Count == 0) + return new SourceData(table.Columns, []); + } var qualifier = named.Alias ?? named.Name; var qualifiedColumns = BuildQualifiedColumns(qualifier, table.Columns); @@ -39494,6 +40435,270 @@ private bool TryCreateTransientEqualityLookup( return false; } + // A range over the INTEGER PRIMARY KEY (`id < ?`, `id >= ?`, `id BETWEEN ? AND ?`) selects a + // contiguous run of the table's rowid order, as SQLite's rowid-range SEARCH does: binary-search + // the shared rowid scan order instead of evaluating the predicate on every row. The run is a + // superset of the matches only when the bound compares exactly, so the probe value must be an + // INTEGER, or a REAL small enough to compare with every rowid without rounding. Rows come back + // in rowid order; the caller still applies the whole WHERE. + private SourceData? TryGetRowidRangeRows( + NamedTableSource source, + EmbeddedTable table, + Expression predicate, + SqlValue[] parameters, + QueryContext context, + SourceRow? outerRow, + bool preserveErrors) + { + if (!table.HasRowid + || table.RowidAliasColumnIndex < 0 + || table.HasMethodIndexes + || table.RowIds.Count != table.Rows.Count + || context.ConcurrentMvStore is not null) + { + return null; + } + + var outputColumns = GetOutputColumns(source, context); + bool IsRowidColumn(Expression expression) + => expression is ColumnExpression { BooleanKeyword: null } column + && ResolveJoinSideColumn(column, outputColumns) is not null + && table.TryGetColumnIndex(column.UnqualifiedName ?? column.Name, out var ordinal) + && ordinal == table.RowidAliasColumnIndex; + + bool TryEvaluateBound(Expression expression, out double bound) + { + bound = 0; + var operand = expression is UnaryExpression { Operator: UnaryOperator.Negate or UnaryOperator.Plus } unary + ? unary.Operand + : expression; + if (operand is not (LiteralExpression or ParameterExpression)) + return false; + var value = Evaluate(expression, parameters, outerRow, context); + const double ExactLimit = 9007199254740992d; // 2^53 + switch (value.Kind) + { + case SqlValueKind.Integer when Math.Abs((double)value.AsInteger()) < ExactLimit: + bound = value.AsInteger(); + return true; + case SqlValueKind.Real when !double.IsNaN(value.AsReal()) && Math.Abs(value.AsReal()) < ExactLimit: + bound = value.AsReal(); + return true; + default: + return false; + } + } + + double? lower = null; + double? upper = null; + bool lowerInclusive = true, upperInclusive = true; + void TightenLower(double value, bool inclusive) + { + if (lower is null || value > lower || (value == lower && !inclusive)) + { + lower = value; + lowerInclusive = inclusive; + } + } + + void TightenUpper(double value, bool inclusive) + { + if (upper is null || value < upper || (value == upper && !inclusive)) + { + upper = value; + upperInclusive = inclusive; + } + } + + var pending = new Stack(); + pending.Push(predicate); + var found = false; + while (pending.Count > 0) + { + var conjunct = pending.Pop(); + if (conjunct is BinaryExpression { Operator: BinaryOperator.And } and) + { + pending.Push(and.Right); + pending.Push(and.Left); + continue; + } + + if (conjunct is BinaryExpression + { + Operator: BinaryOperator.LessThan or BinaryOperator.LessThanOrEqual + or BinaryOperator.GreaterThan or BinaryOperator.GreaterThanOrEqual, + } comparison) + { + var columnOnLeft = IsRowidColumn(comparison.Left); + var columnOnRight = !columnOnLeft && IsRowidColumn(comparison.Right); + if ((columnOnLeft || columnOnRight) + && TryEvaluateBound(columnOnLeft ? comparison.Right : comparison.Left, out var bound)) + { + // Normalize to `rowid op bound`. + var op = comparison.Operator; + if (columnOnRight) + { + op = op switch + { + BinaryOperator.LessThan => BinaryOperator.GreaterThan, + BinaryOperator.LessThanOrEqual => BinaryOperator.GreaterThanOrEqual, + BinaryOperator.GreaterThan => BinaryOperator.LessThan, + _ => BinaryOperator.LessThanOrEqual, + }; + } + + switch (op) + { + case BinaryOperator.LessThan: TightenUpper(bound, inclusive: false); break; + case BinaryOperator.LessThanOrEqual: TightenUpper(bound, inclusive: true); break; + case BinaryOperator.GreaterThan: TightenLower(bound, inclusive: false); break; + default: TightenLower(bound, inclusive: true); break; + } + + found = true; + continue; + } + } + else if (conjunct is BetweenExpression { Negated: false } between + && IsRowidColumn(between.Value) + && TryEvaluateBound(between.Lower, out var betweenLower) + && TryEvaluateBound(between.Upper, out var betweenUpper)) + { + TightenLower(betweenLower, inclusive: true); + TightenUpper(betweenUpper, inclusive: true); + found = true; + continue; + } + + if (preserveErrors && ExpressionCanFail(conjunct)) + break; + } + + if (!found) + return null; + + var order = table.GetRowidScanOrder(); + var rowIds = table.RowIds; + bool AboveLower(int orderIndex) + => lower is not { } bound || (lowerInclusive ? rowIds[order[orderIndex]] >= bound : rowIds[order[orderIndex]] > bound); + bool BelowUpper(int orderIndex) + => upper is not { } bound || (upperInclusive ? rowIds[order[orderIndex]] <= bound : rowIds[order[orderIndex]] < bound); + + // First index at or above the lower bound, and first index past the upper bound. + int low = 0, high = order.Length; + while (low < high) + { + var middle = (low + high) >>> 1; + if (AboveLower(middle)) + high = middle; + else + low = middle + 1; + } + + var start = low; + high = order.Length; + while (low < high) + { + var middle = (low + high) >>> 1; + if (BelowUpper(middle)) + low = middle + 1; + else + high = middle; + } + + var end = low; + var qualifier = source.Alias ?? source.Name; + var qualifiedColumns = BuildQualifiedColumns(qualifier, table.Columns); + var qualifiedColumnDefinitions = BuildQualifiedColumnDefinitions(qualifier, table.ColumnDefinitions); + var rows = new SourceRow[Math.Max(0, end - start)]; + for (var index = start; index < end; index++) + { + var position = order[index]; + rows[index - start] = new SourceRow( + table.Columns, + table.Rows[position], + qualifiedColumns, + outerRow, + RowId: rowIds[position], + RowIdQualifier: qualifier, + ColumnDefinitions: table.ColumnDefinitions, + QualifiedColumnDefinitions: qualifiedColumnDefinitions); + } + + return new SourceData(table.Columns, rows); + } + + // The IN-list counterpart of TryCreateTransientEqualityLookup: the first conjunct shaped + // `column IN (literal or parameter, ...)` over the scanned table. Its values carry no affinity + // or collation (SQLite treats `a IN (x, y)` as `a = +x OR a = +y`), which literals and + // parameters already lack, so each value probes the same equality lookup as `a = value`. + private bool TryCreateTransientInListLookup( + NamedTableSource source, + EmbeddedTable table, + Expression predicate, + QueryContext context, + SourceRow? outerRow, + [System.Diagnostics.CodeAnalysis.NotNullWhen(true)] out TransientEqualityLookup? lookup, + [System.Diagnostics.CodeAnalysis.NotNullWhen(true)] out IReadOnlyList? values, + bool preserveErrors = false) + { + var outputColumns = GetOutputColumns(source, context); + var pending = new Stack(); + pending.Push(predicate); + while (pending.Count > 0) + { + var conjunct = pending.Pop(); + if (conjunct is BinaryExpression { Operator: BinaryOperator.And } and) + { + pending.Push(and.Right); + pending.Push(and.Left); + continue; + } + + if (conjunct is InExpression { Negated: false, Values.Count: > 0 } inList + && inList.Values.All(value => value is LiteralExpression or ParameterExpression) + && TryMatchTransientEquality( + inList.Value, + inList.Values[0], + columnSideIsLeftOperand: true, + source, + table, + outputColumns, + outerRow, + out var first)) + { + var consistent = true; + for (var index = 1; index < inList.Values.Count && consistent; index++) + { + consistent = TryMatchTransientEquality( + inList.Value, + inList.Values[index], + columnSideIsLeftOperand: true, + source, + table, + outputColumns, + outerRow, + out var other) + && other with { ValueExpression = first.ValueExpression } == first; + } + + if (consistent) + { + lookup = first; + values = inList.Values; + return true; + } + } + + if (preserveErrors && ExpressionCanFail(conjunct)) + break; + } + + lookup = null; + values = null; + return false; + } + private bool TryMatchTransientEquality( Expression columnSide, Expression valueSide, @@ -39766,6 +40971,7 @@ private SourceData GetSideSourceRows( ? TryGetManagedJoinSideIndexRows( named, sidePredicate, + parameters, context, outerRow, sourceOrderBy) @@ -39801,6 +41007,7 @@ private SourceData GetSideSourceRows( private SourceData? TryGetManagedJoinSideIndexRows( NamedTableSource source, Expression predicate, + SqlValue[] parameters, QueryContext context, SourceRow? outerRow, IReadOnlyList? sourceOrderBy) @@ -39816,8 +41023,10 @@ private SourceData GetSideSourceRows( OrderBy: sourceOrderBy ?? [], Limit: null, Offset: null); + // The side predicate and parameters let a SEARCH plan seek the durable index instead of + // materializing and sorting every entry of it; the caller still filters the result. return TryPlanManagedIndexScan(select, context) is { } plan - ? GetManagedIndexRows(plan, context, outerRow) + ? GetManagedIndexRows(plan, context, outerRow, predicate: predicate, parameters: parameters) : null; } @@ -40278,9 +41487,32 @@ private SourceData GetJoinRows( && TryPlanDeclaredIndexLookup(rightNamedSource, source.Condition, context) is { } declaredLookup ? declaredLookup : null; + // A small left side probes a base-table right side per row through the shared equality + // lookup (TryGetTransientLookupRows), as SQLite's index nested loop does, instead of + // materializing every right row and hashing them all for a handful of probes. The probe + // only prunes: the full join condition and right predicate still run on each candidate. + var rightProbesPerLeftRow = !rightIsCorrelatedSource + && rightDeclaredLookup is null + && source.Kind is JoinKind.Inner or JoinKind.Left + && source.Condition is not null + && (sourceOrderBy is null || sourceOrderBy.Count == 0) + && left.Rows.Count > 0 + && source.Right is NamedTableSource rightProbeSource + && !IsSchemaTable(rightProbeSource.Name) + && !IsCommonTableExpression(rightProbeSource, context) + && context.Views?.ContainsKey(rightProbeSource.Name) != true + && !TryGetVirtualTable(context, rightProbeSource, out _) + && context.Tables.TryGetValue(rightProbeSource.Name, out var rightProbeTable) + && left.Rows.Count <= Math.Max(SmallJoinProbeRows, rightProbeTable.Rows.Count / 8) + && TryGetTransientLookupRows( + source.Right, + source.Condition, + parameters, + context, + left.Rows[0] with { Parent = outerRow }) is not null; var right = rightIsCorrelatedSource ? new SourceData(GetSourceColumns(source.Right, context), []) - : rightDeclaredLookup is not null + : rightDeclaredLookup is not null || rightProbesPerLeftRow ? new SourceData(GetSourceColumns(source.Right, context), []) : GetSideSourceRows( source.Right, @@ -40302,7 +41534,7 @@ private SourceData GetJoinRows( var ambiguousQualifiedColumns = GetAmbiguousQualifiedColumns(source, context); var leftWidth = left.Columns.Length; var joinPairs = BuildJoinPairs(source, context); - var joinHashIndex = rightIsCorrelatedSource + var joinHashIndex = rightIsCorrelatedSource || rightProbesPerLeftRow ? null : rightDeclaredLookup is null ? TryBuildJoinHashIndex(source, right, parameters, context) @@ -40334,30 +41566,63 @@ SourceData CreateResult(IReadOnlyList result) result, OmittedVirtualTablePredicates: omittedPredicates); + // Rows drawn from the same pair of sources share their qualified-column map and column + // metadata, so the combined forms are reused while those inputs are the same instances + // instead of rebuilding a dictionary and an array for every joined row. + IReadOnlyDictionary? combinedColumnsLeft = null; + IReadOnlyDictionary? combinedColumnsRight = null; + IReadOnlyDictionary? combinedColumns = null; + IReadOnlyList? combinedDefinitionsLeft = null; + IReadOnlyList? combinedDefinitionsRight = null; + IReadOnlyList? combinedDefinitions = null; + var hasCombinedDefinitions = false; SourceRow CreateJoinedRow(SourceRow leftRow, SourceRow rightRow) - => new( - columns, - leftRow.Values.Concat(rightRow.Values).ToArray(), - CombineQualifiedColumns(leftRow.QualifiedColumns, rightRow.QualifiedColumns, leftWidth), - outerRow, - outputColumns, - QualifiedRowIds: CombineQualifiedRowIds( - GetQualifiedRowIds(leftRow), - GetQualifiedRowIds(rightRow)), - ColumnDefinitions: CombineColumnDefinitions( + { + if (combinedColumns is null + || !ReferenceEquals(combinedColumnsLeft, leftRow.QualifiedColumns) + || !ReferenceEquals(combinedColumnsRight, rightRow.QualifiedColumns)) + { + combinedColumnsLeft = leftRow.QualifiedColumns; + combinedColumnsRight = rightRow.QualifiedColumns; + combinedColumns = CombineQualifiedColumns(leftRow.QualifiedColumns, rightRow.QualifiedColumns, leftWidth); + } + + if (!hasCombinedDefinitions + || !ReferenceEquals(combinedDefinitionsLeft, leftRow.ColumnDefinitions) + || !ReferenceEquals(combinedDefinitionsRight, rightRow.ColumnDefinitions)) + { + combinedDefinitionsLeft = leftRow.ColumnDefinitions; + combinedDefinitionsRight = rightRow.ColumnDefinitions; + combinedDefinitions = CombineColumnDefinitions( leftRow, rightRow, left.Columns.Length, right.Columns.Length, - columnDefinitions), + columnDefinitions); + hasCombinedDefinitions = true; + } + + return new( + columns, + ConcatRowValues(leftRow.Values, rightRow.Values), + combinedColumns, + outerRow, + outputColumns, + QualifiedRowIds: CombineQualifiedRowIds(leftRow, rightRow), + ColumnDefinitions: combinedDefinitions, QualifiedColumnDefinitions: qualifiedColumnDefinitions, AmbiguousQualifiedColumns: ambiguousQualifiedColumns, - QualifiedMethodIndexSources: CombineMethodIndexSources( - GetMethodIndexSources(leftRow), - GetMethodIndexSources(rightRow)), - QualifiedFts5Sources: CombineFts5Sources( - GetFts5Sources(leftRow), - GetFts5Sources(rightRow))); + QualifiedMethodIndexSources: HasNoMethodIndexSources(leftRow) && HasNoMethodIndexSources(rightRow) + ? null + : CombineMethodIndexSources( + GetMethodIndexSources(leftRow), + GetMethodIndexSources(rightRow)), + QualifiedFts5Sources: HasNoFts5Sources(leftRow) && HasNoFts5Sources(rightRow) + ? null + : CombineFts5Sources( + GetFts5Sources(leftRow), + GetFts5Sources(rightRow))); + } if (leftIsReverseCorrelatedSource) { @@ -40397,6 +41662,26 @@ SourceRow CreateJoinedRow(SourceRow leftRow, SourceRow rightRow) return CreateResult(rows); } + SourceData ProbeRightRows(SourceRow leftRow) + { + var probed = TryGetTransientLookupRows( + source.Right, + source.Condition, + parameters, + context, + leftRow with { Parent = outerRow }) + ?? (rightLookupFallback ??= GetSideSourceRows( + source.Right, + rightPredicate, + parameters, + context, + outerRow, + sourceOrderBy)); + return rightPredicate is null || ReferenceEquals(probed, rightLookupFallback) + ? probed + : FilterSourceRows(probed, rightPredicate, parameters, context); + } + var rightMatched = new bool[right.Rows.Count]; foreach (var leftRow in left.Rows) { @@ -40423,12 +41708,14 @@ SourceRow CreateJoinedRow(SourceRow leftRow, SourceRow rightRow) context, outerRow, sourceOrderBy)) + : rightProbesPerLeftRow + ? ProbeRightRows(leftRow) : right; AddOmittedPredicates(rowsForLeft.OmittedVirtualTablePredicates); var matched = false; var candidateIndices = joinHashIndex is not null ? joinHashIndex.Probe(leftRow) - : rightIsCorrelatedSource || rightDeclaredLookup is not null + : rightIsCorrelatedSource || rightDeclaredLookup is not null || rightProbesPerLeftRow ? Enumerable.Range(0, rowsForLeft.Rows.Count) : allRightIndices ??= Enumerable.Range(0, rowsForLeft.Rows.Count); foreach (var rightIndex in candidateIndices) @@ -41386,10 +42673,27 @@ public IReadOnlyList Probe(SourceRow leftRow) } private static string ToNumericKeyBits(double number) + => GetNumericKeyBits(number).ToString("X16", CultureInfo.InvariantCulture); + + /// The bits a numeric canonical key ("N" + 16 hex digits) encodes. + internal static long GetNumericKeyBits(double number) { if (number == 0) number = 0; // normalize -0 so 0 and -0 share a bucket - return BitConverter.DoubleToInt64Bits(number).ToString("X16", CultureInfo.InvariantCulture); + return BitConverter.DoubleToInt64Bits(number); + } + + /// Parses the bits back out of a numeric canonical key. + internal static bool TryGetNumericKeyBits(string key, out long bits) + { + bits = 0; + return key.Length == 17 + && key[0] == 'N' + && long.TryParse( + key.AsSpan(1), + NumberStyles.AllowHexSpecifier, + CultureInfo.InvariantCulture, + out bits); } private static string CanonicalizeJoinKeyText(string text, string collation) @@ -41448,6 +42752,50 @@ private static IReadOnlyDictionary CombineQualifiedColumns( return columns; } + private static SqlValue[] ConcatRowValues(SqlValue[] left, SqlValue[] right) + { + var values = new SqlValue[left.Length + right.Length]; + left.CopyTo(values, 0); + right.CopyTo(values, left.Length); + return values; + } + + // CombineQualifiedRowIds(GetQualifiedRowIds(left), GetQualifiedRowIds(right)) in one map. A join + // gives every row it produces the same qualifiers, so the map shares their layout and stores + // only this row's rowids (see JoinedRowIdMap); any other map shape takes the dictionary path. + private static IReadOnlyDictionary CombineQualifiedRowIds(SourceRow left, SourceRow right) + { + if (left.QualifiedRowIds is null or JoinedRowIdMap + && right.QualifiedRowIds is null or JoinedRowIdMap) + { + return JoinedRowIdMap.Combine(left, right); + } + + var rowIds = new Dictionary(StringComparer.OrdinalIgnoreCase); + foreach (var row in (ReadOnlySpan)[left, right]) + { + if (row.QualifiedRowIds is not null) + { + foreach (var (qualifier, rowId) in row.QualifiedRowIds) + rowIds.TryAdd(qualifier, rowId); + } + + if (row.RowIdQualifier is not null) + rowIds.TryAdd(row.RowIdQualifier, row.RowId); + } + + return rowIds; + } + + // Whether GetMethodIndexSources / GetFts5Sources would return an empty map for the row. + private static bool HasNoMethodIndexSources(SourceRow row) + => (row.QualifiedMethodIndexSources is null || row.QualifiedMethodIndexSources.Count == 0) + && (row.MethodIndexSource is null || row.RowIdQualifier is null); + + private static bool HasNoFts5Sources(SourceRow row) + => (row.QualifiedFts5Sources is null || row.QualifiedFts5Sources.Count == 0) + && (row.Fts5Source is null || row.RowIdQualifier is null); + private static IReadOnlyDictionary CombineQualifiedRowIds( IReadOnlyDictionary left, IReadOnlyDictionary right) @@ -41673,14 +43021,31 @@ private static IReadOnlyDictionary GetSourceQualifiedCol } } + // Scans and DML build one evaluation row per table row, each asking for the same map; the + // last one built on this thread is reused while its qualifier and column names still match. + [ThreadStatic] + private static (string Qualifier, string[] Columns, IReadOnlyDictionary Map)? t_lastQualifiedColumns; + private static IReadOnlyDictionary BuildQualifiedColumns( string qualifier, IReadOnlyList columns) { - var qualifiedColumns = new Dictionary(StringComparer.OrdinalIgnoreCase); + if (t_lastQualifiedColumns is { } last + && string.Equals(last.Qualifier, qualifier, StringComparison.Ordinal) + && last.Columns.Length == columns.Count) + { + var matches = true; + for (var index = 0; index < columns.Count && matches; index++) + matches = string.Equals(last.Columns[index], columns[index], StringComparison.Ordinal); + if (matches) + return last.Map; + } + + var qualifiedColumns = new Dictionary(columns.Count, StringComparer.OrdinalIgnoreCase); for (var index = 0; index < columns.Count; index++) qualifiedColumns.TryAdd($"{qualifier}.{columns[index]}", index); + t_lastQualifiedColumns = (qualifier, columns.ToArray(), qualifiedColumns); return qualifiedColumns; } @@ -41983,22 +43348,18 @@ IEnumerable EnumerateSourceRows() if (maximumRows is { } maxRows && maxRows < rowCount) rowCount = (int)maxRows; - var rowOrder = Enumerable.Range(0, table.Rows.Count).ToArray(); - if (table.HasRowid) - { - // A table B-tree cursor rewinds to its smallest integer key. Keep the evaluator's - // heap-backed table representation from leaking insertion order into physical scans. - Array.Sort( - rowOrder, - (left, right) => table.RowIds[left].CompareTo(table.RowIds[right])); - } + // A table B-tree cursor rewinds to its smallest integer key. Keep the evaluator's + // heap-backed table representation from leaking insertion order into physical scans. + // The rowid order is shared across statements (EmbeddedTable.GetRowidScanOrder). + var rowOrder = table.HasRowid && table.RowIds.Count == table.Rows.Count + ? table.GetRowidScanOrder() + : null; - var sourceRows = new SourceRow[rowCount]; - for (var outputIndex = 0; outputIndex < rowCount; outputIndex++) + SourceRow CreateRow(int outputIndex) { - var index = rowOrder[outputIndex]; + var index = rowOrder is null ? outputIndex : rowOrder[outputIndex]; var rowid = index < table.RowIds.Count ? table.RowIds[index] : index + 1; - sourceRows[outputIndex] = new SourceRow( + return new SourceRow( table.Columns, table.Rows[index], qualifiedColumns, @@ -42010,6 +43371,15 @@ IEnumerable EnumerateSourceRows() MethodIndexSource: methodIndexSource); } + // A bounded scan (LIMIT/OFFSET paging) builds rows on first access, so a caller that + // starts at the OFFSET never builds the rows it skips (see ExecuteSelect). + if (maximumRows is not null) + return new SourceData(table.Columns, new IndexedSourceRowList(rowCount, CreateRow)); + + var sourceRows = new SourceRow[rowCount]; + for (var outputIndex = 0; outputIndex < rowCount; outputIndex++) + sourceRows[outputIndex] = CreateRow(outputIndex); + return new SourceData(table.Columns, sourceRows); } @@ -51773,50 +53143,58 @@ private int Compare(SqlValue left, SqlValue right, string? collation = null) if (left.Kind == SqlValueKind.Real && right.Kind == SqlValueKind.Real) return left.AsReal().CompareTo(right.AsReal()); if (left.Kind == SqlValueKind.Text && right.Kind == SqlValueKind.Text) - { - if (_collations.TryGetValue(collation ?? "BINARY", out var compare)) - { - return InvokeManagedCallback( - () => compare(left.AsText(), right.AsText())); - } - - // Consult the external resolver BEFORE the built-in BINARY/NOCASE/RTRIM fallback. - // The own-registry check above only sees registrations made directly on this - // instance; the private index-expression/partial-predicate evaluator (see - // _externalCollationResolver) never registers anything of its own and relies - // entirely on this resolver to see the owning connection's actual collations — - // including an application override of a built-in name. Checking the hard-coded - // fallback first would let such a BINARY/NOCASE/RTRIM override silently miss inside - // a partial-index predicate or expression-index key while it is honored everywhere - // else. On the owning connection itself _externalCollationResolver is always null - // (it is only ever set via SetExternalCollationResolver on a throwaway evaluator - // instance) and any override is already caught by the own-registry check above, so - // this preserves ordinary EmbeddedDatabase.Compare precedence there. The resolver's - // own delegate already routes the invocation through the owning connection's - // InvokeManagedCallback (see BuildCollationResolver), so it is called directly - // rather than re-wrapped here. - var external = _externalCollationResolver?.Invoke(collation ?? "BINARY"); - if (external is not null) - return external(left.AsText(), right.AsText()); - - if (collation is null || string.Equals(collation, "BINARY", StringComparison.OrdinalIgnoreCase)) - return string.CompareOrdinal(left.AsText(), right.AsText()); - if (string.Equals(collation, "NOCASE", StringComparison.OrdinalIgnoreCase)) - return SqliteIndexRecordComparer.CompareNoCaseText(left.AsText(), right.AsText()); - if (string.Equals(collation, "RTRIM", StringComparison.OrdinalIgnoreCase)) - return SqliteIndexRecordComparer.CompareRTrimText(left.AsText(), right.AsText()); - - if (LocaleCollationRegistry.TryResolve(collation, out var localeCompare)) - return InvokeManagedCallback(() => localeCompare!(left.AsText(), right.AsText())); - - throw new EmbeddedSqlException($"no such collation sequence: {collation}"); - } + return CompareText(left.AsText(), right.AsText(), collation); if (left.Kind == SqlValueKind.Blob && right.Kind == SqlValueKind.Blob) return left.AsBlob().Span.SequenceCompareTo(right.AsBlob().Span); return left.Kind.CompareTo(right.Kind); } + // Separate from Compare because its callback lambdas capture the operands: C# allocates the + // closure on entry to the method that declares them, which made every comparison of any kind + // allocate. + private int CompareText(string leftText, string rightText, string? collation) + { + if (_collations.TryGetValue(collation ?? "BINARY", out var compare)) + { + return InvokeCollation(compare, leftText, rightText); + } + + // Consult the external resolver BEFORE the built-in BINARY/NOCASE/RTRIM fallback. + // The own-registry check above only sees registrations made directly on this + // instance; the private index-expression/partial-predicate evaluator (see + // _externalCollationResolver) never registers anything of its own and relies + // entirely on this resolver to see the owning connection's actual collations — + // including an application override of a built-in name. Checking the hard-coded + // fallback first would let such a BINARY/NOCASE/RTRIM override silently miss inside + // a partial-index predicate or expression-index key while it is honored everywhere + // else. On the owning connection itself _externalCollationResolver is always null + // (it is only ever set via SetExternalCollationResolver on a throwaway evaluator + // instance) and any override is already caught by the own-registry check above, so + // this preserves ordinary EmbeddedDatabase.Compare precedence there. The resolver's + // own delegate already routes the invocation through the owning connection's + // InvokeManagedCallback (see BuildCollationResolver), so it is called directly + // rather than re-wrapped here. + var external = _externalCollationResolver?.Invoke(collation ?? "BINARY"); + if (external is not null) + return external(leftText, rightText); + + if (collation is null || string.Equals(collation, "BINARY", StringComparison.OrdinalIgnoreCase)) + return string.CompareOrdinal(leftText, rightText); + if (string.Equals(collation, "NOCASE", StringComparison.OrdinalIgnoreCase)) + return SqliteIndexRecordComparer.CompareNoCaseText(leftText, rightText); + if (string.Equals(collation, "RTRIM", StringComparison.OrdinalIgnoreCase)) + return SqliteIndexRecordComparer.CompareRTrimText(leftText, rightText); + + if (LocaleCollationRegistry.TryResolve(collation, out var localeCompare)) + return InvokeCollation(localeCompare!, leftText, rightText); + + throw new EmbeddedSqlException($"no such collation sequence: {collation}"); + } + + private int InvokeCollation(Func compare, string leftText, string rightText) + => InvokeManagedCallback(() => compare(leftText, rightText)); + private static int CompareIntegerAndReal(long integer, double real) { if (real < long.MinValue) @@ -54750,7 +56128,7 @@ private static long ToSqliteInteger(SqlValue value) if (TryGetExactWindowInteger(value, out var integer)) return integer; if (value.Kind == SqlValueKind.Real) - return (long)value.AsReal(); + return SaturatingToInt64(value.AsReal()); if (value.Kind == SqlValueKind.Text && double.TryParse( EmbeddedTable.TrimAsciiWhitespace(value.AsText()), @@ -54758,12 +56136,26 @@ private static long ToSqliteInteger(SqlValue value) CultureInfo.InvariantCulture, out var real)) { - return (long)real; + return SaturatingToInt64(real); } return 0; } + // SQLite's doubleToInt64: out-of-range values clamp to the int64 limits and NaN becomes 0. A + // plain cast is runtime-dependent there (.NET 9+ saturates; .NET 8 on x64 yields long.MinValue + // for any out-of-range value, which turned substr('abc', 1.8e19) into the whole string). + private static long SaturatingToInt64(double value) + { + if (double.IsNaN(value)) + return 0; + if (value <= long.MinValue) + return long.MinValue; + if (value >= long.MaxValue) + return long.MaxValue; + return (long)value; + } + private int CompareRows( SourceRow left, SourceRow right, @@ -61249,8 +62641,7 @@ public EmbeddedStatement Prepare(string sql) ThrowIfRecursiveTriggerCallbackReentry(); ThrowIfDisposed(); ThrowIfInsideHookCallback(); - var parameterMap = SqlParameterMap.Parse(sql); - var statement = SqlParser.Parse(sql, parameterMap, IsKnownTableOrViewName); + var (parameterMap, statement) = ParseCached(sql); if (statement is CreateTypeStatement or CreateDomainStatement && !ExperimentalCustomTypesEnabled) throw new EmbeddedSqlException("Custom types are experimental and are not enabled for this connection."); if (_hooks.Authorizer is not null) @@ -61259,6 +62650,59 @@ public EmbeddedStatement Prepare(string sql) return new EmbeddedStatement(this, statement, parameterMap, sql); } + private const int MaximumParseCacheEntries = 128; + + // Parsed statements by SQL text. ADO.NET executes a command by preparing its text again (a + // reused or Prepare()d command included), so re-parsing showed up on every execution. The + // parse consults the schema only through IsKnownTableOrViewName, so an entry records each + // name it asked about with the answer, and is reused only while every answer is unchanged. + // Statement trees are immutable, so executions can share one. + private readonly Dictionary _parseCache = new(StringComparer.Ordinal); + + private sealed record CachedParse( + SqlParameterMap ParameterMap, + ParsedStatement Statement, + (string Name, bool Known)[] NameProbes); + + private (SqlParameterMap ParameterMap, ParsedStatement Statement) ParseCached(string sql) + { + CachedParse? cached; + lock (_parseCache) + _parseCache.TryGetValue(sql, out cached); + if (cached is not null) + { + var current = true; + foreach (var (name, known) in cached.NameProbes) + { + if (IsKnownTableOrViewName(name) != known) + { + current = false; + break; + } + } + + if (current) + return (cached.ParameterMap, cached.Statement); + } + + var probes = new List<(string Name, bool Known)>(); + var parameterMap = SqlParameterMap.Parse(sql); + var statement = SqlParser.Parse(sql, parameterMap, name => + { + var known = IsKnownTableOrViewName(name); + probes.Add((name, known)); + return known; + }); + lock (_parseCache) + { + if (_parseCache.Count >= MaximumParseCacheEntries && !_parseCache.ContainsKey(sql)) + _parseCache.Clear(); + _parseCache[sql] = new CachedParse(parameterMap, statement, [.. probes]); + } + + return (parameterMap, statement); + } + /// /// The callbacks registered on this connection. Mutating the returned instance takes effect /// on the next prepared statement or execution, mirroring SQLite's hook registration APIs. @@ -61464,6 +62908,15 @@ public IReadOnlyList PrepareScript(string sql) } public void ResetForPooling() + => ResetForPooling(adoptCommittedChanges: true); + + /// + /// Resets connection state like . With + /// false it skips adopting other connections' + /// commits, which costs file-stat calls and is wasted on a connection going idle: renting it + /// adopts them again, as does every statement. + /// + internal void ResetForPooling(bool adoptCommittedChanges) { ThrowIfRecursiveTriggerCallbackReentry(); ThrowIfDisposed(); @@ -61503,7 +62956,8 @@ public void ResetForPooling() attachment.Dispose(); _attachedDatabases.Clear(); ResetTemporaryDatabase(); - _database.RefreshFileCatalogForPooling(); + if (adoptCommittedChanges) + _database.RefreshFileCatalogForPooling(); } public void RegisterScalarFunction(string name, int arity, Func, SqlValue> function) @@ -61957,7 +63411,7 @@ or AttachDatabaseStatement BeginConcurrentSchemaChange(); } - if (_transactionDatabases is null) + if (_transactionDatabases is null && !ReadsOnlyConnectionState(statement)) { _database.RefreshForeignCatalogForStatementIfNeeded(); _database.RefreshOwnedCatalogForStatementIfNeeded(); @@ -62890,6 +64344,19 @@ private EmbeddedDatabase ResolveSchemaDatabase(string schema) private EmbeddedDatabase ResolvePragmaDatabase(string? schema) => schema is null ? _database : ResolveSchemaDatabase(schema); + /// + /// Pragmas that only read or set this connection's own flags. Like SQLite, they never open a + /// read transaction, so they skip adopting peers' commits (which costs file-stat calls); the + /// next statement that reads the database still adopts them. Data source providers issue + /// PRAGMA foreign_keys on every pooled open. + /// + private static bool ReadsOnlyConnectionState(ParsedStatement statement) + => statement is PragmaForeignKeysStatement + or PragmaDeferForeignKeysStatement + or PragmaRecursiveTriggersStatement + or PragmaCountChangesStatement + or PragmaBusyTimeoutStatement; + private void ValidatePragmaSchema(string? schema) { if (schema is not null) @@ -63854,6 +65321,52 @@ private RoutedStatement RoutePragmaTableMetadataStatement( private string ResolveExistingObjectSchema(string objectName, ManagedSchemaObjectKind kind) => FindExistingObjectSchema(objectName, kind) ?? "main"; + /// + /// An identity for the base table an unqualified resolves to, as + /// PRAGMA table_info/index_list resolve it (temp, main, then attached schemas, each + /// through this connection's transaction catalog when one is open). Equal identities mean the + /// table's columns, declared types and indexes are unchanged, so metadata derived from those + /// pragmas can be reused. Null for anything else (views, virtual tables, qualified names), + /// which callers must resolve through the pragmas themselves. + /// + internal object? TryGetTableSchemaIdentity(string tableName) + { + if (tableName.Contains('.', StringComparison.Ordinal) + || FindExistingObjectSchema(tableName, ManagedSchemaObjectKind.Table) is not { } schema) + { + return null; + } + + var database = schema switch + { + "temp" => _tempDatabase, + "main" => _database, + _ => _attachedDatabases.TryGetValue(schema, out var attached) ? attached.Database : null, + }; + if (database is null) + return null; + + var tables = GetTransactionState(database)?.Catalog.Tables ?? database.LiveCatalog.Tables; + if (!tables.TryGetValue(tableName, out var table)) + return null; + + var signature = new StringBuilder(); + foreach (var column in table.ColumnDefinitions) + signature.Append(column.Name).Append(':').Append(column.DeclaredType).Append(column.NotNull ? "!" : string.Empty).Append(column.PrimaryKey ? "*" : string.Empty).Append(';'); + signature.Append('|'); + foreach (var index in table.Indexes) + { + signature.Append(index.Name).Append(index.Unique ? "+U" : string.Empty).Append(index.IsPartial ? "+P" : string.Empty).Append('('); + foreach (var term in index.Columns) + signature.Append(term.Name).Append(term.ExpressionSql).Append(','); + signature.Append(");"); + } + + return new TableSchemaIdentity(schema, table.ColumnDefinitions, signature.ToString()); + } + + private sealed record TableSchemaIdentity(string Schema, EmbeddedColumn[] Columns, string Signature); + private string? FindExistingObjectSchema(string objectName, ManagedSchemaObjectKind kind) { if (GetTransactionState(_tempDatabase) is { } tempState @@ -69498,6 +71011,17 @@ internal sealed class RowStore : IList, IReadOnlyList public int Count => _rows.Count; + /// + /// Process-unique identity of the current row contents (see + /// ). Unlike it also changes + /// on , and it never repeats across diverging clones. + /// + public long ContentStamp => _rows.ContentStamp; + + /// See . + internal bool SharesChunkWith(RowStore other, int chunkIndex) + => _rows.SharesChunkWith(other._rows, chunkIndex); + public bool IsReadOnly => false; public SqlValue[] this[int index] @@ -69642,12 +71166,14 @@ internal sealed class EmbeddedTable private InsertConflictAlgorithm? _effectivePrimaryKeyConflictAlgorithm; private int[]? _cachedRowidScanOrder; private long _cachedRowidScanOrderRevision = -1; - private Dictionary? _cachedRowIdPositions; - private long _cachedRowIdPositionsRevision = -1; - private int _cachedRowIdPositionsCount = -1; + private RowIdLookup? _cachedRowIdLookup; + private long _cachedRowIdLookupRowIdsStamp; private readonly Dictionary _cachedIndexScanOrders = new(StringComparer.OrdinalIgnoreCase); + // Shared with every Clone() of this table; see TableDerivedCache. + private TableDerivedCache _derivedCache = new(); + public EmbeddedTable( string name, IReadOnlyList columns, @@ -69736,6 +71262,44 @@ public EmbeddedTable( ValidateSchemaExpressions(); } + /// + /// Copies this table's already-validated schema state for and + /// , which every statement runs for every table of its working + /// catalog. The public constructor re-derives and re-validates everything (generated-column + /// order, primary-key schema, foreign keys, constraint indexes, schema expressions) from the + /// definition, which made catalog cloning dominate small statements. Every value copied here + /// is either immutable or replaced (never mutated) by the instance methods that change it, + /// except the column-name map, which mutates and is therefore copied. + /// Rows, rowids, caches and explicit indexes are left to the caller exactly as before. + /// + private EmbeddedTable(EmbeddedTable source) + { + Name = source.Name; + ColumnDefinitions = source.ColumnDefinitions; + Columns = source.Columns; + _columnIndices = new Dictionary(source._columnIndices, StringComparer.OrdinalIgnoreCase); + WithoutRowid = source.WithoutRowid; + Strict = source.Strict; + TableLevelPrimaryKey = source.TableLevelPrimaryKey; + TableUniqueConstraints = source.TableUniqueConstraints; + CheckConstraints = source.CheckConstraints; + TableForeignKeys = source.TableForeignKeys; + TablePrimaryKeyConflictAlgorithm = source.TablePrimaryKeyConflictAlgorithm; + TablePrimaryKeyConstraintName = source.TablePrimaryKeyConstraintName; + TablePrimaryKeyDeclarationOrder = source.TablePrimaryKeyDeclarationOrder; + PrimaryKeyColumns = source.PrimaryKeyColumns; + PrimaryKeySchema = source.PrimaryKeySchema; + RowidAliasColumnIndex = source.RowidAliasColumnIndex; + IsAutoIncrement = source.IsAutoIncrement; + GeneratedColumnOrder = source.GeneratedColumnOrder; + ForeignKeys = source.ForeignKeys; + WithoutRowidPrimaryKeyIndexName = source.WithoutRowidPrimaryKeyIndexName; + PrimaryKeyConstraintOrdinal = source.PrimaryKeyConstraintOrdinal; + _hasEffectivePrimaryKeyConflictAlgorithm = source._hasEffectivePrimaryKeyConflictAlgorithm; + _effectivePrimaryKeyConflictAlgorithm = source._effectivePrimaryKeyConflictAlgorithm; + Indexes.AddRange(source.Indexes.Where(index => index.Origin != EmbeddedIndexOrigin.Explicit)); + } + private void CreateConstraintIndexes() { if (WithoutRowid) @@ -70433,10 +71997,11 @@ public CowChunkedList RowIds /// internal (long LineageId, long Revision) RowStorageIdentity => (_rowsStore.LineageId, _rowsStore.Revision); - // The rowids as an immutable set, cached against the row store's identity. Every rowid change - // is paired with a row change that bumps RowStore.Revision (RowIds and Rows are index-aligned), - // so a matching lineage, revision and count proves the set current. It is immutable so a clone - // can share it and each side can extend it without affecting the other. + // The rowids as an immutable set, cached against the rowid list's content stamp, which changes + // exactly when a rowid does (CowChunkedList.ContentStamp): an UPDATE that rewrites row values + // keeps the set current, where keying it on the row store's revision rebuilt it after every + // write. It is immutable so a clone can share it and each side can extend it without + // affecting the other. private RowIdSet? _rowIdSet; /// The table's current rowids and their maximum, rebuilt only when stale. @@ -70457,8 +72022,7 @@ internal RowIdSet GetRowIdSet() } var built = new RowIdSet( - _rowsStore.LineageId, - _rowsStore.Revision, + rowIds.ContentStamp, rowIds.Count, builder.ToImmutable(), max); @@ -70471,13 +72035,47 @@ internal RowIdSet GetRowIdSet() { var rowIds = RowIds; return _rowIdSet is { } cached - && cached.LineageId == _rowsStore.LineageId - && cached.Revision == _rowsStore.Revision + && cached.RowIdsStamp == rowIds.ContentStamp && cached.Count == rowIds.Count ? cached : null; } + /// + /// Narrows , the set that was current immediately before + /// were deleted, so the next INSERT does not rebuild it. + /// + internal void RecordRemovedRowIds(RowIdSet? before, IReadOnlyList removed) + { + if (before is null || before.Count - removed.Count != RowIds.Count) + return; + + var builder = before.Ids.ToBuilder(); + var removedMaximum = false; + foreach (var rowId in removed) + { + if (!builder.Remove(rowId)) + return; + removedMaximum |= rowId == before.Max; + } + + // The next allocated rowid is max + 1, so a removed maximum is recomputed exactly; a scan + // of the rowid list is far cheaper than rebuilding the immutable set. + var max = before.Max; + if (removedMaximum) + { + max = long.MinValue; + var rowIds = RowIds; + for (var index = 0; index < rowIds.Count; index++) + { + if (rowIds[index] > max) + max = rowIds[index]; + } + } + + _rowIdSet = new RowIdSet(RowIds.ContentStamp, RowIds.Count, builder.ToImmutable(), max); + } + // Unique-index key sets, cached against the row store's identity exactly like _rowIdSet and // keyed by an index-definition signature (a clone rebuilds its PRIMARY KEY/UNIQUE autoindex // objects, so the index reference cannot be the key). Immutable so clones share them. @@ -70523,8 +72121,7 @@ internal void RecordAppendedRowIds(RowIdSet? before, IReadOnlyList inserte } _rowIdSet = new RowIdSet( - _rowsStore.LineageId, - _rowsStore.Revision, + RowIds.ContentStamp, RowIds.Count, builder.ToImmutable(), max); @@ -70725,10 +72322,7 @@ internal IEnumerable GetScanOrderIndices() if (RowIds.Count != Rows.Count) throw new InvalidOperationException($"Table '{Name}' has inconsistent row identity metadata."); - _cachedRowidScanOrder = Enumerable.Range(0, Rows.Count).ToArray(); - Array.Sort( - _cachedRowidScanOrder, - (left, right) => RowIds[left].CompareTo(RowIds[right])); + _cachedRowidScanOrder = GetSharedRowidScanOrder(); _cachedRowidScanOrderRevision = Rows.Revision; } @@ -70736,68 +72330,172 @@ internal IEnumerable GetScanOrderIndices() yield return index; } - // STAT4 validates a small sample set repeatedly while planning. Cache the rowid map by - // RowStore revision, but verify the live slot so same-count rowid replacements fail closed. - internal int TryGetRowIdPosition(long rowId, out bool cacheRebuilt) + /// + /// Row positions in ascending rowid order, shared with every table holding the same rows and + /// rowids. The returned array must not be modified. + /// + internal int[] GetRowidScanOrder() => GetSharedRowidScanOrder(); + + private int[] GetSharedRowidScanOrder() { - cacheRebuilt = false; - if (!HasRowid || RowIds.Count != Rows.Count) + var rowsStamp = Rows.ContentStamp; + var rowIdsStamp = RowIds.ContentStamp; + if (_derivedCache.TryGet(DerivedCacheKind.RowidScanOrder, rowsStamp, rowIdsStamp, out var shared)) + return shared; + + var count = Rows.Count; + var order = new int[count]; + var sorted = true; + var previous = long.MinValue; + for (var index = 0; index < count; index++) { - InvalidateCache(); - return -1; + order[index] = index; + var rowId = RowIds[index]; + if (index > 0 && rowId < previous) + sorted = false; + previous = rowId; } - var builtThisCall = false; - if (_cachedRowIdPositions is null - || _cachedRowIdPositionsRevision != Rows.Revision - || _cachedRowIdPositionsCount != RowIds.Count) + // Rows are almost always stored in rowid order (page loads and appends); only sort when not. + if (!sorted) { - RebuildCache(); - cacheRebuilt = true; - builtThisCall = true; + var keys = new long[count]; + for (var index = 0; index < count; index++) + keys[index] = RowIds[index]; + Array.Sort(keys, order); } - if (TryReadLivePosition(rowId, out var position)) - return position; - if (builtThisCall) - return -1; + _derivedCache.Set(DerivedCacheKind.RowidScanOrder, rowsStamp, rowIdsStamp, order); + return order; + } - // A same-count rowid replacement can bypass RowStore.Revision. Rebuild once on a stale - // hit or miss, then let the caller's live index-key validation decide whether to trust it. - RebuildCache(); - cacheRebuilt = true; - return TryReadLivePosition(rowId, out position) ? position : -1; + /// + /// A read-only structure derived from exactly this table's current rows and rowids, built by + /// any clone sharing them. Values must not be mutated (see ). + /// + /// The derived structure for if one is current, without building it. + internal bool TryGetDerived(object key, [System.Diagnostics.CodeAnalysis.NotNullWhen(true)] out T? value) + where T : class + => _derivedCache.TryGet(key, Rows.ContentStamp, RowIds.ContentStamp, out value); + + internal T GetOrCreateDerived(object key, Func build) + where T : class + { + var rowsStamp = Rows.ContentStamp; + var rowIdsStamp = RowIds.ContentStamp; + if (_derivedCache.TryGet(key, rowsStamp, rowIdsStamp, out var shared)) + return shared; + + var built = build(); + _derivedCache.Set(key, rowsStamp, rowIdsStamp, built); + return built; + } + + private enum DerivedCacheKind + { + RowidScanOrder, + RowIdPositions, + } - bool TryReadLivePosition(long id, out int found) + // STAT4 validates a small sample set repeatedly while planning, and rowid lookups probe it + // once per statement. The lookup is shared across clones and keyed by the exact rowids content + // stamp, so a rowid replacement that bypasses RowStore.Revision still invalidates it; the live + // slot is verified anyway. Row contents do not affect positions (the counts are checked live), + // so an in-place UPDATE keeps it. Rowids are normally stored in ascending order, and then a + // binary search over them replaces the position map: confirming the order is one sequential + // pass, where rebuilding the map after every INSERT or DELETE hashed every rowid. + internal int TryGetRowIdPosition(long rowId, out bool cacheRebuilt) + { + cacheRebuilt = false; + if (!HasRowid || RowIds.Count != Rows.Count) { - if (_cachedRowIdPositions!.TryGetValue(id, out found) - && found >= 0 - && found < RowIds.Count - && RowIds[found] == id) + _cachedRowIdLookup = null; + return -1; + } + + const long rowsStamp = 0; + var rowIds = RowIds; + var rowIdsStamp = rowIds.ContentStamp; + if (_cachedRowIdLookup is null + || _cachedRowIdLookupRowIdsStamp != rowIdsStamp) + { + if (!_derivedCache.TryGet( + DerivedCacheKind.RowIdPositions, + rowsStamp, + rowIdsStamp, + out var lookup)) { - return true; + lookup = RowIdsAreAscending(rowIds) + ? RowIdLookup.Ascending + : new RowIdLookup(BuildRowIdPositions(rowIds)); + _derivedCache.Set(DerivedCacheKind.RowIdPositions, rowsStamp, rowIdsStamp, lookup); + cacheRebuilt = true; } - found = -1; - return false; + _cachedRowIdLookup = lookup; + _cachedRowIdLookupRowIdsStamp = rowIdsStamp; } - void RebuildCache() + int found; + if (_cachedRowIdLookup.Positions is { } positions) { - var positions = new Dictionary(RowIds.Count); - for (var index = 0; index < RowIds.Count; index++) - positions.TryAdd(RowIds[index], index); - _cachedRowIdPositions = positions; - _cachedRowIdPositionsRevision = Rows.Revision; - _cachedRowIdPositionsCount = RowIds.Count; + if (!positions.TryGetValue(rowId, out found)) + return -1; } + else + { + var low = 0; + var high = rowIds.Count - 1; + found = -1; + while (low <= high) + { + var middle = low + ((high - low) / 2); + var candidate = rowIds[middle]; + if (candidate == rowId) + { + found = middle; + break; + } - void InvalidateCache() + if (candidate < rowId) + low = middle + 1; + else + high = middle - 1; + } + } + + return found >= 0 + && found < rowIds.Count + && rowIds[found] == rowId + ? found + : -1; + } + + private static bool RowIdsAreAscending(CowChunkedList rowIds) + { + for (var index = 1; index < rowIds.Count; index++) { - _cachedRowIdPositions = null; - _cachedRowIdPositionsRevision = -1; - _cachedRowIdPositionsCount = -1; + if (rowIds[index] <= rowIds[index - 1]) + return false; } + + return true; + } + + private static Dictionary BuildRowIdPositions(CowChunkedList rowIds) + { + var positions = new Dictionary(rowIds.Count); + for (var index = 0; index < rowIds.Count; index++) + positions.TryAdd(rowIds[index], index); + return positions; + } + + // Positions null: the rowids are strictly ascending and are binary-searched directly. + private sealed class RowIdLookup(Dictionary? positions) + { + public static readonly RowIdLookup Ascending = new(null); + + public Dictionary? Positions { get; } = positions; } internal IReadOnlyList GetOrCreateIndexScanOrder( @@ -70958,6 +72656,33 @@ public bool ColumnHasNumericAffinity(int columnIndex) return affinity is ColumnAffinity.Integer or ColumnAffinity.Real or ColumnAffinity.Numeric; } + // Indexes of the REAL-affinity columns, derived from ColumnDefinitions (replaced, never + // mutated, by schema changes) and cached against that array instance. + private (EmbeddedColumn[] Definitions, int[] Columns)? _realAffinityColumns; + + /// + /// The columns with REAL affinity. A stored record may hold an integral REAL value as an + /// integer (SQLite's on-disk form), which reads back as REAL; see + /// . + /// + internal int[] GetRealAffinityColumns() + { + var definitions = ColumnDefinitions; + if (_realAffinityColumns is { } cached && ReferenceEquals(cached.Definitions, definitions)) + return cached.Columns; + + var columns = new List(); + for (var index = 0; index < definitions.Length; index++) + { + if (GetColumnAffinity(definitions[index]) == ColumnAffinity.Real) + columns.Add(index); + } + + var result = columns.ToArray(); + _realAffinityColumns = (definitions, result); + return result; + } + public ColumnAffinity GetColumnAffinity(EmbeddedColumn column) => column.Domain is not null || column.IdentityType is not null ? ColumnAffinity.Integer @@ -72337,18 +74062,7 @@ string[] RenameAll(IReadOnlyList columns) public EmbeddedTable Clone() { - var clone = new EmbeddedTable( - Name, - ColumnDefinitions, - WithoutRowid, - TableLevelPrimaryKey, - TableUniqueConstraints, - CheckConstraints, - TablePrimaryKeyConflictAlgorithm, - TablePrimaryKeyConstraintName, - TablePrimaryKeyDeclarationOrder, - TableForeignKeys, - Strict); + var clone = new EmbeddedTable(this); clone.SchemaSqlCompact = SchemaSqlCompact; clone.Sql = Sql; if (!TryCopyPendingRowLoadTo(clone)) @@ -72359,6 +74073,8 @@ public EmbeddedTable Clone() clone._uniqueIndexKeySets = _uniqueIndexKeySets; } + clone._derivedCache = _derivedCache; + clone.Indexes.RemoveAll(index => index.Origin == EmbeddedIndexOrigin.Explicit); clone.Indexes.AddRange(Indexes.Where(index => index.Origin == EmbeddedIndexOrigin.Explicit)); CopyMethodAttachmentsTo(clone); @@ -72458,18 +74174,7 @@ IReadOnlyList RewriteChecks(IReadOnlyList chec // that turned N-row bulk inserts into O(N^2) minute-long hangs. public EmbeddedTable CloneShallow() { - var clone = new EmbeddedTable( - Name, - ColumnDefinitions, - WithoutRowid, - TableLevelPrimaryKey, - TableUniqueConstraints, - CheckConstraints, - TablePrimaryKeyConflictAlgorithm, - TablePrimaryKeyConstraintName, - TablePrimaryKeyDeclarationOrder, - TableForeignKeys, - Strict); + var clone = new EmbeddedTable(this); clone.SchemaSqlCompact = SchemaSqlCompact; clone.Sql = Sql; clone.Rows.ShareRowsWithFreshIdentity(Rows); @@ -73366,16 +75071,28 @@ private bool TryGetValue(string name, bool allowQualifiedLookup, out SqlValue va // Columns joined with USING/NATURAL are coalesced: an unqualified reference to // such a column must resolve to COALESCE(left, right) so RIGHT/FULL joins report // the surviving side rather than the NULL-padded one. - var coalesced = OutputColumns? - .Where(output => string.Equals(output.Name, name, StringComparison.OrdinalIgnoreCase) - && (output.CoalesceIndex is not null - || output.AdditionalCoalesceIndices is { Count: > 0 })) - .ToArray(); - if (coalesced is { Length: > 1 }) - ThrowAmbiguousColumn(name); - if (coalesced is { Length: 1 }) + // A plain loop: a LINQ filter here captured the name, allocating on every lookup. + OutputColumn? coalescedOutput = null; + if (OutputColumns is not null) + { + for (var position = 0; position < OutputColumns.Count; position++) + { + var candidate = OutputColumns[position]; + if (!string.Equals(candidate.Name, name, StringComparison.OrdinalIgnoreCase) + || (candidate.CoalesceIndex is null + && candidate.AdditionalCoalesceIndices is not { Count: > 0 })) + { + continue; + } + + if (coalescedOutput is not null) + ThrowAmbiguousColumn(name); + coalescedOutput = candidate; + } + } + + if (coalescedOutput is { } output) { - var output = coalesced[0]; value = Values[output.Index]; if (value.Kind != SqlValueKind.Null) return true; @@ -73443,6 +75160,175 @@ private static void ThrowAmbiguousColumn(string name) } +/// +/// A read-only qualifier-to-rowid map for one joined row. The qualifiers, their order and where each +/// value comes from depend only on the joined sides' own layouts, so they live in a shared +/// and a row stores just its values. Keys compare case-insensitively and +/// enumerate in insertion order, like the dictionary it replaces. +/// +internal sealed class JoinedRowIdMap : IReadOnlyDictionary +{ + private readonly Layout _layout; + private readonly long?[] _values; + + private JoinedRowIdMap(Layout layout, long?[] values) + { + _layout = layout; + _values = values; + } + + // The last few layouts built on this thread; nested joins alternate between a handful. + [ThreadStatic] + private static Layout?[]? t_layouts; + + [ThreadStatic] + private static int t_nextLayout; + + public static JoinedRowIdMap Combine(SourceRow left, SourceRow right) + { + var leftLayout = (left.QualifiedRowIds as JoinedRowIdMap)?._layout; + var rightLayout = (right.QualifiedRowIds as JoinedRowIdMap)?._layout; + var layout = FindLayout(leftLayout, left.RowIdQualifier, rightLayout, right.RowIdQualifier); + var values = new long?[layout.Keys.Count]; + for (var slot = 0; slot < values.Length; slot++) + { + var (side, sourceSlot) = layout.Sources[slot]; + var row = side == 0 ? left : right; + values[slot] = sourceSlot < 0 ? row.RowId : ((JoinedRowIdMap)row.QualifiedRowIds!)._values[sourceSlot]; + } + + return new JoinedRowIdMap(layout, values); + } + + private static Layout FindLayout(Layout? leftLayout, string? leftQualifier, Layout? rightLayout, string? rightQualifier) + { + var layouts = t_layouts ??= new Layout?[4]; + foreach (var candidate in layouts) + { + if (candidate is not null + && ReferenceEquals(candidate.LeftLayout, leftLayout) + && ReferenceEquals(candidate.RightLayout, rightLayout) + && string.Equals(candidate.LeftQualifier, leftQualifier, StringComparison.Ordinal) + && string.Equals(candidate.RightQualifier, rightQualifier, StringComparison.Ordinal)) + { + return candidate; + } + } + + var layout = new Layout(leftLayout, leftQualifier, rightLayout, rightQualifier); + layout.AddSide(0, leftLayout, leftQualifier); + layout.AddSide(1, rightLayout, rightQualifier); + layouts[t_nextLayout] = layout; + t_nextLayout = (t_nextLayout + 1) % layouts.Length; + return layout; + } + + public long? this[string key] + => TryGetValue(key, out var value) ? value : throw new KeyNotFoundException(key); + + public IEnumerable Keys => _layout.Keys; + + public IEnumerable Values => _values; + + public int Count => _values.Length; + + public bool ContainsKey(string key) => _layout.Slots.ContainsKey(key); + + public bool TryGetValue(string key, out long? value) + { + if (_layout.Slots.TryGetValue(key, out var slot)) + { + value = _values[slot]; + return true; + } + + value = null; + return false; + } + + public IEnumerator> GetEnumerator() + { + for (var slot = 0; slot < _values.Length; slot++) + yield return new KeyValuePair(_layout.Keys[slot], _values[slot]); + } + + System.Collections.IEnumerator System.Collections.IEnumerable.GetEnumerator() => GetEnumerator(); + + private sealed class Layout(Layout? leftLayout, string? leftQualifier, Layout? rightLayout, string? rightQualifier) + { + public Layout? LeftLayout { get; } = leftLayout; + + public string? LeftQualifier { get; } = leftQualifier; + + public Layout? RightLayout { get; } = rightLayout; + + public string? RightQualifier { get; } = rightQualifier; + + public Dictionary Slots { get; } = new(StringComparer.OrdinalIgnoreCase); + + public List Keys { get; } = []; + + // Per slot: the side it is read from (0 left, 1 right) and that side's slot, or -1 for + // the side row's own RowId. + public List<(int Side, int SourceSlot)> Sources { get; } = []; + + public void AddSide(int side, Layout? mapLayout, string? qualifier) + { + if (mapLayout is not null) + { + for (var slot = 0; slot < mapLayout.Keys.Count; slot++) + Add(mapLayout.Keys[slot], side, slot); + } + + if (qualifier is not null) + Add(qualifier, side, -1); + } + + private void Add(string key, int side, int sourceSlot) + { + if (Slots.TryAdd(key, Keys.Count)) + { + Keys.Add(key); + Sources.Add((side, sourceSlot)); + } + } + } +} + +/// +/// Source rows built on first access and then kept, so every reader of a position sees the same +/// instance (callers key rows by reference). +/// +internal sealed class IndexedSourceRowList(int count, Func create) : IReadOnlyList +{ + private readonly SourceRow?[] _rows = new SourceRow?[count]; + + public int Count => _rows.Length; + + public SourceRow this[int index] + { + get + { + var existing = Volatile.Read(ref _rows[index]); + if (existing is not null) + return existing; + + var created = create(index); + return Interlocked.CompareExchange(ref _rows[index], created, null) ?? created; + } + } + + public IEnumerable EnumerateFrom(int start) + { + for (var index = start; index < _rows.Length; index++) + yield return this[index]; + } + + public IEnumerator GetEnumerator() => EnumerateFrom(0).GetEnumerator(); + + System.Collections.IEnumerator System.Collections.IEnumerable.GetEnumerator() => GetEnumerator(); +} + internal sealed record SourceData( string[] Columns, IReadOnlyList Rows, diff --git a/src/Ahtola.Core/EmbeddedFileStore.cs b/src/Ahtola.Core/EmbeddedFileStore.cs index fff1bba..3bdcb14 100644 --- a/src/Ahtola.Core/EmbeddedFileStore.cs +++ b/src/Ahtola.Core/EmbeddedFileStore.cs @@ -1213,6 +1213,7 @@ private IEnumerable SeekCommittedRowidIndex( Array.Fill(row, SqlValue.Null); for (var position = 0; position < index.Columns.Count; position++) row[index.Columns[position].ColumnIndex] = indexValues[position]; + ApplyStoredRealAffinity(table, row); } else { @@ -1385,6 +1386,7 @@ private IEnumerable ScanCommittedRowidIndexAscending( Array.Fill(row, SqlValue.Null); for (var position = 0; position < index.Columns.Count; position++) row[index.Columns[position].ColumnIndex] = indexValues[position]; + ApplyStoredRealAffinity(table, row); } else { @@ -3023,9 +3025,121 @@ internal static SqlValue[] RestoreRowidTableRecord( source++; } + ApplyStoredRealAffinity(table, row); return row; } + /// + /// Reads a stored INTEGER in a REAL-affinity column back as REAL. SQLite writes an integral + /// REAL value (2.0) in its integer form and converts it on every read (OP_RealAffinity after + /// OP_Column; turso-src/core/vdbe/execute.rs op_real_affinity), so typeof() is 'real' and a + /// data reader yields a double. Without this, rows decoded from a SQLite-written page kept + /// the integer storage class. + /// + internal static void ApplyStoredRealAffinity(EmbeddedTable table, SqlValue[] row) + { + foreach (var columnIndex in table.GetRealAffinityColumns()) + { + if (columnIndex < row.Length && row[columnIndex].Kind == SqlValueKind.Integer) + row[columnIndex] = SqlValue.Real(row[columnIndex].AsInteger()); + } + } + + /// + /// The on-disk form of a REAL-affinity value: SQLite's OP_MakeRecord stores a REAL that is + /// exactly an integer of magnitude below 2^51 (sqlite3RealSameAsInt, zero of either sign + /// included) in integer form, which turns back into + /// REAL on read. Writing the same form keeps records byte-identical to SQLite's and keeps a + /// rewritten row from growing (an 8-byte float where SQLite stored a small integer would + /// overflow full leaves on ordinary UPDATEs). + /// + internal static SqlValue ToStoredRealForm(SqlValue value) + { + if (value.Kind != SqlValueKind.Real) + return value; + + var real = value.AsReal(); + if (real == 0) + return SqlValue.Integer(0); + const double Limit = 2251799813685248d; // 2^51 + if (real >= -Limit && real < Limit) + { + var integer = (long)real; + if ((double)integer == real) + return SqlValue.Integer(integer); + } + + return value; + } + + private static bool IsRealAffinityColumn(EmbeddedTable table, int columnIndex) + => Array.IndexOf(table.GetRealAffinityColumns(), columnIndex) >= 0; + + private static IReadOnlyList ToStoredRealForm(EmbeddedTable table, IReadOnlyList row) + { + SqlValue[]? converted = null; + foreach (var columnIndex in table.GetRealAffinityColumns()) + { + if (columnIndex >= row.Count) + continue; + var stored = ToStoredRealForm(row[columnIndex]); + if (stored.Kind == row[columnIndex].Kind) + continue; + converted ??= [.. row]; + converted[columnIndex] = stored; + } + + return converted ?? row; + } + + // Index keys take the stored form for plain REAL-affinity columns only: an expression term's + // value is whatever the expression produced, which SQLite records without column affinity. + private static void ConvertIndexKeyToStoredRealForm(EmbeddedTable table, EmbeddedIndex index, SqlValue[] values) + { + for (var position = 0; position < index.Columns.Count && position < values.Length; position++) + { + var term = index.Columns[position]; + if (!term.IsExpression && term.ColumnIndex >= 0 && IsRealAffinityColumn(table, term.ColumnIndex)) + values[position] = ToStoredRealForm(values[position]); + } + } + + // Records written before stored values took SQLite's form encode an integral REAL as an + // 8-byte float, so a stored record may differ from the rebuilt one in exactly that + // representation (in either direction) and still hold the same values. + private bool RecordsHoldSameValues(byte[] stored, byte[] rebuilt) + { + if (stored.AsSpan().SequenceEqual(rebuilt)) + return true; + + var storedValues = SqliteRecordCodec.Decode(stored, _textEncoding); + var rebuiltValues = SqliteRecordCodec.Decode(rebuilt, _textEncoding); + if (storedValues.Length != rebuiltValues.Length) + return false; + + for (var index = 0; index < storedValues.Length; index++) + { + var left = storedValues[index]; + var right = rebuiltValues[index]; + if ((left.Kind, right.Kind) is (SqlValueKind.Integer, SqlValueKind.Real) or (SqlValueKind.Real, SqlValueKind.Integer)) + { + var integer = left.Kind == SqlValueKind.Integer ? left : right; + var real = left.Kind == SqlValueKind.Real ? left : right; + if ((double)integer.AsInteger() != real.AsReal()) + return false; + continue; + } + + if (!SqliteRecordCodec.Encode([left], _textEncoding).AsSpan() + .SequenceEqual(SqliteRecordCodec.Encode([right], _textEncoding))) + { + return false; + } + } + + return true; + } + /// /// Validates and durably persists the full managed catalog as SQLite pages in /// a single atomic WAL transaction. Any unsupported schema or data is rejected @@ -3040,7 +3154,8 @@ public FileCatalogVersion Persist( bool forceFullRewrite = false, IReadOnlyDictionary? previousTables = null, IReadOnlyList<(string TableName, EmbeddedTable Table, EmbeddedIndex Index)>? targetedIndexRebuild = null, - uint maximumPageCount = SqlitePageLimits.DefaultMaximumPageCount) + uint maximumPageCount = SqlitePageLimits.DefaultMaximumPageCount, + bool unchangedRowsAreDurable = false) => WithGrowthCeiling( maximumPageCount, () => PersistCore( @@ -3053,7 +3168,8 @@ public FileCatalogVersion Persist( pragmaHeader, forceFullRewrite, previousTables, - targetedIndexRebuild: targetedIndexRebuild)); + targetedIndexRebuild: targetedIndexRebuild, + unchangedRowsAreDurable: unchangedRowsAreDurable)); /// /// Materializes an MVCC checkpoint into pager WAL pages without reclaiming @@ -3526,7 +3642,8 @@ private FileCatalogVersion PersistCore( IReadOnlyDictionary? previousTables = null, bool checkpointAfterCommit = true, SqliteDatabaseHeader? vacuumSourceHeader = null, - IReadOnlyList<(string TableName, EmbeddedTable Table, EmbeddedIndex Index)>? targetedIndexRebuild = null) + IReadOnlyList<(string TableName, EmbeddedTable Table, EmbeddedIndex Index)>? targetedIndexRebuild = null, + bool unchangedRowsAreDurable = false) { ThrowIfDisposed(); ThrowIfPostCommitMaintenanceFaulted(); @@ -3564,7 +3681,8 @@ private FileCatalogVersion PersistCore( triggers, virtualTables, previousTables, - checkpointAfterCommit)) + checkpointAfterCommit, + unchangedRowsAreDurable)) { _committedTables = tables; return CommittedCatalogVersion; @@ -4059,13 +4177,33 @@ internal void RetainOnly(ISet live) } } - /// The number of changed rows above which a complete rewrite is preferred. + /// + /// The minimum number of changed rows the incremental path accepts before a complete rewrite + /// is preferred; larger databases scale it (see ). + /// /// - /// A bulk change touches most of the database anyway, and one rewrite packs - /// its pages far more densely than a long sequence of incremental splits. + /// A change touching a large share of the database is cheaper as one rewrite, which also packs + /// its pages far more densely than a long sequence of incremental splits. But the rewrite + /// costs the whole file, so a fixed budget made a modest batch (a 1,000-row import into one + /// table of a large database) rewrite every table and index on commit. /// private const int MaximumIncrementalChangedRows = 256; + /// A quarter of the catalog's rows, but never less than the fixed minimum. + private static int GetIncrementalChangedRowBudget(IReadOnlyDictionary previousTables) + { + long rows = 0; + foreach (var table in previousTables.Values) + { + // A table whose rows were never loaded still counts toward the database's size, + // but reading it here would force the load; it is simply left out of the estimate. + if (!table.HasPendingRowLoad) + rows += table.Rows.Count; + } + + return (int)Math.Clamp(rows / 4, MaximumIncrementalChangedRows, int.MaxValue); + } + /// /// Declares that is content-identical to what this /// store last made durable, so later writes may compute their page delta @@ -4128,7 +4266,8 @@ private bool TryPersistIncrementalRowMutation( IReadOnlyDictionary triggers, IReadOnlyDictionary virtualTables, IReadOnlyDictionary previousTables, - bool checkpointAfterCommit) + bool checkpointAfterCommit, + bool unchangedRowsAreDurable = false) { if (!HasCurrentSchemaShape(tables, views, triggers, virtualTables, previousTables)) return false; @@ -4147,9 +4286,17 @@ private bool TryPersistIncrementalRowMutation( // Incremental allocation prefers freelist leaves/trunks before appending. // A non-empty freelist is therefore safe here and no longer forces a full // rewrite solely to avoid stranding free pages. - if (!TryCollectRowDeltas(tables, previousTables, out var deltas) || deltas.Count == 0) + if (!TryCollectRowDeltas(tables, previousTables, out var deltas)) return false; + // Every row equals the committed catalog's (an UPDATE that rewrote values with themselves, + // which ORMs issue routinely). When the caller vouches that the committed catalog is the + // file's content, there is nothing to write; otherwise (an MVCC checkpoint materializing + // logical-log commits) the full write path below must run. Declining sent every such + // statement through a complete catalog rewrite. + if (deltas.Count == 0) + return unchangedRowsAreDurable; + var tableNames = tables.Keys.OrderBy(name => name, StringComparer.OrdinalIgnoreCase).ToArray(); var indexesByTable = GetIndexDefinitions(tableNames, tables, views, triggers, virtualTables, previousTables) .GroupBy(definition => definition.TableName, StringComparer.OrdinalIgnoreCase) @@ -4460,11 +4607,21 @@ private bool ApplyIncrementalTableDelta( if (change.Before is not null && change.After is not null) tableTree.Update(rootPage, change.RowId, BuildTableRecord(table, change.After)); } + // Inserts go in as one ascending batch, so rows sharing a leaf (appends especially) are + // merged into a single write of that leaf instead of one rewrite per row. + var inserts = new List<(long RowId, byte[] Record)>(); foreach (var change in delta.Changes) { if (change.Before is null && change.After is not null) - tableTree.Insert(rootPage, change.RowId, BuildTableRecord(table, change.After)); + inserts.Add((change.RowId, BuildTableRecord(table, change.After))); } + inserts.Sort(static (left, right) => left.RowId.CompareTo(right.RowId)); + for (var index = 1; index < inserts.Count; index++) + { + if (inserts[index].RowId == inserts[index - 1].RowId) + throw new SqliteBtreeMaintenanceRequiredException("A row delta inserts the same rowid twice."); + } + tableTree.InsertMany(rootPage, inserts); foreach (var (plan, record) in indexInserts) plan.Tree.Insert(plan.RootPage, record); @@ -4485,7 +4642,7 @@ private static bool TryCollectRowDeltas( if (tables.Count != previousTables.Count) return false; - var changedRowBudget = MaximumIncrementalChangedRows; + var changedRowBudget = GetIncrementalChangedRowBudget(previousTables); foreach (var (name, table) in tables) { if (!previousTables.TryGetValue(name, out var previous)) @@ -4502,6 +4659,14 @@ private static bool TryCollectRowDeltas( return false; } + if (TryCollectAlignedRowChanges(table, previous, changedRowBudget, out var alignedChanges) + && alignedChanges.Count > 0) + { + changedRowBudget -= alignedChanges.Count; + deltas.Add(new TableRowDelta(name, table, previous, alignedChanges)); + continue; + } + var before = new Dictionary(previous.Rows.Count); for (var index = 0; index < previous.Rows.Count; index++) { @@ -4576,6 +4741,7 @@ private static bool TryCollectRowDeltas( var values = new SqlValue[index.Columns.Count + 1]; Array.Copy(key, values, key.Length); values[^1] = SqlValue.Integer(rowId); + ConvertIndexKeyToStoredRealForm(table, index, values); var record = SqliteRecordCodec.Encode(values, _textEncoding); comparer.Validate(record); return record; @@ -9865,15 +10031,72 @@ private static bool HaveSameRows(EmbeddedTable left, EmbeddedTable right) if (left.Rows.Count != right.Rows.Count || left.RowIds.Count != right.RowIds.Count) return false; - for (var rowIndex = 0; rowIndex < left.Rows.Count; rowIndex++) - { - if (left.RowIds[rowIndex] != right.RowIds[rowIndex] - || !left.Rows[rowIndex].AsSpan().SequenceEqual(right.Rows[rowIndex])) + return TryCollectAlignedRowChanges(left, right, changedRowBudget: 0, out _); + } + + /// + /// Diffs two versions of a table whose previous rows all still sit at the same positions with + /// the same rowids, which is how in-place UPDATEs and appending INSERTs leave them; rows past + /// the previous count are inserts (the aligned prefix holds every previous rowid, so theirs + /// are new). Copy-on-write chunks both versions still share are skipped outright, and rows + /// are compared by reference before by value, so a commit that touched a few rows reads only + /// their chunks instead of the whole table. Returns false when the versions are not aligned (a + /// rowid differs at some position, or rows were removed) or more than + /// rows changed. + /// + private static bool TryCollectAlignedRowChanges( + EmbeddedTable table, + EmbeddedTable previous, + int changedRowBudget, + out List changes) + { + changes = []; + var rows = table.Rows; + var previousRows = previous.Rows; + var rowIds = table.RowIds; + var previousRowIds = previous.RowIds; + var count = rows.Count; + var previousCount = previousRows.Count; + if (count != rowIds.Count || previousCount != previousRowIds.Count || count < previousCount) + return false; + + // A chunk is skipped only when both versions share it and it lies wholly inside the + // previous count, where shared storage proves equal elements. + const int chunkSize = CowChunkedList.ChunkSize; + for (var chunk = 0; chunk * chunkSize < previousCount; chunk++) + { + var end = Math.Min(previousCount, (chunk + 1) * chunkSize); + if (end == (chunk + 1) * chunkSize + && rows.SharesChunkWith(previousRows, chunk) + && rowIds.SharesChunkWith(previousRowIds, chunk)) { - return false; + continue; + } + + for (var index = chunk * chunkSize; index < end; index++) + { + var rowId = rowIds[index]; + if (rowId != previousRowIds[index]) + return false; + + var row = rows[index]; + var previousRow = previousRows[index]; + if (ReferenceEquals(row, previousRow) || previousRow.AsSpan().SequenceEqual(row)) + continue; + + changes.Add(new RowChange(rowId, previousRow, row)); + if (changes.Count > changedRowBudget) + return false; } } + for (var index = previousCount; index < count; index++) + { + changes.Add(new RowChange(rowIds[index], null, rows[index])); + if (changes.Count > changedRowBudget) + return false; + } + return true; } @@ -10678,7 +10901,11 @@ private void ValidateTableRepresentable( // storage, even though nothing about it changed. if (primaryKeyCount == 1 && table.RowidAliasColumnIndex >= 0 - && !IsTableRowStorageUnchangedFromPrevious(name, table, previousTables, out _)) + && !IsTableRowStorageUnchangedFromPrevious(name, table, previousTables, out _) + && !RowidAliasValuesAreTheIncreasingRowIds( + table, + primaryKeyIndex, + previousTables is not null && previousTables.TryGetValue(name, out var previousTable) ? previousTable : null)) { var seen = new HashSet(); foreach (var row in table.Rows) @@ -10706,6 +10933,66 @@ private void ValidateTableRepresentable( } } + // The common shape proves the rowid-alias check without hashing every row: rows kept in + // strictly increasing rowid order whose alias column holds exactly the row's rowid are distinct + // integers. Anything else falls back to the full check above. A proof is remembered per exact + // row and rowid contents, so the next commit re-checks only the copy-on-write chunks it no + // longer shares with the proven previous version (plus each chunk's first ordering step). + private static bool RowidAliasValuesAreTheIncreasingRowIds( + EmbeddedTable table, + int aliasIndex, + EmbeddedTable? previous) + { + var proofKey = new IncreasingRowidAliasProofKey(aliasIndex); + if (table.TryGetDerived(proofKey, out _)) + return true; + + var rows = table.Rows; + var rowIds = table.RowIds; + if (rowIds.Count != rows.Count) + return false; + + var previousCount = previous is not null + && !ReferenceEquals(previous, table) + && ReferenceEquals(previous.ColumnDefinitions, table.ColumnDefinitions) + && previous.Rows.Count == previous.RowIds.Count + && previous.Rows.Count <= rows.Count + && previous.TryGetDerived(proofKey, out _) + ? previous.Rows.Count + : 0; + const int chunkSize = CowChunkedList.ChunkSize; + for (var chunkStart = 0; chunkStart < rows.Count; chunkStart += chunkSize) + { + var chunk = chunkStart / chunkSize; + var end = Math.Min(rows.Count, chunkStart + chunkSize); + var shared = chunkStart + chunkSize <= previousCount + && rows.SharesChunkWith(previous!.Rows, chunk) + && rowIds.SharesChunkWith(previous.RowIds, chunk); + for (var index = chunkStart; index < end; index++) + { + var rowId = rowIds[index]; + if (index > 0 && rowId <= rowIds[index - 1]) + return false; + if (shared) + break; + + var value = rows[index][aliasIndex]; + if (value.Kind != SqlValueKind.Integer || value.AsInteger() != rowId) + return false; + } + } + + table.GetOrCreateDerived(proofKey, static () => IncreasingRowidAliasProof.Instance); + return true; + } + + private readonly record struct IncreasingRowidAliasProofKey(int AliasIndex); + + private sealed class IncreasingRowidAliasProof + { + public static readonly IncreasingRowidAliasProof Instance = new(); + } + private static void ValidatePrimaryKeyIndexPrerequisites( string tableName, EmbeddedTable table, @@ -12568,7 +12855,7 @@ private SqliteIndexLeafCell CreateIndexLeafCell( /// private byte[] BuildTableRecord(EmbeddedTable table, IReadOnlyList row) { - var record = ProjectStoredRow(table, row); + var record = ProjectStoredRow(table, ToStoredRealForm(table, row)); var aliasIndex = table.RowidAliasColumnIndex; if (aliasIndex >= 0) { @@ -12638,7 +12925,7 @@ private IReadOnlyList BuildWithoutRowidTableRecords( } var record = SqliteRecordCodec.Encode( - OrderWithoutRowidRecord(tableName, table, primaryKeySchema, row), + OrderWithoutRowidRecord(tableName, table, primaryKeySchema, ToStoredRealForm(table, row)), _textEncoding); comparer.Validate(record); records.Add(new WithoutRowidRecord(record, key)); @@ -12761,6 +13048,7 @@ internal static SqlValue[] RestoreWithoutRowidRecord( if (source < storedValues.Count) throw new InvalidDataException($"Stored WITHOUT ROWID table '{tableName}' record has trailing values."); + ApplyStoredRealAffinity(table, row); return row; } @@ -12812,7 +13100,12 @@ private IReadOnlyList BuildIndexRecords( values = new SqlValue[storageColumns!.Count]; Array.Copy(key, values, key.Length); for (var column = index.Columns.Count; column < storageColumns.Count; column++) - values[column] = row[storageColumns[column].ColumnIndex]; + { + var storageColumn = storageColumns[column].ColumnIndex; + values[column] = IsRealAffinityColumn(table, storageColumn) + ? ToStoredRealForm(row[storageColumn]) + : row[storageColumn]; + } } else { @@ -12820,6 +13113,7 @@ private IReadOnlyList BuildIndexRecords( Array.Copy(key, values, key.Length); values[^1] = SqlValue.Integer(rowId!.Value); } + ConvertIndexKeyToStoredRealForm(table, index, values); var record = SqliteRecordCodec.Encode(values, _textEncoding); // `values` is already the decoded form of `record`: every SqlValue // factory normalises on construction (SqlValue.Real folds NaN to @@ -13498,7 +13792,7 @@ private void ValidateStoredIndex( for (var recordIndex = 0; recordIndex < expectedRecords.Count; recordIndex++) { - if (!actualRecords[recordIndex].AsSpan().SequenceEqual(expectedRecords[recordIndex])) + if (!RecordsHoldSameValues(actualRecords[recordIndex], expectedRecords[recordIndex])) { throw new EmbeddedSqlException( $"Stored index '{entry.Name}' does not match table '{entry.TableName}' at record {recordIndex}."); diff --git a/src/Ahtola.Core/IndexSemantics.cs b/src/Ahtola.Core/IndexSemantics.cs index 158d439..b323d37 100644 --- a/src/Ahtola.Core/IndexSemantics.cs +++ b/src/Ahtola.Core/IndexSemantics.cs @@ -150,6 +150,14 @@ private static string FormatIdentifier(string identifier) internal static class IndexExpressionSemantics { + // Successful round-trip checks. The check is a pure function of the immutable index record and + // the table's name and column definitions (replaced, never mutated, by schema changes), and + // every statement's catalog clone shares both instances, so persisting a write no longer + // regenerates and re-parses every CREATE INDEX statement of the table. + private static readonly System.Runtime.CompilerServices.ConditionalWeakTable s_roundTripProofs = new(); + + private sealed record RoundTripProof(string TableName, EmbeddedColumn[] Columns); + public static void ValidateRoundTrip( string tableName, EmbeddedTable table, @@ -158,6 +166,22 @@ public static void ValidateRoundTrip( if (index.Origin != EmbeddedIndexOrigin.Explicit) return; + if (s_roundTripProofs.TryGetValue(index, out var proof) + && ReferenceEquals(proof.Columns, table.ColumnDefinitions) + && string.Equals(proof.TableName, tableName, StringComparison.Ordinal)) + { + return; + } + + ValidateRoundTripCore(tableName, table, index); + s_roundTripProofs.AddOrUpdate(index, new RoundTripProof(tableName, table.ColumnDefinitions)); + } + + private static void ValidateRoundTripCore( + string tableName, + EmbeddedTable table, + EmbeddedIndex index) + { var sql = IndexSqlFormatter.BuildCreateIndexSql(tableName, index); if (SqlParser.Parse(sql, SqlParameterMap.Parse(sql)) is not CreateIndexStatement statement) throw new EmbeddedSqlException($"Index '{index.Name}' cannot be reconstructed."); diff --git a/src/Ahtola.Core/ManagedLocalAdapters.cs b/src/Ahtola.Core/ManagedLocalAdapters.cs index 5d76b2c..20dbd00 100644 --- a/src/Ahtola.Core/ManagedLocalAdapters.cs +++ b/src/Ahtola.Core/ManagedLocalAdapters.cs @@ -181,6 +181,14 @@ ValueTask PrepareAsync( bool HasAttachedDatabases => true; + /// + /// An opaque value that compares equal () for as long as the + /// columns, declared types and indexes of the base table an unqualified name resolves to stay + /// the same, letting callers reuse metadata read through PRAGMA table_info and + /// index_list. Null when the adapter cannot vouch for it, which disables such reuse. + /// + object? GetTableSchemaIdentity(string tableName) => null; + /// /// How long contended transaction-lock acquisitions wait before reporting busy, /// mirroring sqlite3_busy_timeout. Adapters without a managed transaction @@ -210,6 +218,12 @@ bool ExperimentalCustomTypesEnabled void ResetForPooling() => throw new NotSupportedException("This managed connection adapter does not support pooling."); + /// + /// Resets the connection as it returns to a pool. Renting it calls + /// again, so adapters may skip work whose result only matters to the next user. + /// + void ResetForPoolReturn() => ResetForPooling(); + IManagedIncrementalBlobAdapter OpenBlob( string databaseName, string tableName, @@ -582,6 +596,9 @@ public static ManagedConnectionAdapter Wrap(EmbeddedConnection connection) public ManagedConnectionHooks Hooks => GetConnection().Hooks; + public object? GetTableSchemaIdentity(string tableName) + => GetConnection().TryGetTableSchemaIdentity(tableName); + public TimeSpan BusyTimeout { get => GetConnection().BusyTimeout; @@ -619,6 +636,11 @@ public void ResetForPooling() GetConnection().ResetForPooling(); } + public void ResetForPoolReturn() + { + GetConnection().ResetForPooling(adoptCommittedChanges: false); + } + public IManagedIncrementalBlobAdapter OpenBlob( string databaseName, string tableName, diff --git a/src/Ahtola.Core/RowIdAllocationSet.cs b/src/Ahtola.Core/RowIdAllocationSet.cs index 5a37284..a6e801a 100644 --- a/src/Ahtola.Core/RowIdAllocationSet.cs +++ b/src/Ahtola.Core/RowIdAllocationSet.cs @@ -4,11 +4,10 @@ namespace Ahtola.Core; /// /// The existing rowids of a table as an immutable set plus their maximum, valid for exactly one -/// row-store state (see ). +/// rowid-list state (see ). /// internal sealed record RowIdSet( - long LineageId, - long Revision, + long RowIdsStamp, int Count, ImmutableHashSet Ids, long Max); diff --git a/src/Ahtola.Core/Storage/IFileSystem.cs b/src/Ahtola.Core/Storage/IFileSystem.cs index 994bc45..9225811 100644 --- a/src/Ahtola.Core/Storage/IFileSystem.cs +++ b/src/Ahtola.Core/Storage/IFileSystem.cs @@ -44,6 +44,17 @@ public enum FileSystemOperation /// public readonly record struct FileWriteStamp(long Length, DateTimeOffset LastWriteTimeUtc); +/// +/// Optional capability of an open file that can report its from the +/// handle itself. The pager probes its database and WAL stamps on every statement, and a +/// path-based query opens the file by name each time, which file-system filters make cost +/// tens of microseconds on Windows; the handle-based query reads the same metadata. +/// +internal interface IFileWriteStampSource +{ + FileWriteStamp GetWriteStamp(); +} + /// /// Optional capability that gives a storage backend authority over canonical /// path identity. Host file systems can return absolute paths while browser or diff --git a/src/Ahtola.Core/Storage/PhysicalFileSystem.cs b/src/Ahtola.Core/Storage/PhysicalFileSystem.cs index 69c8550..7e28ec2 100644 --- a/src/Ahtola.Core/Storage/PhysicalFileSystem.cs +++ b/src/Ahtola.Core/Storage/PhysicalFileSystem.cs @@ -233,7 +233,7 @@ void IAtomicFileSystem.ReplaceFileAtomically( /// /// A host file handle exposing positional I/O over . /// -public sealed class PhysicalFile : IFile +public sealed class PhysicalFile : IFile, IFileWriteStampSource { private readonly SafeFileHandle _handle; private readonly string? _pagerLockPath; @@ -257,6 +257,12 @@ public long Length } } + FileWriteStamp IFileWriteStampSource.GetWriteStamp() + { + ThrowIfDisposed(); + return new FileWriteStamp(RandomAccess.GetLength(_handle), File.GetLastWriteTimeUtc(_handle)); + } + public int Read(long position, Span destination) { ThrowIfDisposed(); diff --git a/src/Ahtola.Core/Storage/PhysicalSqliteWalSharedMemoryMapping.cs b/src/Ahtola.Core/Storage/PhysicalSqliteWalSharedMemoryMapping.cs index 4edbf05..d17ce18 100644 --- a/src/Ahtola.Core/Storage/PhysicalSqliteWalSharedMemoryMapping.cs +++ b/src/Ahtola.Core/Storage/PhysicalSqliteWalSharedMemoryMapping.cs @@ -254,7 +254,12 @@ public void Read(long position, Span destination) lock (_gate) { ThrowIfDisposed(); - SynchronizeLengthLocked(); + // Only a read past the current view needs the file's length: the -shm only ever + // grows while this carrier holds its dead-man-switch lock (SQLite truncates it only + // under the exclusive DMS lock), so a range inside the view is always backed. Readers + // probe the header on every page access, and a length query is a syscall each time. + if (!IsWithinMappedViewLocked(position, destination.Length)) + SynchronizeLengthLocked(); ValidateRange(position, destination.Length); if (destination.IsEmpty) return; @@ -443,6 +448,26 @@ internal static void ThrowIfPlatformUnsupported() "Physical SQLite shared-memory mappings are supported only on Windows, 64-bit Linux, and macOS."); } + /// + /// Whether the mapping already covers bytes, querying the + /// file length only when the current view is shorter (see ). + /// + internal bool Covers(long requiredLength) + { + lock (_gate) + { + ThrowIfDisposed(); + if (_view is not null && requiredLength <= _length) + return true; + + SynchronizeLengthLocked(); + return requiredLength <= _length; + } + } + + private bool IsWithinMappedViewLocked(long position, int byteCount) + => _view is not null && position <= _length && byteCount <= _length - position; + private void SynchronizeLengthLocked() { var fileLength = RandomAccess.GetLength(_fileHandle); diff --git a/src/Ahtola.Core/Storage/SqliteIncrementalTableBtree.cs b/src/Ahtola.Core/Storage/SqliteIncrementalTableBtree.cs index b34dbb0..4186fe0 100644 --- a/src/Ahtola.Core/Storage/SqliteIncrementalTableBtree.cs +++ b/src/Ahtola.Core/Storage/SqliteIncrementalTableBtree.cs @@ -69,6 +69,76 @@ public void Insert(uint rootPage, long rowId, ReadOnlySpan record) WriteLeafAndPropagate(path, cells, appendedAtEnd: search.Index == cells.Count - 1); } + /// + /// Inserts , in strictly ascending rowid order, at rowids the tree must + /// not already contain. + /// + /// + /// Equivalent to calling for each row, but every row that lands in the + /// same leaf is merged into one write of that leaf (split as needed), instead of re-reading, + /// re-parsing and rewriting the leaf once per row. A leaf reached for one row receives every + /// following row up to the smallest separator on its path: separators are inclusive upper + /// bounds of their left subtrees, and earlier rows already bound the range from below. + /// + public void InsertMany(uint rootPage, IReadOnlyList<(long RowId, byte[] Record)> rows) + { + ArgumentNullException.ThrowIfNull(rows); + var next = 0; + while (next < rows.Count) + { + var path = Descend(rootPage, rows[next].RowId); + var bound = long.MaxValue; + for (var level = 0; level < path.Count - 1; level++) + { + var interior = ParseInterior(path[level].PageNumber); + var childIndex = path[level].ChildIndex; + if (childIndex < interior.Cells.Count) + bound = Math.Min(bound, interior.Cells[childIndex].Cell.RowId); + } + + var view = ParseLeaf(path[^1].PageNumber); + var cells = view.Cells.Select(cell => cell.Cell).ToList(); + var appendedAtEnd = true; + do + { + var (rowId, record) = rows[next]; + if (next > 0 && rowId <= rows[next - 1].RowId) + throw new ArgumentException("Rows must be in strictly ascending rowid order.", nameof(rows)); + + var index = FindCellIndex(cells, rowId, out var exact); + if (exact) + { + throw new SqliteBtreeMaintenanceRequiredException( + $"Rowid {rowId} already exists, so the caller's view of the committed table is stale."); + } + + appendedAtEnd &= index == cells.Count; + cells.Insert(index, CreateLeafCell(rowId, record)); + next++; + } + while (next < rows.Count && rows[next].RowId <= bound); + + WriteLeafAndPropagate(path, cells, appendedAtEnd); + } + } + + private static int FindCellIndex(List cells, long rowId, out bool exact) + { + var low = 0; + var high = cells.Count; + while (low < high) + { + var middle = low + ((high - low) / 2); + if (cells[middle].RowId < rowId) + low = middle + 1; + else + high = middle; + } + + exact = low < cells.Count && cells[low].RowId == rowId; + return low; + } + /// /// Replaces the record stored at a the tree must /// already contain. diff --git a/src/Ahtola.Core/Storage/SqlitePageStore.cs b/src/Ahtola.Core/Storage/SqlitePageStore.cs index d5ba23b..07998b5 100644 --- a/src/Ahtola.Core/Storage/SqlitePageStore.cs +++ b/src/Ahtola.Core/Storage/SqlitePageStore.cs @@ -407,6 +407,10 @@ internal void EnsurePageMaterialized(uint pageNumber) internal bool SupportsPageMaterialization => _file is IPageMaterializingFile; + /// The open file's write stamp, when its handle can report one. + internal FileWriteStamp? TryGetHandleWriteStamp() + => _file is IFileWriteStampSource source ? source.GetWriteStamp() : null; + internal void RefreshHeader() { ThrowIfDisposed(); diff --git a/src/Ahtola.Core/Storage/SqlitePager.cs b/src/Ahtola.Core/Storage/SqlitePager.cs index 183da95..3d233a2 100644 --- a/src/Ahtola.Core/Storage/SqlitePager.cs +++ b/src/Ahtola.Core/Storage/SqlitePager.cs @@ -72,9 +72,11 @@ public sealed class SqlitePager : IDisposable { /// /// Default maximum number of clean main-database page images retained by one - /// pager instance. + /// pager instance: SQLite's default cache_size=-2000 (2000 KiB) in 4 KiB pages. + /// A smaller cache made every point query over a few-MiB database miss and re-read + /// (and re-stat) the main file for its interior and leaf pages. /// - public const int DefaultPageCacheCapacity = 64; + public const int DefaultPageCacheCapacity = 500; private readonly object _gate = new(); private readonly IFileSystem _fileSystem; @@ -110,6 +112,7 @@ public sealed class SqlitePager : IDisposable private uint _observedWalIndexSalt1; private uint _observedWalIndexSalt2; private bool _hasObservedWalStamp; + private bool _walStampVerifiedBySynchronization; private FileWriteStamp? _observedWalStamp; private bool _exclusiveLockingMode; @@ -3394,8 +3397,54 @@ private SqlitePagerViewToken CaptureCommittedViewTokenUnderLock() _wal?.Header.Salt1 ?? 0, _wal?.Header.Salt2 ?? 0, _committedPageCount, - _fileSystem.GetWriteStamp(_databasePath), - _fileSystem.GetWriteStamp(_walPath)); + GetDatabaseWriteStamp(), + ConsumeSynchronizedWalStamp()); + } + + // The stamps come from the pager's open handles when the file system supports it (see + // IFileWriteStampSource): the same length and last-write metadata without a path-based open. + // A peer cannot replace either file while this pager holds it open (the WAL is deleted only by + // its last connection), so the handle identifies the file the path names. Every WAL stamp read + // goes through GetWalWriteStamp, so observed and probed stamps always compare like with like. + private FileWriteStamp? GetDatabaseWriteStamp() + { + try + { + if (_pageStore.TryGetHandleWriteStamp() is { } stamp) + return stamp; + } + catch (ObjectDisposedException) + { + } + + return _fileSystem.GetWriteStamp(_databasePath); + } + + private FileWriteStamp? GetWalWriteStamp() + { + try + { + if (_wal?.TryGetHandleWriteStamp() is { } stamp) + return stamp; + } + catch (ObjectDisposedException) + { + } + + return _fileSystem.GetWriteStamp(_walPath); + } + + // The -wal stamp SynchronizeCommittedView just proved unchanged under the same locks, so a + // token captured right after it reuses that probe instead of querying the file again. + private FileWriteStamp? ConsumeSynchronizedWalStamp() + { + if (_walStampVerifiedBySynchronization) + { + _walStampVerifiedBySynchronization = false; + return _observedWalStamp; + } + + return GetWalWriteStamp(); } /// @@ -3410,6 +3459,7 @@ private SqlitePagerViewToken CaptureCommittedViewTokenUnderLock() private void SynchronizeCommittedView(SqliteMainFileLockLease? mainFileLock = null) { + _walStampVerifiedBySynchronization = false; try { if (_journalMode == SqliteJournalMode.Delete @@ -3433,7 +3483,12 @@ private void SynchronizeCommittedView(SqliteMainFileLockLease? mainFileLock = nu && !RequiresSharedStorageRescan && !walIndexChanged && !peerWalStampChanged) + { + _walStampVerifiedBySynchronization = UsesWalStorage(_journalMode) + && _wal is not null + && _hasObservedWalStamp; return; + } CommittedViewRescanCount++; if (_fileSystem.FileExists(_journalPath) @@ -3595,7 +3650,7 @@ private void ClearObservedWalIndexIdentity() /// private bool TryDetectPeerWalStampChange() { - var stamp = _fileSystem.GetWriteStamp(_walPath); + var stamp = GetWalWriteStamp(); if (!_hasObservedWalStamp) { ObserveWalStamp(stamp); @@ -3606,7 +3661,7 @@ private bool TryDetectPeerWalStampChange() } private void ObserveCurrentWalStamp() - => ObserveWalStamp(_fileSystem.GetWriteStamp(_walPath)); + => ObserveWalStamp(GetWalWriteStamp()); private void ObserveWalStamp(FileWriteStamp? stamp) { diff --git a/src/Ahtola.Core/Storage/SqliteTableBtreeCursor.cs b/src/Ahtola.Core/Storage/SqliteTableBtreeCursor.cs index 35f3d95..64fbee8 100644 --- a/src/Ahtola.Core/Storage/SqliteTableBtreeCursor.cs +++ b/src/Ahtola.Core/Storage/SqliteTableBtreeCursor.cs @@ -13,9 +13,16 @@ namespace Ahtola.Core.Storage; public sealed class SqliteTableBtreeCursor { private const int MaximumDepth = 64; + private const int MaximumCachedInteriorPages = 16; private readonly ISqliteBtreePageIo _io; + // Validated interior views from earlier seeks. Every seek descends through the same root and + // upper interior pages, and a full parse re-validates every cell, freeblock and child; a view + // is reused only while the page image is byte-for-byte identical to the copy it was parsed + // from, so writes through any layer simply miss. + private Dictionary? _interiorViews; + /// Creates a cursor over one page-access boundary. public SqliteTableBtreeCursor(ISqliteBtreePageIo pageIo) { @@ -345,7 +352,7 @@ private bool TrySeekLeaf( case SqliteBtreePageType.TableInterior: { - var interior = SqliteTableInteriorPageView.Parse(image, _io.UsableSpace, isFirstPage); + var interior = GetInteriorView(pageNumber, image, isFirstPage); pageNumber = interior.SearchChild(rowId).ChildPage; break; } @@ -359,4 +366,21 @@ private bool TrySeekLeaf( throw new InvalidDataException( $"SQLite table b-tree rooted at page {rootPage} is deeper than {MaximumDepth} levels."); } + + private SqliteTableInteriorPageView GetInteriorView(uint pageNumber, byte[] image, bool isFirstPage) + { + if (_interiorViews is not null + && _interiorViews.TryGetValue(pageNumber, out var cached) + && image.AsSpan().SequenceEqual(cached.Image)) + { + return cached.View; + } + + var view = SqliteTableInteriorPageView.Parse(image, _io.UsableSpace, isFirstPage); + _interiorViews ??= []; + if (_interiorViews.Count >= MaximumCachedInteriorPages && !_interiorViews.ContainsKey(pageNumber)) + _interiorViews.Clear(); + _interiorViews[pageNumber] = (image.ToArray(), view); + return view; + } } diff --git a/src/Ahtola.Core/Storage/SqliteWal.cs b/src/Ahtola.Core/Storage/SqliteWal.cs index a143343..f5505a8 100644 --- a/src/Ahtola.Core/Storage/SqliteWal.cs +++ b/src/Ahtola.Core/Storage/SqliteWal.cs @@ -454,6 +454,29 @@ public sealed class SqliteWalFile : IDisposable, IAsyncDisposable /// private ScanPrefix? _scanPrefix; + // The wal-index header last proven to describe this WAL's committed prefix, with the validated + // prefix it was proven against (see IsIndexHeaderValidated). + private (SqliteWalIndexHeader Header, ScanPrefix Prefix)? _validatedIndexHeader; + + /// + /// Whether was already validated against this WAL and nothing has + /// invalidated that since: the recorded validated prefix is still the current one (every rescan + /// that finds a new commit, and every reset or truncation, replaces or clears it). Frames up to + /// the header's boundary are immutable while it stands; a WAL restart changes the salts and so + /// the header itself. + /// + internal bool IsIndexHeaderValidated(SqliteWalIndexHeader header) + => _validatedIndexHeader is { } validated + && _scanPrefix is { } prefix + && ReferenceEquals(validated.Prefix, prefix) + && validated.Header.Equals(header); + + internal void RecordIndexHeaderValidated(SqliteWalIndexHeader header) + { + if (_scanPrefix is { } prefix && prefix.CommittedFrameNumber == header.MaximumFrame) + _validatedIndexHeader = (header, prefix); + } + private sealed record ScanPrefix( SqliteWalHeader Seed, byte[] WalHeaderBytes, @@ -1213,6 +1236,9 @@ public SqliteWalFrame ReadFrame(long frameNumber) $"Frame number is out of range for {fullFrameCount} complete SQLite WAL frame(s)."); } + if (TryReadFrameInValidatedPrefix(frameNumber, fullFrameCount) is { } trusted) + return trusted; + var previousChecksum = (Header.Checksum1, Header.Checksum2); for (var currentFrameNumber = 1L; currentFrameNumber <= frameNumber; currentFrameNumber++) { @@ -1246,6 +1272,42 @@ public SqliteWalFrame ReadFrame(long frameNumber) throw new InvalidOperationException("SQLite WAL frame traversal ended unexpectedly."); } + /// + /// Reads a frame inside the recorded validated prefix (re-proven against the on-disk WAL + /// header and boundary frame header, as ScanCore does) without re-walking the chain: within + /// that prefix each frame's stored checksum is its chain value, so the frame is validated + /// against its predecessor's. Opening a read snapshot reads its boundary frame, and walking + /// from frame one made that cost every committed frame since the last checkpoint. + /// + private SqliteWalFrame? TryReadFrameInValidatedPrefix(long frameNumber, long fullFrameCount) + { + var file = GetSyncFile(); + var walHeaderBytes = new byte[SqliteWalHeader.Size]; + if (file.Read(0, walHeaderBytes) != walHeaderBytes.Length + || TryResumeScanPrefix(file, fullFrameCount, walHeaderBytes) is not { } prefix + || prefix.CommittedFrameNumber < frameNumber) + { + return null; + } + + var previousChecksum = (Header.Checksum1, Header.Checksum2); + if (frameNumber > 1) + { + Span previousHeaderBytes = stackalloc byte[SqliteWalFrameHeader.Size]; + if (file.Read(FrameOffset(frameNumber - 1), previousHeaderBytes) != previousHeaderBytes.Length) + return null; + + var previousHeader = SqliteWalFrameHeader.Parse(previousHeaderBytes); + if (previousHeader.Salt1 != Header.Salt1 || previousHeader.Salt2 != Header.Salt2) + return null; + previousChecksum = (previousHeader.Checksum1, previousHeader.Checksum2); + } + + var frame = ReadFrameBytes(FrameOffset(frameNumber)); + var frameHeader = ValidateFrame(frame, previousChecksum, out _); + return DecodeFrame(frame, frameHeader); + } + /// /// Asynchronously reads one frame after validating every checksum in its /// chain from the WAL header through that frame. @@ -1338,11 +1400,34 @@ public IReadOnlyList ReadFrameRange(long firstFrameNumber, long var frameSize = checked((int)FrameSize); var results = new List(checked((int)(lastFrameNumber - firstFrameNumber + 1))); var previousChecksum = (Header.Checksum1, Header.Checksum2); + + // A commit publishes the frames it just appended, right after the recovery scan that + // validated them, so the chain walk can start at the first requested frame whenever the + // recorded validated prefix (re-proven against the on-disk WAL header and boundary frame + // header, as ScanCore does) reaches the frame before it. Walking from frame one made every + // commit re-read and re-checksum the whole WAL, which now holds up to the checkpoint + // threshold of committed frames. + var startFrameNumber = 1L; + var trustedThrough = 0L; + if (firstFrameNumber > 1) + { + var walHeaderBytes = new byte[SqliteWalHeader.Size]; + if (GetSyncFile().Read(0, walHeaderBytes) == walHeaderBytes.Length + && TryResumeScanPrefix(GetSyncFile(), fullFrameCount, walHeaderBytes) is { } prefix + && prefix.CommittedFrameNumber >= firstFrameNumber - 1) + { + startFrameNumber = firstFrameNumber; + trustedThrough = prefix.CommittedFrameNumber; + if (trustedThrough == firstFrameNumber - 1) + previousChecksum = prefix.ChecksumAfterCommit; + } + } + var rented = ArrayPool.Shared.Rent(frameSize); try { var frame = rented.AsSpan(0, frameSize); - for (var frameNumber = 1L; frameNumber <= lastFrameNumber; frameNumber++) + for (var frameNumber = startFrameNumber; frameNumber <= lastFrameNumber; frameNumber++) { var read = GetSyncFile().Read(FrameOffset(frameNumber), frame); if (read != frame.Length) @@ -1351,8 +1436,21 @@ public IReadOnlyList ReadFrameRange(long firstFrameNumber, long $"Short read on SQLite WAL frame: expected {frame.Length} bytes, got {read} bytes."); } - var frameHeader = ValidateFrame(frame, previousChecksum, out var checksum); - previousChecksum = checksum; + SqliteWalFrameHeader frameHeader; + if (frameNumber <= trustedThrough) + { + // Inside the validated prefix the stored checksum is the chain value. + frameHeader = SqliteWalFrameHeader.Parse(frame[..SqliteWalFrameHeader.Size]); + if (frameHeader.Salt1 != Header.Salt1 || frameHeader.Salt2 != Header.Salt2) + throw new InvalidDataException("SQLite WAL frame salts do not match the WAL header."); + previousChecksum = (frameHeader.Checksum1, frameHeader.Checksum2); + } + else + { + frameHeader = ValidateFrame(frame, previousChecksum, out var checksum); + previousChecksum = checksum; + } + if (frameNumber < firstFrameNumber) continue; @@ -2314,6 +2412,10 @@ private void ThrowIfReadOnly() private void ThrowIfDisposed() => ObjectDisposedException.ThrowIf(_disposed, this); + /// The open WAL file's write stamp, when its handle can report one. + internal FileWriteStamp? TryGetHandleWriteStamp() + => _file is IFileWriteStampSource source ? source.GetWriteStamp() : null; + private IFile GetSyncFile() => _file ?? throw new InvalidOperationException( diff --git a/src/Ahtola.Core/Storage/SqliteWalIndex.cs b/src/Ahtola.Core/Storage/SqliteWalIndex.cs index 1cbaf06..14338d3 100644 --- a/src/Ahtola.Core/Storage/SqliteWalIndex.cs +++ b/src/Ahtola.Core/Storage/SqliteWalIndex.cs @@ -1405,7 +1405,9 @@ private void EnsureMappedBlocks(int blockCount) throw new ArgumentOutOfRangeException(nameof(blockCount), "SQLite WAL-index block count must be positive."); var requiredLength = checked((long)blockCount * SqliteWalIndexLayout.BlockSize); - if (_mapping.Length < requiredLength) + if (_mapping is PhysicalSqliteWalSharedMemoryMapping physical + ? !physical.Covers(requiredLength) + : _mapping.Length < requiredLength) { throw new InvalidDataException( $"SQLite WAL-index mapping is {_mapping.Length} bytes but requires at least {requiredLength} bytes."); @@ -1540,6 +1542,11 @@ private static void ValidateHeaderAgainstWal(SqliteWalIndexHeader header, Sqlite if (header.Salt1 != walHeader.Salt1 || header.Salt2 != walHeader.Salt2) throw new InvalidDataException("SQLite WAL-index salts do not match the WAL header."); + // A commit validates the same header several times (synchronization, read-mark and index + // publication); each pass rescanned the WAL tail and re-read the commit frame. + if (wal.IsIndexHeaderValidated(header)) + return; + var recovery = wal.ScanRecovery(); if (recovery.StopReason != SqliteWalRecoveryStopReason.EndOfFile) { @@ -1580,6 +1587,8 @@ private static void ValidateHeaderAgainstWal(SqliteWalIndexHeader header, Sqlite throw new InvalidDataException( "SQLite WAL-index frame checksum does not match its maximum WAL frame."); } + + wal.RecordIndexHeaderValidated(header); } private static void ValidateMatchedFrame( diff --git a/src/Ahtola.Core/Storage/SqliteWalReadSnapshotCoordinator.cs b/src/Ahtola.Core/Storage/SqliteWalReadSnapshotCoordinator.cs index a93ddbd..5d499ee 100644 --- a/src/Ahtola.Core/Storage/SqliteWalReadSnapshotCoordinator.cs +++ b/src/Ahtola.Core/Storage/SqliteWalReadSnapshotCoordinator.cs @@ -570,6 +570,7 @@ public sealed class SqliteWalReadSnapshot : IDisposable private readonly SqliteWalIndexSharedMemory _index; private SqliteWalByteRangeLockLease? _readMarkLease; private Exception? _fault; + private SqliteWalIndexHeader? _walValidatedHeader; internal SqliteWalReadSnapshot( SqliteWalReadSnapshotCoordinator owner, @@ -680,7 +681,18 @@ internal void EnsureStillValid() private void ValidatePinnedBoundary() { ThrowIfUnavailable(); - var region = _index.ReadValidatedHeader(_wal); + // Readers re-check the pinned boundary on every page access. Cross-validating the header + // against the WAL rescans its tail and re-reads the commit frame (file I/O each time), so + // do it only when the header changed: an identical header (change counter, boundary, + // salts and the chained frame checksums) was already proven to describe this WAL's + // committed prefix, which frames appended after it cannot alter. + var region = _index.ReadStableHeader(); + if (!Equals(region.Header, _walValidatedHeader)) + { + region = _index.ReadValidatedHeader(_wal); + _walValidatedHeader = region.Header; + } + if (region.Header.MaximumFrame < MaximumFrame) { throw new InvalidDataException( diff --git a/src/Ahtola.Core/TableDerivedCache.cs b/src/Ahtola.Core/TableDerivedCache.cs new file mode 100644 index 0000000..6b74825 --- /dev/null +++ b/src/Ahtola.Core/TableDerivedCache.cs @@ -0,0 +1,55 @@ +using System.Diagnostics.CodeAnalysis; + +namespace Ahtola.Core; + +/// +/// Read-only structures derived from a table's rows (rowid scan order, rowid positions, +/// equality lookups, index scan orders), shared by a table and every clone made from it. +/// +/// +/// Every statement executes against a clone of the catalog. Caches held on the +/// instance and validated by were +/// therefore rebuilt by every statement, which made a primary-key lookup or a rowid-ordered +/// scan cost O(table rows) or O(table rows · log) per statement. Entries here are validated by +/// the of the rows and rowids they were derived +/// from instead: a stamp is process-unique per content state, so an entry built by any clone +/// is valid for every other table holding exactly the same rows, and a clone that diverges +/// (including one whose statement is later rolled back) can never satisfy another clone's +/// lookup. Values must never be mutated after . +/// +internal sealed class TableDerivedCache +{ + private readonly object _gate = new(); + private Dictionary? _entries; + + public bool TryGet(object key, long rowsStamp, long rowIdsStamp, [NotNullWhen(true)] out T? value) + where T : class + { + lock (_gate) + { + if (_entries is not null + && _entries.TryGetValue(key, out var entry) + && entry.RowsStamp == rowsStamp + && entry.RowIdsStamp == rowIdsStamp + && entry.Value is T typed) + { + value = typed; + return true; + } + } + + value = null; + return false; + } + + public void Set(object key, long rowsStamp, long rowIdsStamp, object value) + { + lock (_gate) + { + _entries ??= []; + _entries[key] = new Entry(rowsStamp, rowIdsStamp, value); + } + } + + private readonly record struct Entry(long RowsStamp, long RowIdsStamp, object Value); +} diff --git a/src/Ahtola.Core/TransactionMutationOverlay.cs b/src/Ahtola.Core/TransactionMutationOverlay.cs index edca88b..ddae079 100644 --- a/src/Ahtola.Core/TransactionMutationOverlay.cs +++ b/src/Ahtola.Core/TransactionMutationOverlay.cs @@ -1,3 +1,4 @@ +using System.Collections.Immutable; using Ahtola.Core.Storage; namespace Ahtola.Core; @@ -19,7 +20,9 @@ namespace Ahtola.Core; /// internal sealed class TransactionTableOverlay { - private readonly Dictionary? _byRowId; + // Immutable so a checkpoint (taken before every statement of a transaction) captures it in + // O(1); copying a mutable map there made a long transaction quadratic in its row changes. + private ImmutableDictionary? _byRowId; private readonly List<(SqlValue[] Key, SqlValue[]? Current)>? _byPrimaryKey; private readonly SqliteIndexRecordComparer? _primaryKeyComparer; @@ -28,7 +31,7 @@ public TransactionTableOverlay(bool hasRowid, SqliteIndexRecordComparer? primary HasRowid = hasRowid; if (hasRowid) { - _byRowId = new Dictionary(); + _byRowId = ImmutableDictionary.Empty; } else { @@ -48,7 +51,7 @@ public void SetRowId(long rowId, SqlValue[]? current) { if (!HasRowid) throw new InvalidOperationException("This overlay tracks primary keys, not rowids."); - _byRowId![rowId] = current; + _byRowId = _byRowId!.SetItem(rowId, current); } public bool TryGetByRowId(long rowId, out SqlValue[]? current) @@ -134,7 +137,7 @@ public IEnumerable EnumerateCurrentRows() /// Captures a point-in-time copy of this table's overlay for a SAVEPOINT. public TransactionTableOverlayCheckpoint CreateCheckpoint() => HasRowid - ? new TransactionTableOverlayCheckpoint(true, new Dictionary(_byRowId!), null) + ? new TransactionTableOverlayCheckpoint(true, _byRowId, null) : new TransactionTableOverlayCheckpoint( false, null, @@ -149,13 +152,7 @@ public void RestoreCheckpoint(TransactionTableOverlayCheckpoint checkpoint) { if (HasRowid) { - _byRowId!.Clear(); - if (checkpoint.ByRowId is { } byRowId) - { - foreach (var pair in byRowId) - _byRowId[pair.Key] = pair.Value; - } - + _byRowId = checkpoint.ByRowId ?? ImmutableDictionary.Empty; return; } @@ -167,7 +164,8 @@ public void RestoreCheckpoint(TransactionTableOverlayCheckpoint checkpoint) /// Discards all recorded mutations, used when restoring a checkpoint predating this table's first touch. public void Clear() { - _byRowId?.Clear(); + if (_byRowId is not null) + _byRowId = ImmutableDictionary.Empty; _byPrimaryKey?.Clear(); } } @@ -175,7 +173,7 @@ public void Clear() /// An immutable point-in-time copy of one table's mutation overlay for SAVEPOINT support. internal sealed record TransactionTableOverlayCheckpoint( bool HasRowid, - Dictionary? ByRowId, + ImmutableDictionary? ByRowId, List<(SqlValue[] Key, SqlValue[]? Current)>? ByPrimaryKey); /// diff --git a/src/Ahtola.Data.Sqlite/SqliteConnection.cs b/src/Ahtola.Data.Sqlite/SqliteConnection.cs index af8c502..5c15a3b 100644 --- a/src/Ahtola.Data.Sqlite/SqliteConnection.cs +++ b/src/Ahtola.Data.Sqlite/SqliteConnection.cs @@ -1279,6 +1279,14 @@ internal IManagedConnectionAdapter ManagedConnection internal bool IsManagedConnection => _managedDatabase is not null; + /// + /// Declared-type metadata the data reader read for source tables, keyed by table name and + /// tagged with the engine's schema identity for the table it was read from. Entries are only + /// reused while that identity still compares equal, so they need no invalidation. + /// + internal Dictionary TableColumnCache { get; } + = new(StringComparer.OrdinalIgnoreCase); + internal bool RequiresAsyncExecution => _managedDatabaseFactory is not null; /// diff --git a/src/Ahtola.Data.Sqlite/SqliteDataReader.cs b/src/Ahtola.Data.Sqlite/SqliteDataReader.cs index 6f25a17..137f015 100644 --- a/src/Ahtola.Data.Sqlite/SqliteDataReader.cs +++ b/src/Ahtola.Data.Sqlite/SqliteDataReader.cs @@ -852,11 +852,16 @@ public override object GetValue(int ordinal) EnsureOpen(); EnsureHasCurrentRow(); var value = ReadValue(ordinal); - var declaredType = GetDeclaredTypeName(ordinal); - if (IsGuidType(declaredType) && value.Kind is ReaderValueKind.Blob or ReaderValueKind.Text) - return ToGuid(ordinal, value); - if (ShouldMaterializeTextGuid(ordinal, declaredType, value)) - return ToGuid(ordinal, value).ToString("D", CultureInfo.InvariantCulture).ToUpperInvariant(); + // Only TEXT and BLOB values can map to a Guid; resolving the declared type runs schema + // PRAGMAs once per result set, which numeric and NULL values never need. + if (value.Kind is ReaderValueKind.Blob or ReaderValueKind.Text) + { + var declaredType = GetDeclaredTypeName(ordinal); + if (IsGuidType(declaredType)) + return ToGuid(ordinal, value); + if (ShouldMaterializeTextGuid(ordinal, declaredType, value)) + return ToGuid(ordinal, value).ToString("D", CultureInfo.InvariantCulture).ToUpperInvariant(); + } return value.Kind switch { @@ -2114,14 +2119,38 @@ private static List SplitSelectList(string selectList) private Dictionary GetTableColumns(string tableName) { - var columns = new Dictionary(StringComparer.OrdinalIgnoreCase); if (_command.Connection is null) - return columns; + return new Dictionary(StringComparer.OrdinalIgnoreCase); if (_command.Connection.RequiresAsyncExecution) return GetManagedTableColumns(_command.Connection.ManagedConnection, tableName); - using var suspension = _command.Connection.SuspendHooks(); - using (var command = _command.Connection.CreateCommand()) + // Declared-type metadata comes from three or more pragma statements per source table and + // result set, which made a Dapper-style GetValue on a TEXT column cost several statements + // per query. It is reused on this connection while the engine reports the same table schema. + var identity = _command.Connection.IsManagedConnection + ? _command.Connection.ManagedConnection.GetTableSchemaIdentity(tableName) + : null; + var cache = _command.Connection.TableColumnCache; + if (identity is not null + && cache.TryGetValue(tableName, out var cached) + && Equals(cached.Identity, identity) + && cached.Columns is Dictionary cachedColumns) + { + return new Dictionary(cachedColumns, StringComparer.OrdinalIgnoreCase); + } + + var columns = ReadTableColumns(_command.Connection, tableName); + if (identity is not null) + cache[tableName] = (identity, new Dictionary(columns, StringComparer.OrdinalIgnoreCase)); + return columns; + } + + private Dictionary ReadTableColumns(SqliteConnection connection, string tableName) + { + var columns = new Dictionary(StringComparer.OrdinalIgnoreCase); + + using var suspension = connection.SuspendHooks(); + using (var command = connection.CreateCommand()) { command.CommandText = $"PRAGMA table_info({QuoteIdentifier(tableName)});"; using var reader = command.ExecuteReader(); @@ -2137,7 +2166,7 @@ private Dictionary GetTableColumns(string tableName) } } - using (var indexCommand = _command.Connection.CreateCommand()) + using (var indexCommand = connection.CreateCommand()) { indexCommand.CommandText = $"PRAGMA index_list({QuoteIdentifier(tableName)});"; using var indexes = indexCommand.ExecuteReader(); @@ -2147,7 +2176,7 @@ private Dictionary GetTableColumns(string tableName) continue; var indexName = indexes.GetString(1); - using var infoCommand = _command.Connection.CreateCommand(); + using var infoCommand = connection.CreateCommand(); infoCommand.CommandText = $"PRAGMA index_info({QuoteIdentifier(indexName)});"; using var indexInfo = infoCommand.ExecuteReader(); var indexedColumns = new List(); diff --git a/src/Ahtola.Data/ManagedConnectionPool.cs b/src/Ahtola.Data/ManagedConnectionPool.cs index 91d9a04..b6a77b8 100644 --- a/src/Ahtola.Data/ManagedConnectionPool.cs +++ b/src/Ahtola.Data/ManagedConnectionPool.cs @@ -258,7 +258,7 @@ public void Release(bool reusable) try { - released.Database.Connection.ResetForPooling(); + released.Database.Connection.ResetForPoolReturn(); pool.Return(released); } catch diff --git a/src/Ahtola.Tests/ForeignReadOnlyOpenTests.cs b/src/Ahtola.Tests/ForeignReadOnlyOpenTests.cs index 4909b75..d80c6ed 100644 --- a/src/Ahtola.Tests/ForeignReadOnlyOpenTests.cs +++ b/src/Ahtola.Tests/ForeignReadOnlyOpenTests.cs @@ -138,6 +138,40 @@ public void OwnerRecreatingWalBetweenStatementsIsAdoptedByForeignReader() } } + [Test] + public void OwnerCheckpointWithinOneTimestampTickIsAdoptedByForeignReader() + { + // A peer that commits in WAL mode and checkpoints on close leaves the main file's size + // and change counter as they were; only its last-write time moves, and only by whole + // clock ticks. Pin it back to its earlier value to reproduce a write landing in the same + // tick the foreign reader captured its view in. + var path = CreateDatabasePath("wal-same-tick"); + try + { + CreateSqliteDatabase(path, "PRAGMA journal_mode=WAL;"); + var lastWrite = File.GetLastWriteTimeUtc(path); + var length = new FileInfo(path).Length; + + using var foreign = OpenForeign(path); + QueryStrings(foreign, "SELECT name FROM packages ORDER BY id;").Should().Equal("one", "two"); + + using (var owner = OpenSqlite(path)) + ExecuteSqlite(owner, "UPDATE packages SET name = 'uno' WHERE name = 'one';"); + File.Exists(path + "-wal").Should().BeFalse(); + new FileInfo(path).Length.Should().Be(length); + File.SetLastWriteTimeUtc(path, lastWrite); + + QueryStrings(foreign, "SELECT name FROM packages ORDER BY id;") + .Should().Equal( + new[] { "uno", "two" }, + "a stamp captured within the timestamp granularity of a write cannot prove the file unchanged"); + } + finally + { + DeleteDatabase(path); + } + } + [Test] public void CleanlyClosedDeleteJournalDatabaseOpensAndMatchesSqlite() { diff --git a/src/Ahtola.Tests/HotspotAccessPathParityTests.cs b/src/Ahtola.Tests/HotspotAccessPathParityTests.cs new file mode 100644 index 0000000..12be3c9 --- /dev/null +++ b/src/Ahtola.Tests/HotspotAccessPathParityTests.cs @@ -0,0 +1,403 @@ +using System.Data.Common; +using System.Globalization; +using AwesomeAssertions; +using AhtolaSqliteConnection = Ahtola.Data.Sqlite.SqliteConnection; +using MicrosoftSqliteConnection = Microsoft.Data.Sqlite.SqliteConnection; + +namespace Ahtola.Tests; + +/// +/// Differential checks for the access paths that serve common .NET query shapes without a full +/// scan: rowid and secondary-index equality, IN lists, rowid ranges, ORDER BY satisfied by an +/// index or by rowid order with LIMIT/OFFSET, covering COUNT and existence probes, small-left +/// join probes, DML candidate pruning, and RETURNING with and without subqueries. Every query +/// runs against the same SQLite-written file through Microsoft.Data.Sqlite and through Ahtola's +/// facade, and must return identical typed rows (in order whenever the SQL orders them). +/// +[NonParallelizable] +public sealed class HotspotAccessPathParityTests +{ + private string _root = null!; + + [SetUp] + public void SetUp() + { + _root = Path.Combine(Path.GetTempPath(), "ahtola-hotspot-parity-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(_root); + } + + [TearDown] + public void TearDown() + { + AhtolaSqliteConnection.ClearAllPools(); + MicrosoftSqliteConnection.ClearAllPools(); + try + { + Directory.Delete(_root, recursive: true); + } + catch (IOException) + { + } + catch (UnauthorizedAccessException) + { + } + } + + [TestCase("SELECT id FROM items WHERE id IN (@a, @b, @c) ORDER BY id;", 3L, 5L, 7L)] + [TestCase("SELECT id FROM items WHERE id IN (@a, @b, @c) ORDER BY id;", 2.0, "4", 9.5)] + [TestCase("SELECT id FROM items WHERE id IN (@a, @b, @c) ORDER BY id;", null, -2L, 1000000L)] + [TestCase("SELECT id FROM items WHERE category IN (@a, @b, @c) ORDER BY id;", 1L, "2", 3.0)] + [TestCase("SELECT id FROM items WHERE name IN (@a, @b, @c) ORDER BY id;", "ALPHA", "beta", "Gamma ")] + [TestCase("SELECT id FROM items WHERE code IN (@a, @b, @c) AND category = 1 ORDER BY id;", "c3", "c4", "c11")] + [TestCase("SELECT id FROM items WHERE id = @a;", 2.0, null, null)] + [TestCase("SELECT id FROM items WHERE id = @a;", "6", null, null)] + [TestCase("SELECT id, name FROM items WHERE category = @a ORDER BY id;", "1", null, null)] + public void InListAndEqualityProbesMatchSqlite(string sql, object? a, object? b, object? c) + => AssertSameResults(sql, ("@a", a), ("@b", b), ("@c", c)); + + [TestCase("SELECT id FROM items WHERE id <= 5 ORDER BY id;")] + [TestCase("SELECT id FROM items WHERE id < 2.5 ORDER BY id;")] + [TestCase("SELECT id FROM items WHERE id > -3 AND id < 4 ORDER BY id;")] + [TestCase("SELECT id FROM items WHERE id BETWEEN -2 AND 4 ORDER BY id;")] + [TestCase("SELECT id FROM items WHERE 5 >= id AND category = 1 ORDER BY id;")] + [TestCase("SELECT id FROM items WHERE id >= '3' ORDER BY id;")] + [TestCase("SELECT id FROM items WHERE id > 3 AND id > 10 AND id <= 1000000 ORDER BY id;")] + [TestCase("SELECT id FROM items WHERE id < -100;")] + public void RowidRangesMatchSqlite(string sql) + => AssertSameResults(sql); + + [TestCase("SELECT id, tag FROM items ORDER BY tag DESC LIMIT 5;")] + [TestCase("SELECT id, tag FROM items ORDER BY tag LIMIT 5;")] + [TestCase("SELECT id, category FROM items WHERE category > 0 ORDER BY category LIMIT 3 OFFSET 2;")] + [TestCase("SELECT id, category FROM items WHERE category < 3 ORDER BY category DESC LIMIT 4;")] + [TestCase("SELECT id, name FROM items ORDER BY name LIMIT 4;")] + [TestCase("SELECT id, name FROM items ORDER BY name COLLATE BINARY LIMIT 4;")] + [TestCase("SELECT id, name FROM items ORDER BY id LIMIT 3 OFFSET 2;")] + [TestCase("SELECT id FROM items ORDER BY rowid LIMIT 2;")] + [TestCase("SELECT id FROM items ORDER BY id DESC LIMIT 2;")] + [TestCase("SELECT id FROM items ORDER BY id LIMIT 0;")] + [TestCase("SELECT id FROM items ORDER BY id LIMIT -1 OFFSET 12;")] + public void OrderedLimitsMatchSqlite(string sql) + => AssertSameResults(sql); + + [TestCase("SELECT id, category FROM items WHERE category > 0 AND category <= 2 ORDER BY category LIMIT 6;")] + [TestCase("SELECT id, category FROM items WHERE category > 0 AND category <= 2 ORDER BY category, id;")] + [TestCase("SELECT id, name FROM items WHERE name >= 'b' ORDER BY name LIMIT 3;")] + [TestCase("SELECT id, name FROM items WHERE name < 'GAMMA' ORDER BY name DESC LIMIT 3;")] + [TestCase("SELECT id, price FROM items WHERE price < 5 ORDER BY price DESC LIMIT 2;")] + [TestCase("SELECT id, price FROM items WHERE price >= 2.5 AND price < 9 ORDER BY price;")] + [TestCase("SELECT id, category FROM items WHERE category BETWEEN '1' AND 2 ORDER BY category LIMIT 4;")] + [TestCase("SELECT id, category FROM items WHERE category > '1' ORDER BY category LIMIT 4;")] + [TestCase("SELECT id, tag FROM items WHERE tag < 'c' ORDER BY tag DESC LIMIT 3;")] + [TestCase("SELECT id FROM items WHERE category > NULL ORDER BY category LIMIT 3;")] + [TestCase("SELECT id FROM items WHERE 2 >= category ORDER BY category LIMIT 5;")] + [TestCase("SELECT COUNT(*) FROM items WHERE category >= 1 AND category < 3;")] + public void IndexRangesMatchSqlite(string sql) + { + // Twice on one connection: the second run is served from the warm shared index order. + AssertSameScript(sql, sql); + } + + [TestCase("SELECT id, price FROM items ORDER BY id;")] + [TestCase("SELECT id, typeof(price), price FROM items ORDER BY id;")] + [TestCase("SELECT price FROM items WHERE price = 2 ORDER BY price;")] + [TestCase("SELECT COUNT(*), MIN(price), MAX(price) FROM items WHERE price >= 3;")] + [TestCase("SELECT price FROM items WHERE price BETWEEN 4 AND 9 ORDER BY price DESC;")] + [TestCase("SELECT code, rate, typeof(rate), weight, typeof(weight) FROM rates ORDER BY code;")] + [TestCase("SELECT rate FROM rates WHERE code = 'a';")] + [TestCase("SELECT COUNT(*) FROM items WHERE category = 1;")] + [TestCase("SELECT COUNT(*) FROM items WHERE category = '1';")] + [TestCase("SELECT COUNT(*) FROM items WHERE category = NULL;")] + [TestCase("SELECT 1 FROM items WHERE category = 2 LIMIT 1;")] + [TestCase("SELECT COUNT(*), 7 FROM items WHERE category = 2 AND category IN (1, 2);")] + [TestCase("SELECT COUNT(*) FROM items WHERE category = 1 AND id > 2;")] + public void CoveringProbesMatchSqlite(string sql) + => AssertSameResults(sql); + + [TestCase("SELECT o.id, i.name FROM orders AS o JOIN items AS i ON i.id = o.item_id WHERE o.item_id = 3 ORDER BY o.id;")] + [TestCase("SELECT o.id, i.name FROM orders AS o LEFT JOIN items AS i ON i.id = o.item_id WHERE o.qty = 9 ORDER BY o.id;")] + [TestCase("SELECT o.id, i.name FROM orders AS o JOIN items AS i ON i.id = o.item_id AND i.category = 1 WHERE o.qty < 3 ORDER BY o.id;")] + [TestCase("SELECT i.category, COUNT(*), SUM(o.qty) FROM orders AS o JOIN items AS i ON i.id = o.item_id WHERE i.category < 3 GROUP BY i.category ORDER BY i.category;")] + [TestCase("SELECT o.rowid, i.rowid, o.id FROM orders AS o JOIN items AS i ON i.id = o.item_id ORDER BY o.id;")] + [TestCase("SELECT a.rowid, b.rowid, i.rowid, i._rowid_ FROM orders AS a JOIN orders AS b ON b.item_id = a.item_id AND b.id <> a.id JOIN items AS i ON i.id = b.item_id ORDER BY a.id, b.id;")] + [TestCase("SELECT o.id, i.rowid FROM orders AS o LEFT JOIN items AS i ON i.id = o.item_id ORDER BY o.id;")] + public void JoinProbesMatchSqlite(string sql) + => AssertSameResults(sql); + + [TestCase("UPDATE items SET price = price + 1 WHERE id = 2.0;")] + [TestCase("UPDATE items SET price = price WHERE id = 4;")] + [TestCase("DELETE FROM orders WHERE id = '3';")] + [TestCase("DELETE FROM orders WHERE item_id = 5 AND qty > 1;")] + [TestCase("DELETE FROM orders WHERE id = 1 RETURNING id, qty;")] + [TestCase("DELETE FROM orders WHERE id = 2 RETURNING id, (SELECT COUNT(*) FROM orders);")] + [TestCase("INSERT INTO orders(item_id, qty) VALUES (1, 5) RETURNING id, qty * 2;")] + [TestCase("INSERT INTO orders(item_id, qty) VALUES (1, 5), (2, 6) RETURNING id, (SELECT MAX(id) FROM orders);")] + [TestCase("UPDATE orders SET qty = qty + 1 WHERE id = 4 RETURNING id, qty, (SELECT SUM(qty) FROM orders);")] + [TestCase("UPDATE items SET price = 1 WHERE id = 2.5;")] + [TestCase("UPDATE items SET price = 1 WHERE id = '3';")] + [TestCase("UPDATE items SET price = 1 WHERE id = 8;")] + [TestCase("UPDATE items SET price = 1 WHERE id = 1000000.0;")] + [TestCase("UPDATE items SET price = 1 WHERE id = -2 AND category = 2;")] + [TestCase("DELETE FROM items WHERE id = 9007199254740993;")] + public void DmlMatchesSqlite(string sql) + { + AssertSameResults(sql); + AssertSameResults("SELECT id, item_id, qty FROM orders ORDER BY id;", reuseFixture: true); + AssertSameResults("SELECT id, price FROM items ORDER BY id;", reuseFixture: true); + } + + [Test] + public void RepeatedStatementsSeeEachOthersWrites() + { + // Shared per-table caches (rowid order, equality buckets, sorted index order) must follow + // every write, including a delete of the maximum rowid followed by an insert. + AssertSameScript( + "SELECT id FROM items WHERE category = 1 ORDER BY id;", + "INSERT INTO items(id, code, name, category, tag) VALUES (50, 'c50', 'zeta', 1, 'z');", + "SELECT id FROM items WHERE category = 1 ORDER BY id;", + "SELECT id, tag FROM items ORDER BY tag DESC LIMIT 3;", + "UPDATE items SET category = 2, tag = 'zz' WHERE id = 50;", + "SELECT id FROM items WHERE category = 1 ORDER BY id;", + "SELECT id, tag FROM items ORDER BY tag DESC LIMIT 3;", + "SELECT id FROM items WHERE id IN (49, 50, 51);", + "DELETE FROM items WHERE id = 50;", + "INSERT INTO items(code, name, category) VALUES ('c51', 'eta', 1) RETURNING id;", + "SELECT id FROM items WHERE id > 7 ORDER BY id;", + "SELECT COUNT(*) FROM items WHERE category = 1;", + "BEGIN; INSERT INTO orders(item_id, qty) VALUES (2, 4) RETURNING id; DELETE FROM orders WHERE item_id = 2 AND qty = 4; COMMIT;", + "SELECT o.id, i.name FROM orders AS o JOIN items AS i ON i.id = o.item_id WHERE o.item_id = 2 ORDER BY o.id;"); + } + + [Test] + public void RepeatedIndexSeeksMatchSqliteAcrossTheWarmOrderSwitch() + { + // Enough repeated equality seeks on an unchanged table switch the index to an in-memory + // order; a write restarts the count. Results must not change at either switch. + var statements = new List(); + for (var round = 0; round < 40; round++) + { + statements.Add($"SELECT id FROM items WHERE category = {round % 4} ORDER BY id;"); + statements.Add($"SELECT id, name FROM items WHERE name = '{(round % 2 == 0 ? "alpha" : "Gamma ")}' ORDER BY id;"); + statements.Add($"SELECT id FROM items WHERE tag = '{(char)('a' + (round % 6))}';"); + if (round == 36) + statements.Add("UPDATE items SET category = 3, name = 'ALPHA', tag = 'a' WHERE id = 9;"); + if (round == 38) + statements.Add("INSERT INTO items(id, code, name, category, tag) VALUES (12, 'c12', 'alpha', 2, 'b');"); + } + + AssertSameScript([.. statements]); + } + + [Test] + public void ChunkSpanningUpdatesAndAppendsPersistLikeSqlite() + { + // Commits diff the committed and new row versions chunk by chunk (1024 rows); changes in + // several chunks, at chunk edges, and appended in the same transaction must all persist. + string[] statements = + [ + "CREATE TABLE wide(id INTEGER PRIMARY KEY, v INTEGER, note TEXT);", + "CREATE INDEX ix_wide_v ON wide(v);", + "WITH RECURSIVE n(x) AS (SELECT 1 UNION ALL SELECT x + 1 FROM n WHERE x < 3000) INSERT INTO wide SELECT x, x % 97, 'n' || x FROM n;", + "UPDATE wide SET v = -1 WHERE id = 1024;", + "UPDATE wide SET v = -2 WHERE id = 1025;", + "BEGIN; UPDATE wide SET note = 'first' WHERE id = 1; UPDATE wide SET v = -3 WHERE id = 2999; INSERT INTO wide VALUES (3001, -4, 'appended'); COMMIT;", + "INSERT INTO wide(v, note) VALUES (-5, 'tail');", + "UPDATE wide SET v = v WHERE id = 1500;", + "BEGIN; WITH RECURSIVE n(x) AS (SELECT 1 UNION ALL SELECT x + 1 FROM n WHERE x < 200) INSERT INTO wide SELECT 3100 + x * 2, -x, 'batch' FROM n; WITH RECURSIVE n(x) AS (SELECT 1 UNION ALL SELECT x + 1 FROM n WHERE x < 200) INSERT INTO wide SELECT 3600 + x, x, 'append' FROM n; COMMIT;", + "DELETE FROM wide WHERE id > 3100 AND id % 4 = 0;", + "BEGIN; WITH RECURSIVE n(x) AS (SELECT 1 UNION ALL SELECT x + 1 FROM n WHERE x < 100) INSERT INTO wide SELECT 3100 + x * 4, x, 'refill' FROM n; COMMIT;", + "SELECT id, v, note FROM wide WHERE v < 0 ORDER BY id;", + "SELECT COUNT(*), SUM(v), MAX(id) FROM wide;", + ]; + var (sqlitePath, ahtolaPath) = CreateFixtures(); + using (var sqlite = new MicrosoftSqliteConnection($"Data Source={sqlitePath};Pooling=False")) + using (var ahtola = new AhtolaSqliteConnection($"Data Source={ahtolaPath}")) + { + sqlite.Open(); + ahtola.Open(); + foreach (var sql in statements) + Read(ahtola, sql, []).Should().Equal(Read(sqlite, sql, []), sql); + } + + AhtolaSqliteConnection.ClearAllPools(); + const string dump = "SELECT id, v, note FROM wide ORDER BY id;"; + var expected = Run(new MicrosoftSqliteConnection($"Data Source={sqlitePath};Pooling=False"), dump, []); + Run(new MicrosoftSqliteConnection($"Data Source={ahtolaPath};Pooling=False"), dump, []).Should().Equal(expected); + Run(new MicrosoftSqliteConnection($"Data Source={ahtolaPath};Pooling=False"), "PRAGMA integrity_check;", []) + .Should().Equal("T:ok"); + } + + [Test] + public void IntegralRealsWrittenByAhtolaReadBackAsRealInBothEngines() + { + // SQLite stores an integral REAL in integer form and reads it back as REAL; both engines + // must agree on what Ahtola wrote, and SQLite must accept the file, index included. + var (_, ahtolaPath) = CreateFixtures(); + using (var ahtola = new AhtolaSqliteConnection($"Data Source={ahtolaPath}")) + { + ahtola.Open(); + Read(ahtola, "UPDATE items SET price = 5 WHERE id = 1;", []); + Read(ahtola, "UPDATE items SET price = 8.0, name = name || '!' WHERE id = 2;", []); + Read(ahtola, "INSERT INTO items(id, code, price) VALUES (12, 'c12', -3.0), (13, 'c13', 0.0), (14, 'c14', 1e300);", []); + Read(ahtola, "INSERT INTO rates VALUES ('d', 7, 7.5);", []); + } + + AhtolaSqliteConnection.ClearAllPools(); + const string query = "SELECT id, typeof(price), price FROM items ORDER BY id;"; + const string rates = "SELECT code, typeof(rate), rate, typeof(weight), weight FROM rates ORDER BY code;"; + const string indexed = "SELECT price FROM items INDEXED BY ix_items_price WHERE price >= 0 ORDER BY price;"; + var fromAhtola = Run(new AhtolaSqliteConnection($"Data Source={ahtolaPath}"), query, []); + var ratesFromAhtola = Run(new AhtolaSqliteConnection($"Data Source={ahtolaPath}"), rates, []); + var indexedFromAhtola = Run(new AhtolaSqliteConnection($"Data Source={ahtolaPath}"), indexed, []); + AhtolaSqliteConnection.ClearAllPools(); + + Run(new MicrosoftSqliteConnection($"Data Source={ahtolaPath};Pooling=False"), query, []).Should().Equal(fromAhtola); + Run(new MicrosoftSqliteConnection($"Data Source={ahtolaPath};Pooling=False"), rates, []).Should().Equal(ratesFromAhtola); + Run(new MicrosoftSqliteConnection($"Data Source={ahtolaPath};Pooling=False"), indexed, []).Should().Equal(indexedFromAhtola); + Run(new MicrosoftSqliteConnection($"Data Source={ahtolaPath};Pooling=False"), "PRAGMA integrity_check;", []) + .Should().Equal("T:ok"); + fromAhtola.Should().Contain(["I:1|T:real|R:5", "I:2|T:real|R:8", "I:12|T:real|R:-3", "I:13|T:real|R:0"]); + } + + private void AssertSameResults(string sql, params (string Name, object? Value)[] parameters) + => AssertSameResults(sql, reuseFixture: false, parameters); + + private (string SqlitePath, string AhtolaPath)? _fixture; + + private void AssertSameResults(string sql, bool reuseFixture, params (string Name, object? Value)[] parameters) + { + var (sqlitePath, ahtolaPath) = reuseFixture && _fixture is { } existing ? existing : CreateFixtures(); + _fixture = (sqlitePath, ahtolaPath); + var expected = Run(new MicrosoftSqliteConnection($"Data Source={sqlitePath};Pooling=False"), sql, parameters); + var actual = Run(new AhtolaSqliteConnection($"Data Source={ahtolaPath}"), sql, parameters); + actual.Should().Equal(expected, sql); + } + + private void AssertSameScript(params string[] statements) + { + var (sqlitePath, ahtolaPath) = CreateFixtures(); + using var sqlite = new MicrosoftSqliteConnection($"Data Source={sqlitePath};Pooling=False"); + using var ahtola = new AhtolaSqliteConnection($"Data Source={ahtolaPath}"); + sqlite.Open(); + ahtola.Open(); + foreach (var sql in statements) + { + var expected = Read(sqlite, sql, []); + var actual = Read(ahtola, sql, []); + actual.Should().Equal(expected, sql); + } + } + + private (string SqlitePath, string AhtolaPath) CreateFixtures() + { + var name = Guid.NewGuid().ToString("N"); + var sqlitePath = Path.Combine(_root, name + "-sqlite.db"); + var ahtolaPath = Path.Combine(_root, name + "-ahtola.db"); + using (var connection = new MicrosoftSqliteConnection($"Data Source={sqlitePath};Pooling=False")) + { + connection.Open(); + Execute(connection, "PRAGMA journal_mode=WAL;"); + Execute( + connection, + """ + -- Integral prices exercise SQLite's on-disk form of a REAL: stored as an + -- integer, read back as REAL. + CREATE TABLE items( + id INTEGER PRIMARY KEY, + code TEXT UNIQUE, + name TEXT COLLATE NOCASE, + category INTEGER, + price REAL, + tag TEXT); + CREATE INDEX ix_items_category ON items(category); + CREATE INDEX ix_items_name ON items(name); + CREATE INDEX ix_items_tag ON items(tag DESC); + CREATE INDEX ix_items_price ON items(price); + CREATE TABLE rates(code TEXT PRIMARY KEY, rate REAL, weight FLOAT) WITHOUT ROWID; + INSERT INTO rates VALUES ('a', 1.0, 2), ('b', 2.5, 3.0), ('c', -4.0, NULL); + CREATE TABLE orders(id INTEGER PRIMARY KEY, item_id INTEGER, qty INTEGER); + CREATE INDEX ix_orders_item ON orders(item_id); + INSERT INTO items VALUES (-2, 'cm2', 'Omega', 2, 1.5, NULL); + INSERT INTO items VALUES (1, 'c1', 'alpha', 1, 2.0, 'b'); + INSERT INTO items VALUES (2, 'c2', 'Beta', 1, 2.5, 'a'); + INSERT INTO items VALUES (3, 'c3', 'gamma', 2, 3.0, NULL); + INSERT INTO items VALUES (4, 'c4', 'ALPHA', 1, 4.0, 'c'); + INSERT INTO items VALUES (5, 'c5', 'delta', '7', 5.5, 'b'); + INSERT INTO items VALUES (6, 'c6', 'epsilon', NULL, 6.0, 'd'); + INSERT INTO items VALUES (7, 'c7', 'Gamma ', 3, 7.0, 'a'); + INSERT INTO items VALUES (9, 'c9', 'beta', 2, 9.0, 'e'); + INSERT INTO items VALUES (11, 'c11', 'theta', 1, 11.0, 'c'); + INSERT INTO items VALUES (1000000, 'cbig', 'iota', 0, 0.5, 'f'); + INSERT INTO orders VALUES (1, 3, 1); + INSERT INTO orders VALUES (2, 3, 2); + INSERT INTO orders VALUES (3, 5, 9); + INSERT INTO orders VALUES (4, 42, 9); + INSERT INTO orders VALUES (5, 1, 2); + INSERT INTO orders VALUES (6, 5, 3); + INSERT INTO orders VALUES (7, 2, 1); + """); + } + + File.Copy(sqlitePath, ahtolaPath); + return (sqlitePath, ahtolaPath); + } + + private static List Run(DbConnection connection, string sql, (string Name, object? Value)[] parameters) + { + using (connection) + { + connection.Open(); + return Read(connection, sql, parameters); + } + } + + private static List Read(DbConnection connection, string sql, (string Name, object? Value)[] parameters) + { + using var command = connection.CreateCommand(); + command.CommandText = sql; + foreach (var (name, value) in parameters) + { + if (!sql.Contains(name, StringComparison.Ordinal)) + continue; + var parameter = command.CreateParameter(); + parameter.ParameterName = name; + parameter.Value = value ?? DBNull.Value; + command.Parameters.Add(parameter); + } + + var rows = new List(); + using var reader = command.ExecuteReader(); + do + { + while (reader.Read()) + { + var cells = new string[reader.FieldCount]; + for (var ordinal = 0; ordinal < cells.Length; ordinal++) + { + cells[ordinal] = reader.IsDBNull(ordinal) + ? "NULL" + : reader.GetValue(ordinal) switch + { + long integer => "I:" + integer.ToString(CultureInfo.InvariantCulture), + double real => "R:" + real.ToString("R", CultureInfo.InvariantCulture), + string text => "T:" + text, + byte[] blob => "B:" + Convert.ToHexString(blob), + var other => "?:" + other, + }; + } + + rows.Add(string.Join('|', cells)); + } + } + while (reader.NextResult()); + + return rows; + } + + private static void Execute(DbConnection connection, string sql) + { + using var command = connection.CreateCommand(); + command.CommandText = sql; + command.ExecuteNonQuery(); + } +} diff --git a/src/Ahtola.Tests/ManagedIncrementalPagerTests.cs b/src/Ahtola.Tests/ManagedIncrementalPagerTests.cs index d048b65..177740c 100644 --- a/src/Ahtola.Tests/ManagedIncrementalPagerTests.cs +++ b/src/Ahtola.Tests/ManagedIncrementalPagerTests.cs @@ -289,6 +289,47 @@ public void ColumnLocatorDoesNotMaterializeLargePrecedingBodies() "locating a column must allocate for the leaf/header, not preceding value bodies"); } + [Test] + public void InsertManyMatchesRowByRowInsertsAcrossLeavesAndAppends() + { + var pageIo = new CountingPageIo(pageSize: 4096, usableSpace: 4096, initialPageCount: 2); + pageIo.WritePage(2u, SqliteTableLeafPageBuilderImage(pageIo)); + var writer = new SqliteIncrementalTableBtree(pageIo); + var records = new Dictionary(); + for (var rowId = 3L; rowId <= 6000; rowId += 3) + { + records[rowId] = Record(rowId); + writer.Insert(2, rowId, records[rowId]); + } + + // Gap fills spread over many leaves, a run of appends past the maximum, and one + // overflow-sized record, all in one ascending batch. + var batch = new List<(long RowId, byte[] Record)>(); + for (var rowId = 1L; rowId <= 6000; rowId += 7) + { + if (rowId % 3 != 0) + batch.Add((rowId, Record(rowId))); + } + for (var rowId = 6001L; rowId <= 7500; rowId++) + batch.Add((rowId, rowId == 7000 ? new byte[10_000] : Record(rowId))); + foreach (var (rowId, record) in batch) + records[rowId] = record; + + writer.InsertMany(2, batch); + + var cursor = new SqliteTableBtreeCursor(pageIo); + for (var rowId = 0L; rowId <= 7600; rowId++) + { + var found = cursor.TrySeek(2, rowId, out var record); + found.Should().Be(records.ContainsKey(rowId), $"rowid {rowId}"); + if (found) + record.Should().Equal(records[rowId], $"rowid {rowId}"); + } + + Assert.Throws( + () => writer.InsertMany(2, [(9, Record(9))])); + } + [Test] public void ColumnReadsRejectNonByteValuesAndMalformedHeaders() { diff --git a/src/Ahtola.Tests/ManagedReaderDeclaredTypeCacheTests.cs b/src/Ahtola.Tests/ManagedReaderDeclaredTypeCacheTests.cs new file mode 100644 index 0000000..fc1d5ca --- /dev/null +++ b/src/Ahtola.Tests/ManagedReaderDeclaredTypeCacheTests.cs @@ -0,0 +1,171 @@ +using System.Data; +using AwesomeAssertions; +using Ahtola.Data.Sqlite; + +namespace Ahtola.Tests; + +/// +/// The data reader reuses the declared-type metadata it reads for a source table while the +/// engine reports the same schema for it. These pin that every schema change the pragmas would +/// observe is observed through the reused metadata too. +/// +public sealed class ManagedReaderDeclaredTypeCacheTests +{ + private static readonly Guid Value = Guid.Parse("0f8fad5b-d9cb-469f-a165-70867728950e"); + + private string _root = null!; + + [SetUp] + public void SetUp() + { + _root = Path.Combine(Path.GetTempPath(), "ahtola-decltype-cache-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(_root); + } + + [TearDown] + public void TearDown() + { + SqliteConnection.ClearAllPools(); + try + { + Directory.Delete(_root, recursive: true); + } + catch (IOException) + { + } + } + + [Test] + public void TempTableShadowingAndDroppingIsObserved() + { + using var connection = new SqliteConnection("Data Source=:memory:;Local Provider=Managed"); + connection.Open(); + connection.ExecuteNonQuery($""" + CREATE TABLE t(id INTEGER PRIMARY KEY, a TEXT); + INSERT INTO t VALUES (1, '{Value}'); + """); + + ReadFirstValue(connection).Should().Be(Value.ToString()); + + connection.ExecuteNonQuery($""" + CREATE TEMP TABLE t(id INTEGER PRIMARY KEY, a GUID); + INSERT INTO temp.t VALUES (1, '{Value}'); + """); + ReadFirstValue(connection).Should().Be(Value); + + connection.ExecuteNonQuery("DROP TABLE temp.t;"); + ReadFirstValue(connection).Should().Be(Value.ToString()); + } + + [Test] + public void AddedColumnsAndIndexesAreObserved() + { + using var connection = new SqliteConnection("Data Source=:memory:;Local Provider=Managed"); + connection.Open(); + connection.ExecuteNonQuery(""" + CREATE TABLE t(id INTEGER PRIMARY KEY, a TEXT); + INSERT INTO t VALUES (1, 'x'); + """); + + ReadSchema(connection).Should().Equal(("id", "INTEGER", true, false), ("a", "TEXT", false, false)); + + connection.ExecuteNonQuery($"ALTER TABLE t ADD COLUMN b GUID; UPDATE t SET b = '{Value}';"); + ReadSchema(connection).Should().Equal( + ("id", "INTEGER", true, false), ("a", "TEXT", false, false), ("b", "GUID", false, false)); + using (var command = connection.CreateCommand()) + { + command.CommandText = "SELECT b FROM t;"; + command.ExecuteScalar().Should().Be(Value); + } + + connection.ExecuteNonQuery("CREATE UNIQUE INDEX ux_t_a ON t(a);"); + ReadSchema(connection).Should().Equal( + ("id", "INTEGER", true, false), ("a", "TEXT", false, true), ("b", "GUID", false, false)); + + connection.ExecuteNonQuery("DROP INDEX ux_t_a;"); + ReadSchema(connection).Should().Equal( + ("id", "INTEGER", true, false), ("a", "TEXT", false, false), ("b", "GUID", false, false)); + } + + [Test] + public void RolledBackSchemaChangesAreObserved() + { + using var connection = new SqliteConnection("Data Source=:memory:;Local Provider=Managed"); + connection.Open(); + connection.ExecuteNonQuery($""" + CREATE TABLE t(id INTEGER PRIMARY KEY, a TEXT); + INSERT INTO t VALUES (1, '{Value}'); + """); + ReadFirstValue(connection).Should().Be(Value.ToString()); + + using (var transaction = connection.BeginTransaction()) + { + using (var command = connection.CreateCommand()) + { + command.Transaction = transaction; + command.CommandText = $""" + DROP TABLE t; + CREATE TABLE t(id INTEGER PRIMARY KEY, a GUID); + INSERT INTO t VALUES (1, '{Value}'); + """; + command.ExecuteNonQuery(); + command.CommandText = "SELECT a FROM t;"; + command.ExecuteScalar().Should().Be(Value); + } + + transaction.Rollback(); + } + + ReadFirstValue(connection).Should().Be(Value.ToString()); + } + + [Test] + public void SchemaChangesFromAnotherConnectionAreObserved() + { + var path = Path.Combine(_root, "shared.db"); + using var reader = new SqliteConnection($"Data Source={path}"); + reader.Open(); + reader.ExecuteNonQuery($""" + CREATE TABLE t(id INTEGER PRIMARY KEY, a TEXT); + INSERT INTO t VALUES (1, '{Value}'); + """); + ReadFirstValue(reader).Should().Be(Value.ToString()); + + using (var writer = new SqliteConnection($"Data Source={path}")) + { + writer.Open(); + writer.ExecuteNonQuery($""" + DROP TABLE t; + CREATE TABLE t(id INTEGER PRIMARY KEY, a GUID); + INSERT INTO t VALUES (1, '{Value}'); + """); + } + + ReadFirstValue(reader).Should().Be(Value); + } + + private static object ReadFirstValue(SqliteConnection connection) + { + using var command = connection.CreateCommand(); + command.CommandText = "SELECT id, a FROM t;"; + using var reader = command.ExecuteReader(); + reader.Read().Should().BeTrue(); + reader.GetDataTypeName(1).Should().Be(reader.GetValue(1) is Guid ? "GUID" : "TEXT"); + return reader.GetValue(1); + } + + private static List<(string Name, string Type, bool IsKey, bool IsUnique)> ReadSchema(SqliteConnection connection) + { + using var command = connection.CreateCommand(); + command.CommandText = "SELECT * FROM t;"; + using var reader = command.ExecuteReader(); + var schema = reader.GetSchemaTable(); + return schema.Rows.Cast() + .Select(row => ( + (string)row["ColumnName"], + (string)row["DataTypeName"], + row["IsKey"] is true, + row["IsUnique"] is true)) + .ToList(); + } +} diff --git a/src/Ahtola.Tests/PlannerAccessPathDepthTests.cs b/src/Ahtola.Tests/PlannerAccessPathDepthTests.cs index 0bd64d7..cf61a3f 100644 --- a/src/Ahtola.Tests/PlannerAccessPathDepthTests.cs +++ b/src/Ahtola.Tests/PlannerAccessPathDepthTests.cs @@ -207,7 +207,7 @@ public void Stat4CapturesSkewAndInvalidOrStaleSamplesFallBack() } [Test] - public void Stat4SampleValidationReusesOneRowIdMapPerTableRevision() + public void Stat4SampleValidationReusesOneRowIdMapPerRowIdSet() { using var database = new EmbeddedDatabase(); using var connection = database.Connect(); @@ -235,11 +235,20 @@ public void Stat4SampleValidationReusesOneRowIdMapPerTableRevision() database.PlannerAccessPathMetrics.Stat4SampleRowIdLookups.Should().Be(sampleCount * 2); database.PlannerAccessPathMetrics.Stat4RowIdCacheRebuilds.Should().Be(1); + // Positions depend only on the rowids: an in-place UPDATE keeps the map (sample keys are + // re-read from the live rows), while renumbering a row changes the rowid set and rebuilds it. Execute(connection, "UPDATE cached_stats SET payload='changed' WHERE id=1;"); database.ResetJoinOrderDiagnostics(); PlanDetail(connection, "SELECT id FROM cached_stats WHERE bucket=42;") .Should().Contain("cached_stats_bucket"); database.PlannerAccessPathMetrics.Stat4SampleRowIdLookups.Should().Be(sampleCount); + database.PlannerAccessPathMetrics.Stat4RowIdCacheRebuilds.Should().Be(0); + + Execute(connection, "UPDATE cached_stats SET id = 5000 WHERE id = 1;"); + database.ResetJoinOrderDiagnostics(); + PlanDetail(connection, "SELECT id FROM cached_stats WHERE bucket=42;") + .Should().Contain("cached_stats_bucket"); + database.PlannerAccessPathMetrics.Stat4SampleRowIdLookups.Should().BePositive(); database.PlannerAccessPathMetrics.Stat4RowIdCacheRebuilds.Should().Be(1); } diff --git a/src/Ahtola.Tests/TransactionDirectIndexOverlayTests.cs b/src/Ahtola.Tests/TransactionDirectIndexOverlayTests.cs index 091f167..899dd35 100644 --- a/src/Ahtola.Tests/TransactionDirectIndexOverlayTests.cs +++ b/src/Ahtola.Tests/TransactionDirectIndexOverlayTests.cs @@ -1,3 +1,4 @@ +using System.Collections.Immutable; using AwesomeAssertions; using Ahtola.Core; using Ahtola.Core.Storage; @@ -992,7 +993,7 @@ public void RestoreCheckpointRemovesOverlayEntriesForTablesAbsentFromTheCheckpoi { ["ghost"] = new TransactionTableOverlayCheckpoint( true, - new Dictionary { [1] = [SqlValue.Integer(1)] }, + new Dictionary { [1] = [SqlValue.Integer(1)] }.ToImmutableDictionary(), null), }); overlay.RestoreCheckpoint(seedCheckpoint); @@ -1040,7 +1041,7 @@ public void RestoreCheckpointReconstructsAMismatchedShapeOverlayFromTheCheckpoin { ["shifted"] = new TransactionTableOverlayCheckpoint( true, - new Dictionary { [5] = [SqlValue.Text("recreated")] }, + new Dictionary { [5] = [SqlValue.Text("recreated")] }.ToImmutableDictionary(), null), }); overlay.RestoreCheckpoint(rowidCheckpoint); diff --git a/src/Benchmarks/DotNetHotspotBenchmarks.cs b/src/Benchmarks/DotNetHotspotBenchmarks.cs new file mode 100644 index 0000000..9ebffab --- /dev/null +++ b/src/Benchmarks/DotNetHotspotBenchmarks.cs @@ -0,0 +1,643 @@ +using System.Data; +using System.Data.Common; +using System.Text; +using BenchmarkDotNet.Attributes; +using BenchmarkDotNet.Configs; +using AhtolaSqliteConnection = Ahtola.Data.Sqlite.SqliteConnection; +using MicrosoftSqliteConnection = Microsoft.Data.Sqlite.SqliteConnection; + +namespace Benchmarks; + +/// +/// The access patterns that dominate typical .NET applications over SQLite, measured +/// through the Microsoft.Data.Sqlite-compatible facade a migrating consumer uses +/// () against Microsoft.Data.Sqlite itself. +/// +/// +/// Every case runs against a file-backed WAL database with synchronous=NORMAL, the +/// configuration most .NET applications ship. The scenarios mirror what ADO.NET code, +/// Dapper and EF Core issue in practice: pooled connection-per-operation, a fresh +/// parameterized command per query, reused commands, typed and GetValue-based +/// materialization, unique-key lookups, paging, IN lists, aggregates over joins, +/// autocommit and batched writes, UPSERT and RETURNING, multi-statement batches and +/// blobs. Fixtures are seeded identically for both providers and validated before timing. +/// +[MemoryDiagnoser] +[CategoriesColumn] +[GroupBenchmarksBy(BenchmarkLogicalGroupRule.ByCategory)] +public class DotNetHotspotBenchmarks +{ + public enum Provider + { + Sqlite, + Ahtola, + } + + private const int ItemCount = 10_000; + private const int OrderCount = 20_000; + private const int CategoryCount = 100; + private const int BatchRows = 1_000; + private const int ReadAllRows = 1_000; + private const int InListSize = 50; + private const int BlobBytes = 64 * 1024; + + private string _root = null!; + private string _path = null!; + private string _connectionString = null!; + private DbConnection _connection = null!; + private DbCommand _reusedLookup = null!; + private DbParameter _reusedLookupId = null!; + private Guid[] _guids = null!; + private byte[] _blob = null!; + private int _cursor; + + [Params(Provider.Sqlite, Provider.Ahtola)] + public Provider Engine { get; set; } + + [GlobalSetup] + public void GlobalSetup() + { + _root = Path.Combine(Path.GetTempPath(), "ahtola-hotspot-bench-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(_root); + _path = Path.Combine(_root, "app.db"); + _connectionString = $"Data Source={_path}"; + _guids = new Guid[ItemCount]; + var random = new Random(1234); + Span bytes = stackalloc byte[16]; + for (var index = 0; index < _guids.Length; index++) + { + random.NextBytes(bytes); + _guids[index] = new Guid(bytes); + } + + _blob = new byte[BlobBytes]; + random.NextBytes(_blob); + + // Both engines open the same SQLite-written file, so the fixture is byte-identical and + // its construction cost stays out of setup (bulk loading is measured by its own case). + using (var seed = new MicrosoftSqliteConnection($"Data Source={_path};Pooling=False")) + { + seed.Open(); + Seed(seed); + } + + _connection = CreateConnection(); + _connection.Open(); + Execute(_connection, "PRAGMA synchronous=NORMAL;"); + Validate(_connection); + + _reusedLookup = _connection.CreateCommand(); + _reusedLookup.CommandText = "SELECT id, guid, name, category, price, created FROM items WHERE id = @id;"; + _reusedLookupId = AddParameter(_reusedLookup, "@id", 1L); + _reusedLookup.Prepare(); + } + + [GlobalCleanup] + public void GlobalCleanup() + { + _reusedLookup.Dispose(); + _connection.Dispose(); + AhtolaSqliteConnection.ClearAllPools(); + MicrosoftSqliteConnection.ClearAllPools(); + try + { + Directory.Delete(_root, recursive: true); + } + catch (IOException) + { + } + catch (UnauthorizedAccessException) + { + } + } + + [IterationSetup(Target = nameof(InsertBatchInTransaction))] + public void ResetLog() => Execute(_connection, "DELETE FROM log;"); + + // ---- Connection lifecycle ------------------------------------------------------------ + + [BenchmarkCategory("Connection")] + [Benchmark(Description = "Pooled open + close")] + public int PooledOpenClose() + { + using var connection = CreateConnection(); + connection.Open(); + return connection.State == ConnectionState.Open ? 1 : 0; + } + + [BenchmarkCategory("Connection")] + [Benchmark(Description = "Pooled open + PK lookup + close")] + public long PooledOpenLookupClose() + { + using var connection = CreateConnection(); + connection.Open(); + return LookupById(connection, NextId()); + } + + // ---- Point reads --------------------------------------------------------------------- + + [BenchmarkCategory("PointRead")] + [Benchmark(Description = "PK lookup, new command per call")] + public long LookupByIdNewCommand() => LookupById(_connection, NextId()); + + [BenchmarkCategory("PointRead")] + [Benchmark(Description = "PK lookup, reused prepared command")] + public long LookupByIdReusedCommand() + { + _reusedLookupId.Value = (long)NextId(); + using var reader = _reusedLookup.ExecuteReader(); + return reader.Read() ? reader.GetInt64(0) + reader.GetString(2).Length : 0; + } + + [BenchmarkCategory("PointRead")] + [Benchmark(Description = "PK lookup, Dapper-style GetValue mapping")] + public long LookupByIdDapperStyle() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT id, guid, name, category, price, created, notes FROM items WHERE id = @id;"; + AddParameter(command, "@id", (long)NextId()); + using var reader = command.ExecuteReader(); + long checksum = 0; + while (reader.Read()) + { + for (var ordinal = 0; ordinal < reader.FieldCount; ordinal++) + checksum += reader.IsDBNull(ordinal) ? 0 : reader.GetValue(ordinal).GetHashCode() & 0xFF; + } + + return checksum; + } + + [BenchmarkCategory("PointRead")] + [Benchmark(Description = "Unique TEXT (GUID) key lookup")] + public long LookupByGuid() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT id, name, price FROM items WHERE guid = @guid;"; + AddParameter(command, "@guid", _guids[NextId() - 1].ToString()); + using var reader = command.ExecuteReader(); + return reader.Read() ? reader.GetInt64(0) : 0; + } + + [BenchmarkCategory("PointRead")] + [Benchmark(Description = "ExecuteScalar COUNT(*) on index")] + public long ScalarCountByCategory() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT COUNT(*) FROM items WHERE category = @category;"; + AddParameter(command, "@category", (long)(NextId() % CategoryCount)); + return Convert.ToInt64(command.ExecuteScalar()); + } + + [BenchmarkCategory("PointRead")] + [Benchmark(Description = "EF-style schema probe on sqlite_master")] + public long SchemaProbe() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT COUNT(*) FROM \"sqlite_master\" WHERE \"name\" = @name AND \"type\" = 'table' AND \"rootpage\" IS NOT NULL;"; + AddParameter(command, "@name", "items"); + return Convert.ToInt64(command.ExecuteScalar()); + } + + // ---- Multi-row reads ----------------------------------------------------------------- + + [BenchmarkCategory("Materialize")] + [Benchmark(Description = "1,000 rows, typed getters")] + public long ReadRowsTyped() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT id, guid, name, category, price, created, notes FROM items WHERE id <= @limit;"; + AddParameter(command, "@limit", (long)ReadAllRows); + using var reader = command.ExecuteReader(); + long checksum = 0; + while (reader.Read()) + { + checksum += reader.GetInt64(0) + + reader.GetString(1).Length + + reader.GetString(2).Length + + reader.GetInt32(3) + + (long)reader.GetDouble(4) + + reader.GetString(5).Length + + (reader.IsDBNull(6) ? 0 : reader.GetString(6).Length); + } + + return checksum; + } + + [BenchmarkCategory("Materialize")] + [Benchmark(Description = "1,000 rows, Dapper-style GetValue mapping")] + public long ReadRowsDapperStyle() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT id, guid, name, category, price, created, notes FROM items WHERE id <= @limit;"; + AddParameter(command, "@limit", (long)ReadAllRows); + using var reader = command.ExecuteReader(); + var fieldCount = reader.FieldCount; + long checksum = 0; + for (var ordinal = 0; ordinal < fieldCount; ordinal++) + checksum += reader.GetName(ordinal).Length + reader.GetFieldType(ordinal).Name.Length; + var values = new object[fieldCount]; + while (reader.Read()) + { + for (var ordinal = 0; ordinal < fieldCount; ordinal++) + values[ordinal] = reader.IsDBNull(ordinal) ? DBNull.Value : reader.GetValue(ordinal); + checksum += (long)values[0] + ((string)values[2]).Length; + } + + return checksum; + } + + [BenchmarkCategory("Materialize")] + [Benchmark(Description = "1,000 rows, GetFieldValue by GetOrdinal")] + public long ReadRowsGetFieldValue() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT id, guid, name, category, price, created FROM items WHERE id <= @limit;"; + AddParameter(command, "@limit", (long)ReadAllRows); + using var reader = command.ExecuteReader(); + var id = reader.GetOrdinal("id"); + var guid = reader.GetOrdinal("guid"); + var price = reader.GetOrdinal("price"); + var created = reader.GetOrdinal("created"); + long checksum = 0; + while (reader.Read()) + { + checksum += reader.GetFieldValue(id) + + reader.GetFieldValue(guid).GetHashCode() + + (long)reader.GetFieldValue(price) + + reader.GetFieldValue(created).Day; + } + + return checksum; + } + + [BenchmarkCategory("Query")] + [Benchmark(Description = "Indexed equality, ~100 rows")] + public long IndexedEquality() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT id, name, price FROM items WHERE category = @category;"; + AddParameter(command, "@category", (long)(NextId() % CategoryCount)); + return Drain(command); + } + + [BenchmarkCategory("Query")] + [Benchmark(Description = "Paging ORDER BY id LIMIT 50 OFFSET n")] + public long PagedOrderByLimitOffset() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT id, name, price FROM items ORDER BY id LIMIT @take OFFSET @skip;"; + AddParameter(command, "@take", 50L); + AddParameter(command, "@skip", (long)(NextId() % 100 * 50)); + return Drain(command); + } + + [BenchmarkCategory("Query")] + [Benchmark(Description = "Keyset ORDER BY indexed col DESC LIMIT 20")] + public long KeysetOrderByIndexDesc() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT id, name, created FROM items WHERE created < @before ORDER BY created DESC LIMIT 20;"; + AddParameter(command, "@before", CreatedAt(NextId())); + return Drain(command); + } + + [BenchmarkCategory("Query")] + [Benchmark(Description = "WHERE id IN (50 parameters)")] + public long InListParameters() + { + using var command = _connection.CreateCommand(); + var sql = new StringBuilder("SELECT id, name FROM items WHERE id IN ("); + var start = NextId() % (ItemCount - InListSize * 3); + for (var index = 0; index < InListSize; index++) + { + if (index > 0) + sql.Append(", "); + sql.Append("@p").Append(index); + AddParameter(command, "@p" + index, (long)(start + index * 3 + 1)); + } + + sql.Append(");"); + command.CommandText = sql.ToString(); + return Drain(command); + } + + [BenchmarkCategory("Query")] + [Benchmark(Description = "LIKE prefix search LIMIT 20")] + public long LikePrefix() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT id, name FROM items WHERE name LIKE @pattern LIMIT 20;"; + AddParameter(command, "@pattern", "Item 9" + (NextId() % 10) + "%"); + return Drain(command); + } + + [BenchmarkCategory("Query")] + [Benchmark(Description = "JOIN + GROUP BY aggregate")] + public long JoinGroupBy() + { + using var command = _connection.CreateCommand(); + command.CommandText = + """ + SELECT i.category, COUNT(*), SUM(o.quantity) + FROM orders AS o JOIN items AS i ON i.id = o.item_id + WHERE i.category < @maxCategory + GROUP BY i.category + ORDER BY i.category; + """; + AddParameter(command, "@maxCategory", 5L); + return Drain(command); + } + + [BenchmarkCategory("Query")] + [Benchmark(Description = "Correlated lookup: orders of one item")] + public long ForeignKeyLookup() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT o.id, o.quantity, i.name FROM orders AS o JOIN items AS i ON i.id = o.item_id WHERE o.item_id = @itemId;"; + AddParameter(command, "@itemId", (long)NextId()); + return Drain(command); + } + + [BenchmarkCategory("Query")] + [Benchmark(Description = "Two result sets via NextResult")] + public long MultipleResultSets() + { + using var command = _connection.CreateCommand(); + command.CommandText = "SELECT COUNT(*) FROM items WHERE category = @category; SELECT id, name FROM items WHERE category = @category ORDER BY id LIMIT 10;"; + AddParameter(command, "@category", (long)(NextId() % CategoryCount)); + using var reader = command.ExecuteReader(); + long checksum = 0; + do + { + while (reader.Read()) + checksum += reader.GetInt64(0); + } + while (reader.NextResult()); + + return checksum; + } + + // ---- Writes -------------------------------------------------------------------------- + + [BenchmarkCategory("Write")] + [Benchmark(Description = "Autocommit single-row INSERT")] + public int InsertAutocommit() + { + using var command = _connection.CreateCommand(); + command.CommandText = "INSERT INTO log(at, level, message) VALUES (@at, @level, @message);"; + AddParameter(command, "@at", "2026-01-01T00:00:00"); + AddParameter(command, "@level", 2L); + AddParameter(command, "@message", "request completed"); + return command.ExecuteNonQuery(); + } + + [BenchmarkCategory("Write")] + [Benchmark(Description = "1,000-row INSERT in transaction, reused command", OperationsPerInvoke = BatchRows)] + public int InsertBatchInTransaction() + { + using var transaction = _connection.BeginTransaction(); + using var command = _connection.CreateCommand(); + command.Transaction = transaction; + command.CommandText = "INSERT INTO log(at, level, message) VALUES (@at, @level, @message);"; + var at = AddParameter(command, "@at", string.Empty); + var level = AddParameter(command, "@level", 0L); + var message = AddParameter(command, "@message", string.Empty); + command.Prepare(); + var rows = 0; + for (var row = 0; row < BatchRows; row++) + { + at.Value = CreatedAt(row); + level.Value = (long)(row % 5); + message.Value = "batch message " + row; + rows += command.ExecuteNonQuery(); + } + + transaction.Commit(); + return rows; + } + + [BenchmarkCategory("Write")] + [Benchmark(Description = "Autocommit UPDATE by PK")] + public int UpdateByPrimaryKey() + { + using var command = _connection.CreateCommand(); + command.CommandText = "UPDATE items SET price = @price WHERE id = @id;"; + var id = NextId(); + AddParameter(command, "@price", id * 1.25); + AddParameter(command, "@id", (long)id); + return command.ExecuteNonQuery(); + } + + [BenchmarkCategory("Write")] + [Benchmark(Description = "Autocommit UPSERT ON CONFLICT DO UPDATE")] + public int Upsert() + { + using var command = _connection.CreateCommand(); + command.CommandText = "INSERT INTO settings(key, value) VALUES (@key, @value) ON CONFLICT(key) DO UPDATE SET value = excluded.value;"; + var id = NextId(); + AddParameter(command, "@key", "setting-" + (id % 500)); + AddParameter(command, "@value", "value-" + id); + return command.ExecuteNonQuery(); + } + + [BenchmarkCategory("Write")] + [Benchmark(Description = "EF-style unit of work: BEGIN, UPDATE, INSERT RETURNING, COMMIT")] + public long UnitOfWorkReturning() + { + using var transaction = _connection.BeginTransaction(); + long id; + using (var update = _connection.CreateCommand()) + { + update.Transaction = transaction; + update.CommandText = "UPDATE items SET notes = @notes WHERE id = @id; SELECT changes();"; + AddParameter(update, "@notes", "touched"); + AddParameter(update, "@id", (long)NextId()); + update.ExecuteScalar(); + } + + using (var insert = _connection.CreateCommand()) + { + insert.Transaction = transaction; + insert.CommandText = "INSERT INTO orders(item_id, quantity) VALUES (@itemId, @quantity) RETURNING id;"; + AddParameter(insert, "@itemId", (long)NextId()); + AddParameter(insert, "@quantity", 3L); + id = Convert.ToInt64(insert.ExecuteScalar()); + } + + using (var delete = _connection.CreateCommand()) + { + delete.Transaction = transaction; + delete.CommandText = "DELETE FROM orders WHERE id = @id;"; + AddParameter(delete, "@id", id); + delete.ExecuteNonQuery(); + } + + transaction.Commit(); + return id; + } + + [BenchmarkCategory("Write")] + [Benchmark(Description = "64 KiB BLOB write + read back")] + public long BlobRoundTrip() + { + var key = "blob-" + (NextId() % 16); + using (var write = _connection.CreateCommand()) + { + write.CommandText = "INSERT OR REPLACE INTO blobs(key, data) VALUES (@key, @data);"; + AddParameter(write, "@key", key); + AddParameter(write, "@data", _blob); + write.ExecuteNonQuery(); + } + + using var read = _connection.CreateCommand(); + read.CommandText = "SELECT data FROM blobs WHERE key = @key;"; + AddParameter(read, "@key", key); + var data = (byte[])read.ExecuteScalar()!; + return data.Length; + } + + // ---- Helpers ------------------------------------------------------------------------- + + private DbConnection CreateConnection() + => Engine switch + { + Provider.Sqlite => new MicrosoftSqliteConnection(_connectionString), + Provider.Ahtola => new AhtolaSqliteConnection(_connectionString), + _ => throw new InvalidOperationException(), + }; + + private int NextId() + { + _cursor = (_cursor * 7919 + 13) % ItemCount; + return _cursor + 1; + } + + private static string CreatedAt(int row) + => new DateTime(2024, 1, 1, 0, 0, 0, DateTimeKind.Utc).AddMinutes(row).ToString("yyyy-MM-dd HH:mm:ss"); + + private static long LookupById(DbConnection connection, int id) + { + using var command = connection.CreateCommand(); + command.CommandText = "SELECT id, guid, name, category, price, created FROM items WHERE id = @id;"; + AddParameter(command, "@id", (long)id); + using var reader = command.ExecuteReader(); + return reader.Read() ? reader.GetInt64(0) + reader.GetString(2).Length : 0; + } + + private static long Drain(DbCommand command) + { + using var reader = command.ExecuteReader(); + long checksum = 0; + while (reader.Read()) + checksum += reader.GetInt64(0) + 1; + return checksum; + } + + private void Seed(DbConnection connection) + { + Execute(connection, "PRAGMA journal_mode=WAL;"); + Execute( + connection, + """ + CREATE TABLE items( + id INTEGER PRIMARY KEY, + guid TEXT NOT NULL UNIQUE, + name TEXT NOT NULL, + category INTEGER NOT NULL, + price REAL NOT NULL, + created TEXT NOT NULL, + notes TEXT NULL); + """); + Execute(connection, "CREATE INDEX ix_items_category ON items(category);"); + Execute(connection, "CREATE INDEX ix_items_created ON items(created);"); + Execute( + connection, + "CREATE TABLE orders(id INTEGER PRIMARY KEY, item_id INTEGER NOT NULL REFERENCES items(id), quantity INTEGER NOT NULL);"); + Execute(connection, "CREATE INDEX ix_orders_item ON orders(item_id);"); + Execute(connection, "CREATE TABLE log(id INTEGER PRIMARY KEY, at TEXT NOT NULL, level INTEGER NOT NULL, message TEXT NOT NULL);"); + Execute(connection, "CREATE TABLE settings(key TEXT PRIMARY KEY, value TEXT NOT NULL);"); + Execute(connection, "CREATE TABLE blobs(key TEXT PRIMARY KEY, data BLOB NOT NULL);"); + + using var transaction = connection.BeginTransaction(); + using (var insert = connection.CreateCommand()) + { + insert.Transaction = transaction; + insert.CommandText = "INSERT INTO items(id, guid, name, category, price, created, notes) VALUES (@id, @guid, @name, @category, @price, @created, @notes);"; + var id = AddParameter(insert, "@id", 0L); + var guid = AddParameter(insert, "@guid", string.Empty); + var name = AddParameter(insert, "@name", string.Empty); + var category = AddParameter(insert, "@category", 0L); + var price = AddParameter(insert, "@price", 0.0); + var created = AddParameter(insert, "@created", string.Empty); + var notes = AddParameter(insert, "@notes", DBNull.Value); + for (var row = 1; row <= ItemCount; row++) + { + id.Value = (long)row; + guid.Value = _guids[row - 1].ToString(); + name.Value = "Item " + row; + category.Value = (long)(row % CategoryCount); + price.Value = row * 0.5; + created.Value = CreatedAt(row); + notes.Value = row % 3 == 0 ? DBNull.Value : "note " + row; + insert.ExecuteNonQuery(); + } + } + + using (var insert = connection.CreateCommand()) + { + insert.Transaction = transaction; + insert.CommandText = "INSERT INTO orders(item_id, quantity) VALUES (@itemId, @quantity);"; + var itemId = AddParameter(insert, "@itemId", 0L); + var quantity = AddParameter(insert, "@quantity", 0L); + for (var row = 0; row < OrderCount; row++) + { + itemId.Value = (long)(row * 31 % ItemCount + 1); + quantity.Value = (long)(row % 7 + 1); + insert.ExecuteNonQuery(); + } + } + + transaction.Commit(); + Execute(connection, "ANALYZE;"); + } + + private static void Validate(DbConnection connection) + { + Require(connection, "SELECT COUNT(*) FROM items;", ItemCount); + Require(connection, "SELECT COUNT(*) FROM orders;", OrderCount); + long expectedQuantity = 0; + for (var row = 0; row < OrderCount; row++) + { + if (row * 31 % ItemCount == 0) + expectedQuantity += row % 7 + 1; + } + + Require(connection, "SELECT SUM(quantity) FROM orders WHERE item_id = 1;", expectedQuantity); + Require(connection, "SELECT COUNT(*) FROM items WHERE category = 7;", ItemCount / CategoryCount); + } + + private static void Require(DbConnection connection, string sql, long expected) + { + using var command = connection.CreateCommand(); + command.CommandText = sql; + var actual = Convert.ToInt64(command.ExecuteScalar()); + if (actual != expected) + throw new InvalidOperationException($"Fixture validation failed for '{sql}': expected {expected}, got {actual}."); + } + + private static DbParameter AddParameter(DbCommand command, string name, object value) + { + var parameter = command.CreateParameter(); + parameter.ParameterName = name; + parameter.Value = value; + command.Parameters.Add(parameter); + return parameter; + } + + private static void Execute(DbConnection connection, string sql) + { + using var command = connection.CreateCommand(); + command.CommandText = sql; + command.ExecuteNonQuery(); + } +} diff --git a/src/Benchmarks/HotspotQuickRunner.cs b/src/Benchmarks/HotspotQuickRunner.cs new file mode 100644 index 0000000..4c88dbd --- /dev/null +++ b/src/Benchmarks/HotspotQuickRunner.cs @@ -0,0 +1,123 @@ +using System.Diagnostics; +using System.Reflection; +using BenchmarkDotNet.Attributes; + +namespace Benchmarks; + +/// +/// In-process stopwatch loop over for the +/// optimize-measure cycle: every case runs for a fixed time budget on both providers and +/// prints the Ahtola/SQLite ratio. BenchmarkDotNet remains the source of record; this is a +/// development aid (dotnet run -c Release -- --hotspots-quick [filter] [milliseconds] [Sqlite|Ahtola]). +/// +internal static class HotspotQuickRunner +{ + public const string Switch = "--hotspots-quick"; + + public static void Run(string[] arguments) + { + var filter = arguments.Length > 1 ? arguments[1] : string.Empty; + var budget = TimeSpan.FromMilliseconds(arguments.Length > 2 ? int.Parse(arguments[2]) : 400); + var engines = arguments.Length > 3 + ? [Enum.Parse(arguments[3], ignoreCase: true)] + : Enum.GetValues(); + var methods = typeof(DotNetHotspotBenchmarks) + .GetMethods(BindingFlags.Public | BindingFlags.Instance) + .Where(method => method.GetCustomAttribute() is not null) + .Where(method => method.Name.Contains(filter, StringComparison.OrdinalIgnoreCase)) + .ToArray(); + + var results = new Dictionary<(string Method, DotNetHotspotBenchmarks.Provider Engine), (double Mean, double Allocated)>(); + foreach (var engine in engines) + { + var benchmarks = new DotNetHotspotBenchmarks { Engine = engine }; + benchmarks.GlobalSetup(); + try + { + foreach (var method in methods) + { + var result = Measure(benchmarks, method, budget); + results[(method.Name, engine)] = result; + Console.Error.WriteLine($" {engine,-6} {method.Name,-28} {FormatTime(result.Mean),12}"); + } + } + finally + { + benchmarks.GlobalCleanup(); + } + } + + if (engines.Length == 1) + return; + + Console.WriteLine($"{"Case",-28} {"SQLite",12} {"Ahtola",12} {"Ratio",8} {"Alloc SQLite",14} {"Alloc Ahtola",14}"); + foreach (var method in methods) + { + var sqlite = results[(method.Name, DotNetHotspotBenchmarks.Provider.Sqlite)]; + var ahtola = results[(method.Name, DotNetHotspotBenchmarks.Provider.Ahtola)]; + Console.WriteLine( + $"{method.Name,-28} {FormatTime(sqlite.Mean),12} {FormatTime(ahtola.Mean),12} {ahtola.Mean / sqlite.Mean,8:F2} {FormatBytes(sqlite.Allocated),14} {FormatBytes(ahtola.Allocated),14}"); + } + } + + private static (double Mean, double Allocated) Measure(DotNetHotspotBenchmarks benchmarks, MethodInfo method, TimeSpan budget) + { + var attribute = method.GetCustomAttribute()!; + var operations = Math.Max(1, attribute.OperationsPerInvoke); + var iterationSetup = typeof(DotNetHotspotBenchmarks) + .GetMethods() + .FirstOrDefault(candidate => candidate.GetCustomAttribute() is { } setup + && setup.Targets.Contains(method.Name)); + Action invoke = method.ReturnType == typeof(int) + ? CreateInvoker(method.CreateDelegate>(benchmarks)) + : method.ReturnType == typeof(long) + ? CreateInvoker(method.CreateDelegate>(benchmarks)) + : throw new NotSupportedException($"{method.Name} returns {method.ReturnType}."); + + // Warm up the JIT, statement caches and the page cache. + var warmup = Stopwatch.StartNew(); + while (warmup.Elapsed < budget / 4) + { + iterationSetup?.Invoke(benchmarks, null); + invoke(); + } + + var samples = new List(); + long allocated = 0; + long invocations = 0; + var total = Stopwatch.StartNew(); + while (total.Elapsed < budget || samples.Count < 3) + { + iterationSetup?.Invoke(benchmarks, null); + var before = GC.GetAllocatedBytesForCurrentThread(); + var start = Stopwatch.GetTimestamp(); + invoke(); + samples.Add(Stopwatch.GetElapsedTime(start).TotalNanoseconds / operations); + allocated += GC.GetAllocatedBytesForCurrentThread() - before; + invocations++; + } + + samples.Sort(); + // Median resists GC and checkpoint outliers better than the mean in a short loop. + return (samples[samples.Count / 2], (double)allocated / invocations / operations); + } + + private static Action CreateInvoker(Func benchmark) + => () => GC.KeepAlive(benchmark()); + + private static string FormatTime(double nanoseconds) + => nanoseconds switch + { + >= 1_000_000 => $"{nanoseconds / 1_000_000:F2} ms", + >= 1_000 => $"{nanoseconds / 1_000:F2} us", + _ => $"{nanoseconds:F0} ns", + }; + + private static string FormatBytes(double bytes) + => bytes switch + { + >= 1024 * 1024 => $"{bytes / (1024 * 1024):F1} MB", + >= 1024 => $"{bytes / 1024:F1} KB", + _ => $"{bytes:F0} B", + }; +} diff --git a/src/Benchmarks/Program.cs b/src/Benchmarks/Program.cs index 0a329cc..873812b 100644 --- a/src/Benchmarks/Program.cs +++ b/src/Benchmarks/Program.cs @@ -1,4 +1,11 @@ using BenchmarkDotNet.Running; +using Benchmarks; + +if (args.Length > 0 && args[0] == HotspotQuickRunner.Switch) +{ + HotspotQuickRunner.Run(args); + return; +} BenchmarkSwitcher.FromAssembly(typeof(Program).Assembly).Run(args); diff --git a/src/Benchmarks/README.md b/src/Benchmarks/README.md index 3abfef5..c7a65e6 100644 --- a/src/Benchmarks/README.md +++ b/src/Benchmarks/README.md @@ -31,6 +31,28 @@ Results are written below `artifacts/benchmarks//`, including raw BDN reports, normalized JSON, environment metadata, and a historical comparison when `-BenchmarkBaseline` names a prior normalized result. +## .NET hotspots + +`DotNetHotspotBenchmarks` measures the access patterns typical ADO.NET, Dapper and +EF Core applications issue, through the `Microsoft.Data.Sqlite`-compatible facade +against `Microsoft.Data.Sqlite` on the same SQLite-written WAL file +(`synchronous=NORMAL`). Each case is parameterized by `Engine` (`Sqlite`/`Ahtola`): + +```powershell +./build.ps1 benchmark -BenchmarkProfile coverage -BenchmarkFilter '*DotNetHotspot*' +``` + +For the optimize-measure loop, an in-process stopwatch runner prints the +Ahtola/SQLite ratio of every case in about a minute (optionally filtered by case +name, with a per-case time budget in milliseconds and a single engine): + +```powershell +dotnet run -c Release -f net10.0 --project src/Benchmarks/Ahtola.Benchmarks.csproj -- --hotspots-quick +dotnet run -c Release -f net10.0 --project src/Benchmarks/Ahtola.Benchmarks.csproj -- --hotspots-quick Lookup 2000 Ahtola +``` + +It is a development aid: BenchmarkDotNet results remain the source of record. + ## Interpretation - Native SQLite ratios provide context; they are not a regression gate.