From 1ac41d10944bf3321f69a030139733534db6c3d9 Mon Sep 17 00:00:00 2001 From: Widthdom Date: Wed, 23 Sep 2026 04:35:12 +0900 Subject: [PATCH 1/2] Fix bounded line-local find chunk reads (#5415) --- TESTING_GUIDE.md | 14 ++ changelog.d/unreleased/5415.fixed.md | 19 ++ docs/find-scan-controls.md | 21 +++ src/CodeIndex/Database/IndexedFindPipeline.cs | 2 +- .../Database/IndexedFindQuerySource.cs | 92 +++++++--- tests/CodeIndex.Tests/FindChunkReadTests.cs | 167 ++++++++++++++++++ 6 files changed, 293 insertions(+), 22 deletions(-) create mode 100644 changelog.d/unreleased/5415.fixed.md create mode 100644 tests/CodeIndex.Tests/FindChunkReadTests.cs diff --git a/TESTING_GUIDE.md b/TESTING_GUIDE.md index 98a27d1bac..8658ab5c0b 100644 --- a/TESTING_GUIDE.md +++ b/TESTING_GUIDE.md @@ -1,5 +1,12 @@ # Testing Guide +`FindChunkReadTests` (#5415) shares a many-chunk fixture across literal, exact, +trigram-candidate and line-regex row/count queries on current and read-only legacy +indexes. Keep ordered query-plan checks, bounded metadata pages, reversed insertion, +overlap ownership, missing/live source controls, NULL-content failures, line caps, +and row/count continuation across chunk pages. Run on net8/net9 alongside existing +find, pagination and MCP regressions; no peak-memory or latency claim is implied. + `RunDeps_PythonContextCoordinatesPreserveBothDirections_Issue5401` shares indexed fixtures across ordinary dependencies and cycles for zero/space/tab indentation, short/long names, aliases, relative imports and repeated same-line member calls. @@ -1814,6 +1821,13 @@ Issue #5300 のテストは隣接・入れ子の C# callable、対象行の除 # テストガイド +`FindChunkReadTests`(#5415)は多数のチャンクを持つ共通 fixture を使い、現在の索引と +読取専用の旧索引でリテラル・exact・trigram 候補・行単位正規表現の行/件数検索を検証します。 +順序付きクエリ計画、上限付きメタデータページ、逆順挿入、重複行の優先順、索引本文の欠落と +実ファイルの対照、NULL 本文エラー、行上限、チャンクページをまたぐ行/件数の継続取得を +維持してください。既存の find・ページ分割・MCP 回帰とともに net8/net9 で実行します。 +ピークメモリや実行時間の測定結果を保証するテストではありません。 + `FindMultilineTests` と `McpServerIssue5399Tests` は、索引の LF/CRLF 窓、補助平面文字の UTF-16 座標、重複チャンク、一致範囲の上限内外、アンカー、dot-all、ゼロ幅、非重複、件数と行の 再開、カーソル条件変更、読取上限、取消・タイムアウト、投影と応答サイズを検証します。 diff --git a/changelog.d/unreleased/5415.fixed.md b/changelog.d/unreleased/5415.fixed.md new file mode 100644 index 0000000000..898a57444c --- /dev/null +++ b/changelog.d/unreleased/5415.fixed.md @@ -0,0 +1,19 @@ +--- +category: fixed +issues: + - 5415 +affected: + - src/CodeIndex/Database/IndexedFindQuerySource.cs + - src/CodeIndex/Database/IndexedFindPipeline.cs + - tests/CodeIndex.Tests/FindChunkReadTests.cs + - docs/find-scan-controls.md + - TESTING_GUIDE.md +--- + +## English + +- **Line-local find no longer sorts source content (#5415)** — literal and line-regex queries read chunks in index order, with bounded metadata pages for older indexes. Source content is fetched separately per chunk, preserving overlap ownership, missing-source behavior, counts, scan limits and continuation without requiring reindexing. + +## 日本語 + +- **行単位の find で本文をソートしなくなりました (#5415)** — リテラル・行単位正規表現検索は索引順にチャンクを読み、旧索引では上限付きのメタデータページを使います。本文はチャンクごとに別途取得し、重複行の優先順、本文欠落時の挙動、件数、走査上限、継続取得を維持します。再索引は不要です。 diff --git a/docs/find-scan-controls.md b/docs/find-scan-controls.md index 9a9865d53f..75b3aa2818 100644 --- a/docs/find-scan-controls.md +++ b/docs/find-scan-controls.md @@ -2,6 +2,17 @@ ## English +Literal and line-regex `find` read indexed chunks in line/chunk order without +sorting source content. Current indexes use the ordered partial chunk index; +older indexes page at most 64 scalar chunk records through a metadata-only sort, +then read each selected chunk's content separately. Pages use the last ordering +key rather than an increasing offset. Retained sort data is bounded, but legacy +pages may rescan file metadata and one chunk's content is still read at a time. +A bounded-memory NULL-content check preserves the existing missing-content error +by selecting the all-row fallback for such files; that check may scan the file's +chunk entries. Missing chunks never trigger live-file reads. Overlap ownership, +counts, line caps and continuation positions are unchanged; no reindex is required. + Regex is line-local by default. To match adjacent lines such as `A\nB`, use `--regex --multiline` (MCP `regex:true, multiline:true`); see [bounded multiline windows](find-multiline.md#english). Window mode rejects semantic @@ -196,6 +207,16 @@ text or JSON output when context from `--before`, `--after`, or ## 日本語 +リテラル検索と行単位の正規表現 `find` は、本文をソートせず行・チャンク順に索引を読みます。 +現在の索引では順序付き部分インデックスを使い、旧索引では最大64件のチャンクメタデータだけを +ソートしてから、選択した各チャンクの本文を個別に取得します。ページは増大する offset ではなく +直前の順序キーから再開します。ソートの保持量には上限がありますが、旧索引ではページごとに +ファイルのメタデータを再走査する場合があり、本文もチャンク1個単位では読み取ります。 +NULL 本文の確認は保持メモリを制限して行い、該当するファイルでは全行を対象とする代替経路により +従来の本文欠落エラーを維持します。この確認もファイルのチャンク行を走査する場合があります。 +チャンクが欠けても実ファイルへ読み取りを切り替えません。重複行の優先順、件数、行上限、 +継続位置は従来どおりで、再索引は不要です。 + 通常の正規表現は行単位です。`A\nB` のような隣接行には `--regex --multiline` (MCP は `regex:true, multiline:true`)を使います。[上限付きの複数行窓](find-multiline.md#日本語)を 参照してください。窓モードは意味フィルターを明示的に拒否し、origin をまたぐ証拠を黙って省略しません。 diff --git a/src/CodeIndex/Database/IndexedFindPipeline.cs b/src/CodeIndex/Database/IndexedFindPipeline.cs index 7de7e27cc9..06cf099163 100644 --- a/src/CodeIndex/Database/IndexedFindPipeline.cs +++ b/src/CodeIndex/Database/IndexedFindPipeline.cs @@ -139,7 +139,7 @@ private bool ScanFileLines( var origins = request.SemanticFilters is null ? null : _owner.CreateFindOriginContext(file, request.CancellationToken); var firstContextLine = Math.Max(1, file.FirstEligibleLine - collector.ContextBefore); var stopScanning = false; - foreach (var indexedLine in _querySource.EnumerateIndexedFileLines(file.Id)) + foreach (var indexedLine in _querySource.EnumerateIndexedFileLines(file.Id, request.CancellationToken)) { request.CancellationToken.ThrowIfCancellationRequested(); if (indexedLine.Number > file.TotalLines) diff --git a/src/CodeIndex/Database/IndexedFindQuerySource.cs b/src/CodeIndex/Database/IndexedFindQuerySource.cs index 79bf3151fb..a6f627b140 100644 --- a/src/CodeIndex/Database/IndexedFindQuerySource.cs +++ b/src/CodeIndex/Database/IndexedFindQuerySource.cs @@ -5,6 +5,18 @@ namespace CodeIndex.Database; public partial class DbReader { + internal const int FindChunkPageSize = 64; + + internal static string FindLineChunkPageSql(bool ordered, bool resumed) => $""" + SELECT c.id, c.start_line, c.end_line, c.chunk_index + FROM chunks c{(ordered ? $" INDEXED BY {BoundedResourceReadChunkIndexName}" : "")} + WHERE c.file_id = @fileId + {(ordered ? "AND c.content IS NOT NULL" : "")} + {(resumed ? "AND (c.start_line, c.chunk_index) > (@startLine, @chunkIndex)" : "")} + ORDER BY c.start_line, c.chunk_index + LIMIT {FindChunkPageSize} + """; + private sealed class IndexedFindQuerySource(DbReader owner) { private readonly DbReader _owner = owner; @@ -85,37 +97,75 @@ SELECT DISTINCT find_chunk.file_id return command; } - internal IEnumerable EnumerateIndexedFileLines(long fileId) + internal IEnumerable EnumerateIndexedFileLines(long fileId, CancellationToken cancellationToken) { + // The partial index excludes NULL source. Keep the original read failure for + // such rows by routing that file through the all-row metadata fallback. + // 部分索引から除外される NULL 本文は、全行のメタデータ経路で従来の読取エラーを維持する。 + var ordered = _owner._chunkIndexes.Contains(BoundedResourceReadChunkIndexName) + && !HasNullChunkContent(fileId); + using var pageCommand = _owner._conn.CreateCommand(); + SqliteCommandPolicy.Add(pageCommand, "@fileId", fileId); + var startLine = pageCommand.Parameters.AddWithValue("@startLine", 0); + var chunkIndex = pageCommand.Parameters.AddWithValue("@chunkIndex", 0); using var command = _owner._conn.CreateCommand(); - command.CommandText = @" - SELECT c.start_line, c.end_line, c.content - FROM chunks c - WHERE c.file_id = @fileId - ORDER BY c.start_line, c.chunk_index"; - SqliteCommandPolicy.Add(command, "@fileId", fileId); + command.CommandText = "SELECT content FROM chunks WHERE id = @chunkId"; + var chunkId = command.Parameters.AddWithValue("@chunkId", 0L); var lastEmittedLine = 0; - using var reader = command.ExecuteTrackedReader(); - while (reader.TrackedRead()) + var resumed = false; + while (true) { - var chunkStartLine = reader.GetInt32(0); - var chunkEndLine = reader.GetInt32(1); - var chunkLines = reader.GetString(2).Split('\n'); - var lineCount = chunkEndLine - chunkStartLine + 1; - - for (var index = 0; index < chunkLines.Length && index < lineCount; index++) + cancellationToken.ThrowIfCancellationRequested(); + // LIMIT bounds the legacy top-N sort to scalar metadata. Source is loaded + // only after selection, so neither SQLite nor the page retains file content. + // 旧索引のソートを少量のメタデータに制限し、選択後にだけ本文を取得する。 + pageCommand.CommandText = FindLineChunkPageSql(ordered, resumed); + var count = 0; + using (var page = pageCommand.ExecuteTrackedReader()) { - var absoluteLine = chunkStartLine + index; - if (absoluteLine <= lastEmittedLine) - continue; - - lastEmittedLine = absoluteLine; - yield return new IndexedLine(absoluteLine, chunkLines[index]); + while (page.TrackedRead()) + { + cancellationToken.ThrowIfCancellationRequested(); + count++; + var chunkStartLine = page.GetInt32(1); + var chunkEndLine = page.GetInt32(2); + startLine.Value = chunkStartLine; + chunkIndex.Value = page.GetInt32(3); + chunkId.Value = page.GetInt64(0); + using var reader = command.ExecuteTrackedReader(); + if (!reader.TrackedRead()) + throw new InvalidOperationException("Indexed find chunk is no longer available."); + var chunkLines = reader.GetString(0).Split('\n'); + var lineCount = chunkEndLine - chunkStartLine + 1; + + for (var index = 0; index < chunkLines.Length && index < lineCount; index++) + { + cancellationToken.ThrowIfCancellationRequested(); + var absoluteLine = chunkStartLine + index; + if (absoluteLine <= lastEmittedLine) + continue; + + lastEmittedLine = absoluteLine; + yield return new IndexedLine(absoluteLine, chunkLines[index]); + } + } } + if (count < FindChunkPageSize) + yield break; + resumed = true; } } + private bool HasNullChunkContent(long fileId) + { + using var command = _owner._conn.CreateCommand(); + command.CommandText = "SELECT 1 FROM chunks WHERE file_id = @fileId AND content IS NULL LIMIT 1"; + SqliteCommandPolicy.Add(command, "@fileId", fileId); + using var reader = command.ExecuteTrackedReader(); + return reader.TrackedRead(); + } + internal static IEnumerable EnumerateLineMatches( string lineText, string? fileLang, diff --git a/tests/CodeIndex.Tests/FindChunkReadTests.cs b/tests/CodeIndex.Tests/FindChunkReadTests.cs new file mode 100644 index 0000000000..8be799684c --- /dev/null +++ b/tests/CodeIndex.Tests/FindChunkReadTests.cs @@ -0,0 +1,167 @@ +using CodeIndex.Database; +using CodeIndex.Models; +using Microsoft.Data.Sqlite; + +namespace CodeIndex.Tests; + +public sealed class FindChunkReadTests +{ + [Theory] + [InlineData(false)] + [InlineData(true)] + public void LineChunks_BoundSortingAndPreserveScanSemantics_Issue5415(bool legacy) + { + using var project = TestProjectHelper.CreateTempProjectScope("cdidx_find_chunks_5415"); + var db = TestProjectHelper.CreateProjectDb(project.Root); + using var connection = new SqliteConnection(new SqliteConnectionStringBuilder + { + DataSource = db, Pooling = false + }.ToString()); + connection.Open(); + var writer = new DbWriter(connection); + var chunkCount = DbReader.FindChunkPageSize + 1; + var totalLines = chunkCount * 2 + 1; + var id = writer.UpsertFile(new FileRecord { Path = "source.txt", Lang = "text", Lines = totalLines, Size = 4096 }); + writer.InsertChunks(Enumerable.Range(0, chunkCount).Reverse().Select(index => new ChunkRecord + { + FileId = id, ChunkIndex = 1000 + index, StartLine = index * 2 + 1, EndLine = index * 2 + 3, + Content = "needle\r\ncontext\r\nneedle" + }).Append(new ChunkRecord + { + FileId = id, ChunkIndex = 999, StartLine = 1, EndLine = 3, + Content = "needle needle\r\nowned\r\nneedle" + }).ToList()); + // A missing indexed source must not fall back to the live file. + // 索引本文が欠けた場合も実ファイルへフォールバックしない。 + writer.UpsertFile(new FileRecord { Path = "missing.txt", Lang = "text", Lines = 1, Size = 6 }); + File.WriteAllText(Path.Combine(project.Root, "missing.txt"), "needle"); + File.WriteAllText(Path.Combine(project.Root, "source.txt"), "unindexed live changes"); + if (legacy) + { + using var drop = connection.CreateCommand(); + drop.CommandText = $"DROP INDEX {DbReader.BoundedResourceReadChunkIndexName}; DROP INDEX {DbReader.BoundedResourceReadChunkEndIndexName}"; + drop.ExecuteNonQuery(); + } + using (var readOnly = connection.CreateCommand()) + { + readOnly.CommandText = "PRAGMA query_only = ON"; + readOnly.ExecuteNonQuery(); + } + + foreach (var resumed in new[] { false, true }) + { + using var command = connection.CreateCommand(); + command.CommandText = DbReader.FindLineChunkPageSql(!legacy, resumed); + command.Parameters.AddWithValue("@fileId", id); + command.Parameters.AddWithValue("@startLine", 1); + command.Parameters.AddWithValue("@chunkIndex", 999); + using (var rows = command.ExecuteReader()) + { + Assert.Equal(4, rows.FieldCount); + Assert.Equal(new[] { "id", "start_line", "end_line", "chunk_index" }, + Enumerable.Range(0, rows.FieldCount).Select(rows.GetName)); + var count = 0; + while (rows.Read()) count++; + Assert.Equal(DbReader.FindChunkPageSize, count); + } + command.CommandText = "EXPLAIN QUERY PLAN " + command.CommandText; + using var plan = command.ExecuteReader(); + var details = new List(); + while (plan.Read()) details.Add(plan.GetString(3)); + if (legacy) + Assert.Contains(details, detail => detail.Contains("TEMP B-TREE", StringComparison.OrdinalIgnoreCase)); + else + { + Assert.Contains(details, detail => detail.Contains(DbReader.BoundedResourceReadChunkIndexName, StringComparison.Ordinal)); + Assert.DoesNotContain(details, detail => detail.Contains("TEMP B-TREE", StringComparison.OrdinalIgnoreCase)); + } + } + + using var reader = new DbReader(connection); + foreach (var (query, regex, exact, candidates) in new[] + { + ("needle", false, false, false), ("needle", false, false, true), + ("needle", false, true, false), ("needle", true, false, false) + }) + { + var full = reader.FindInFiles(query, 200, regex: regex, exact: exact, useIndexedLiteralCandidates: candidates, + before: 1, after: 1); + Assert.Equal(chunkCount + 2, full.Count); + Assert.Equal(new[] { 1 }.Concat(Enumerable.Range(0, chunkCount + 1).Select(index => index * 2 + 1)), + full.Select(row => row.Line)); + Assert.Equal(new[] { 1, 8 }, full.Take(2).Select(row => row.Column)); + Assert.Contains("owned", full[0].Snippet, StringComparison.Ordinal); + Assert.All(full, row => Assert.Equal("source.txt", row.Path)); + Assert.False(full.Scan.Truncated); + Assert.Equal(full.Count, reader.CountFindInFiles(query, regex: regex, exact: exact, + useIndexedLiteralCandidates: candidates).Count); + Assert.Empty(reader.FindInFiles(query, 200, regex: regex, pathPatterns: ["missing.txt"])); + + var paged = new List(); + FindScanSummary scan = default; + for (var pageIndex = 0; pageIndex < full.Count; pageIndex++) + { + var page = reader.FindInFiles(query, 5, regex: regex, exact: exact, useIndexedLiteralCandidates: candidates, + before: 1, after: 1, captureContinuation: true, resumePath: scan.NextPath, + resumeLine: scan.NextLine, resumeFileOrdinal: scan.NextFileOrdinal, + resumeMatchOrdinal: scan.NextMatchOrdinal, resumeByteOffset: scan.NextByteOffset); + paged.AddRange(page); + scan = page.Scan; + if (scan.NextPath is null) break; + } + Assert.Null(scan.NextPath); + Assert.Equal(full.Select(Position), paged.Select(Position)); + + var countSum = 0; + scan = default; + for (var pageIndex = 0; pageIndex < totalLines; pageIndex++) + { + var page = reader.CountFindInFiles(query, regex: regex, exact: exact, useIndexedLiteralCandidates: candidates, + maxLinesScanned: 7, resumePath: scan.NextPath, resumeLine: scan.NextLine, + resumeFileOrdinal: scan.NextFileOrdinal); + Assert.InRange(page.Scan.LinesScanned, 0, 7); + countSum += page.Count; + scan = page.Scan; + if (scan.NextPath is null) break; + Assert.Equal("line_scan_limit", scan.TruncationReason); + } + Assert.Null(scan.NextPath); + Assert.Equal(full.Count, countSum); + } + } + + [Theory] + [InlineData(false)] + [InlineData(true)] + public void NullSource_PreservesOrderedFailureAndEarlyStop_Issue5415(bool legacy) + { + using var project = TestProjectHelper.CreateTempProjectScope("cdidx_find_null_5415"); + var db = TestProjectHelper.CreateProjectDb(project.Root); + using var connection = new SqliteConnection(new SqliteConnectionStringBuilder + { + DataSource = db, Pooling = false + }.ToString()); + connection.Open(); + var writer = new DbWriter(connection); + var id = writer.UpsertFile(new FileRecord { Path = "null.txt", Lang = "text", Lines = 2, Size = 6 }); + writer.InsertChunks([new ChunkRecord { FileId = id, ChunkIndex = 1, StartLine = 1, EndLine = 1, Content = "needle" }]); + using (var insert = connection.CreateCommand()) + { + insert.CommandText = "INSERT INTO chunks(file_id, chunk_index, start_line, end_line, content) VALUES (@fileId, 0, 2, 2, NULL)"; + insert.Parameters.AddWithValue("@fileId", id); + insert.ExecuteNonQuery(); + if (legacy) + { + insert.CommandText = $"DROP INDEX {DbReader.BoundedResourceReadChunkIndexName}"; + insert.ExecuteNonQuery(); + } + } + using var reader = new DbReader(connection); + Assert.Equal(1, Assert.Single(reader.FindInFiles("needle", 1)).Line); + Assert.Throws(() => reader.CountFindInFiles("needle")); + Assert.Throws(() => reader.FindInFiles("missing", 10, regex: true)); + } + + private static string Position(FileFindResult row) + => $"{row.Path}:{row.Line}:{row.Column}:{row.Length}:{row.StartLine}:{row.EndLine}:{row.Snippet}"; +} From 60d3b5886aef28cff48470bd1fbb230b810283f3 Mon Sep 17 00:00:00 2001 From: Widthdom Date: Wed, 23 Sep 2026 06:01:37 +0900 Subject: [PATCH 2/2] Fix CI formatting for find chunk regression tests (#5415) --- tests/CodeIndex.Tests/FindChunkReadTests.cs | 16 ++++++++++++---- 1 file changed, 12 insertions(+), 4 deletions(-) diff --git a/tests/CodeIndex.Tests/FindChunkReadTests.cs b/tests/CodeIndex.Tests/FindChunkReadTests.cs index 8be799684c..c80ee8e163 100644 --- a/tests/CodeIndex.Tests/FindChunkReadTests.cs +++ b/tests/CodeIndex.Tests/FindChunkReadTests.cs @@ -15,7 +15,8 @@ public void LineChunks_BoundSortingAndPreserveScanSemantics_Issue5415(bool legac var db = TestProjectHelper.CreateProjectDb(project.Root); using var connection = new SqliteConnection(new SqliteConnectionStringBuilder { - DataSource = db, Pooling = false + DataSource = db, + Pooling = false }.ToString()); connection.Open(); var writer = new DbWriter(connection); @@ -24,11 +25,17 @@ public void LineChunks_BoundSortingAndPreserveScanSemantics_Issue5415(bool legac var id = writer.UpsertFile(new FileRecord { Path = "source.txt", Lang = "text", Lines = totalLines, Size = 4096 }); writer.InsertChunks(Enumerable.Range(0, chunkCount).Reverse().Select(index => new ChunkRecord { - FileId = id, ChunkIndex = 1000 + index, StartLine = index * 2 + 1, EndLine = index * 2 + 3, + FileId = id, + ChunkIndex = 1000 + index, + StartLine = index * 2 + 1, + EndLine = index * 2 + 3, Content = "needle\r\ncontext\r\nneedle" }).Append(new ChunkRecord { - FileId = id, ChunkIndex = 999, StartLine = 1, EndLine = 3, + FileId = id, + ChunkIndex = 999, + StartLine = 1, + EndLine = 3, Content = "needle needle\r\nowned\r\nneedle" }).ToList()); // A missing indexed source must not fall back to the live file. @@ -139,7 +146,8 @@ public void NullSource_PreservesOrderedFailureAndEarlyStop_Issue5415(bool legacy var db = TestProjectHelper.CreateProjectDb(project.Root); using var connection = new SqliteConnection(new SqliteConnectionStringBuilder { - DataSource = db, Pooling = false + DataSource = db, + Pooling = false }.ToString()); connection.Open(); var writer = new DbWriter(connection);