diff --git a/TESTING_GUIDE.md b/TESTING_GUIDE.md index 22e444260..4dd050619 100644 --- a/TESTING_GUIDE.md +++ b/TESTING_GUIDE.md @@ -56,6 +56,13 @@ Also cover index-ordered chunk metadata, bounded legacy fallback, offset-only fi page authority, and combined execution/output caps without synthesized continuation in the CLI and both MCP tools. +`FindChunkReadTests` (#5415) shares a many-chunk fixture across literal, exact, +trigram-candidate and line-regex row/count queries on current and read-only legacy +indexes. Keep ordered query-plan checks, bounded metadata pages, reversed insertion, +overlap ownership, missing/live source controls, NULL-content failures, line caps, +and row/count continuation across chunk pages. Run on net8/net9 alongside existing +find, pagination and MCP regressions; no peak-memory or latency claim is implied. + `InspectCompactCandidateTests` (#5397) shares a real indexed fixture across CLI inspect and MCP compact analysis. Keep zero, exact-limit, probe-overflow and the internal five-definition boundary, partial and unrelated same-name/overload @@ -1867,6 +1874,13 @@ UTF-16 座標、重複チャンク、一致範囲の上限内外、アンカー チャンクの索引順取得と旧索引の上限付き代替処理、offset だけで再開した最終ページの確定性、 実行上限と出力上限が同時に発生しても継続カーソルを作らない規則を CLI と両 MCP ツールで検証します。 +`FindChunkReadTests`(#5415)は多数のチャンクを持つ共通 fixture を使い、現在の索引と +読取専用の旧索引でリテラル・exact・trigram 候補・行単位正規表現の行/件数検索を検証します。 +順序付きクエリ計画、上限付きメタデータページ、逆順挿入、重複行の優先順、索引本文の欠落と +実ファイルの対照、NULL 本文エラー、行上限、チャンクページをまたぐ行/件数の継続取得を +維持してください。既存の find・ページ分割・MCP 回帰とともに net8/net9 で実行します。 +ピークメモリや実行時間の測定結果を保証するテストではありません。 + `ToolsList_SuggestionDisclosesConditionalGitHubPublication_Issue5396` は共通の フィクスチャで、既定・compact・full・ツール選択・ページ付きの一覧を検証します。 送信ツールを呼び出さず、外部サービスとのやり取りを示す注釈、その他の注釈値、 diff --git a/changelog.d/unreleased/5415.fixed.md b/changelog.d/unreleased/5415.fixed.md new file mode 100644 index 000000000..898a57444 --- /dev/null +++ b/changelog.d/unreleased/5415.fixed.md @@ -0,0 +1,19 @@ +--- +category: fixed +issues: + - 5415 +affected: + - src/CodeIndex/Database/IndexedFindQuerySource.cs + - src/CodeIndex/Database/IndexedFindPipeline.cs + - tests/CodeIndex.Tests/FindChunkReadTests.cs + - docs/find-scan-controls.md + - TESTING_GUIDE.md +--- + +## English + +- **Line-local find no longer sorts source content (#5415)** — literal and line-regex queries read chunks in index order, with bounded metadata pages for older indexes. Source content is fetched separately per chunk, preserving overlap ownership, missing-source behavior, counts, scan limits and continuation without requiring reindexing. + +## 日本語 + +- **行単位の find で本文をソートしなくなりました (#5415)** — リテラル・行単位正規表現検索は索引順にチャンクを読み、旧索引では上限付きのメタデータページを使います。本文はチャンクごとに別途取得し、重複行の優先順、本文欠落時の挙動、件数、走査上限、継続取得を維持します。再索引は不要です。 diff --git a/docs/find-scan-controls.md b/docs/find-scan-controls.md index 682a7790b..3430d6dde 100644 --- a/docs/find-scan-controls.md +++ b/docs/find-scan-controls.md @@ -2,6 +2,17 @@ ## English +Literal and line-regex `find` read indexed chunks in line/chunk order without +sorting source content. Current indexes use the ordered partial chunk index; +older indexes page at most 64 scalar chunk records through a metadata-only sort, +then read each selected chunk's content separately. Pages use the last ordering +key rather than an increasing offset. Retained sort data is bounded, but legacy +pages may rescan file metadata and one chunk's content is still read at a time. +A bounded-memory NULL-content check preserves the existing missing-content error +by selecting the all-row fallback for such files; that check may scan the file's +chunk entries. Missing chunks never trigger live-file reads. Overlap ownership, +counts, line caps and continuation positions are unchanged; no reindex is required. + ### CLI cursor validation errors (#5412) Bounded `find` validation honors machine output selected by `--json`, @@ -222,6 +233,16 @@ text or JSON output when context from `--before`, `--after`, or ## 日本語 +リテラル検索と行単位の正規表現 `find` は、本文をソートせず行・チャンク順に索引を読みます。 +現在の索引では順序付き部分インデックスを使い、旧索引では最大64件のチャンクメタデータだけを +ソートしてから、選択した各チャンクの本文を個別に取得します。ページは増大する offset ではなく +直前の順序キーから再開します。ソートの保持量には上限がありますが、旧索引ではページごとに +ファイルのメタデータを再走査する場合があり、本文もチャンク1個単位では読み取ります。 +NULL 本文の確認は保持メモリを制限して行い、該当するファイルでは全行を対象とする代替経路により +従来の本文欠落エラーを維持します。この確認もファイルのチャンク行を走査する場合があります。 +チャンクが欠けても実ファイルへ読み取りを切り替えません。重複行の優先順、件数、行上限、 +継続位置は従来どおりで、再索引は不要です。 + ### CLI カーソルの検証エラー (#5412) 上限付き `find` の検証は、`--json`、`--json=ndjson`、`--format json`、 diff --git a/src/CodeIndex/Database/IndexedFindPipeline.cs b/src/CodeIndex/Database/IndexedFindPipeline.cs index 7de7e27cc..06cf09916 100644 --- a/src/CodeIndex/Database/IndexedFindPipeline.cs +++ b/src/CodeIndex/Database/IndexedFindPipeline.cs @@ -139,7 +139,7 @@ private bool ScanFileLines( var origins = request.SemanticFilters is null ? null : _owner.CreateFindOriginContext(file, request.CancellationToken); var firstContextLine = Math.Max(1, file.FirstEligibleLine - collector.ContextBefore); var stopScanning = false; - foreach (var indexedLine in _querySource.EnumerateIndexedFileLines(file.Id)) + foreach (var indexedLine in _querySource.EnumerateIndexedFileLines(file.Id, request.CancellationToken)) { request.CancellationToken.ThrowIfCancellationRequested(); if (indexedLine.Number > file.TotalLines) diff --git a/src/CodeIndex/Database/IndexedFindQuerySource.cs b/src/CodeIndex/Database/IndexedFindQuerySource.cs index 79bf3151f..a6f627b14 100644 --- a/src/CodeIndex/Database/IndexedFindQuerySource.cs +++ b/src/CodeIndex/Database/IndexedFindQuerySource.cs @@ -5,6 +5,18 @@ namespace CodeIndex.Database; public partial class DbReader { + internal const int FindChunkPageSize = 64; + + internal static string FindLineChunkPageSql(bool ordered, bool resumed) => $""" + SELECT c.id, c.start_line, c.end_line, c.chunk_index + FROM chunks c{(ordered ? $" INDEXED BY {BoundedResourceReadChunkIndexName}" : "")} + WHERE c.file_id = @fileId + {(ordered ? "AND c.content IS NOT NULL" : "")} + {(resumed ? "AND (c.start_line, c.chunk_index) > (@startLine, @chunkIndex)" : "")} + ORDER BY c.start_line, c.chunk_index + LIMIT {FindChunkPageSize} + """; + private sealed class IndexedFindQuerySource(DbReader owner) { private readonly DbReader _owner = owner; @@ -85,37 +97,75 @@ SELECT DISTINCT find_chunk.file_id return command; } - internal IEnumerable EnumerateIndexedFileLines(long fileId) + internal IEnumerable EnumerateIndexedFileLines(long fileId, CancellationToken cancellationToken) { + // The partial index excludes NULL source. Keep the original read failure for + // such rows by routing that file through the all-row metadata fallback. + // 部分索引から除外される NULL 本文は、全行のメタデータ経路で従来の読取エラーを維持する。 + var ordered = _owner._chunkIndexes.Contains(BoundedResourceReadChunkIndexName) + && !HasNullChunkContent(fileId); + using var pageCommand = _owner._conn.CreateCommand(); + SqliteCommandPolicy.Add(pageCommand, "@fileId", fileId); + var startLine = pageCommand.Parameters.AddWithValue("@startLine", 0); + var chunkIndex = pageCommand.Parameters.AddWithValue("@chunkIndex", 0); using var command = _owner._conn.CreateCommand(); - command.CommandText = @" - SELECT c.start_line, c.end_line, c.content - FROM chunks c - WHERE c.file_id = @fileId - ORDER BY c.start_line, c.chunk_index"; - SqliteCommandPolicy.Add(command, "@fileId", fileId); + command.CommandText = "SELECT content FROM chunks WHERE id = @chunkId"; + var chunkId = command.Parameters.AddWithValue("@chunkId", 0L); var lastEmittedLine = 0; - using var reader = command.ExecuteTrackedReader(); - while (reader.TrackedRead()) + var resumed = false; + while (true) { - var chunkStartLine = reader.GetInt32(0); - var chunkEndLine = reader.GetInt32(1); - var chunkLines = reader.GetString(2).Split('\n'); - var lineCount = chunkEndLine - chunkStartLine + 1; - - for (var index = 0; index < chunkLines.Length && index < lineCount; index++) + cancellationToken.ThrowIfCancellationRequested(); + // LIMIT bounds the legacy top-N sort to scalar metadata. Source is loaded + // only after selection, so neither SQLite nor the page retains file content. + // 旧索引のソートを少量のメタデータに制限し、選択後にだけ本文を取得する。 + pageCommand.CommandText = FindLineChunkPageSql(ordered, resumed); + var count = 0; + using (var page = pageCommand.ExecuteTrackedReader()) { - var absoluteLine = chunkStartLine + index; - if (absoluteLine <= lastEmittedLine) - continue; - - lastEmittedLine = absoluteLine; - yield return new IndexedLine(absoluteLine, chunkLines[index]); + while (page.TrackedRead()) + { + cancellationToken.ThrowIfCancellationRequested(); + count++; + var chunkStartLine = page.GetInt32(1); + var chunkEndLine = page.GetInt32(2); + startLine.Value = chunkStartLine; + chunkIndex.Value = page.GetInt32(3); + chunkId.Value = page.GetInt64(0); + using var reader = command.ExecuteTrackedReader(); + if (!reader.TrackedRead()) + throw new InvalidOperationException("Indexed find chunk is no longer available."); + var chunkLines = reader.GetString(0).Split('\n'); + var lineCount = chunkEndLine - chunkStartLine + 1; + + for (var index = 0; index < chunkLines.Length && index < lineCount; index++) + { + cancellationToken.ThrowIfCancellationRequested(); + var absoluteLine = chunkStartLine + index; + if (absoluteLine <= lastEmittedLine) + continue; + + lastEmittedLine = absoluteLine; + yield return new IndexedLine(absoluteLine, chunkLines[index]); + } + } } + if (count < FindChunkPageSize) + yield break; + resumed = true; } } + private bool HasNullChunkContent(long fileId) + { + using var command = _owner._conn.CreateCommand(); + command.CommandText = "SELECT 1 FROM chunks WHERE file_id = @fileId AND content IS NULL LIMIT 1"; + SqliteCommandPolicy.Add(command, "@fileId", fileId); + using var reader = command.ExecuteTrackedReader(); + return reader.TrackedRead(); + } + internal static IEnumerable EnumerateLineMatches( string lineText, string? fileLang, diff --git a/tests/CodeIndex.Tests/FindChunkReadTests.cs b/tests/CodeIndex.Tests/FindChunkReadTests.cs new file mode 100644 index 000000000..c80ee8e16 --- /dev/null +++ b/tests/CodeIndex.Tests/FindChunkReadTests.cs @@ -0,0 +1,175 @@ +using CodeIndex.Database; +using CodeIndex.Models; +using Microsoft.Data.Sqlite; + +namespace CodeIndex.Tests; + +public sealed class FindChunkReadTests +{ + [Theory] + [InlineData(false)] + [InlineData(true)] + public void LineChunks_BoundSortingAndPreserveScanSemantics_Issue5415(bool legacy) + { + using var project = TestProjectHelper.CreateTempProjectScope("cdidx_find_chunks_5415"); + var db = TestProjectHelper.CreateProjectDb(project.Root); + using var connection = new SqliteConnection(new SqliteConnectionStringBuilder + { + DataSource = db, + Pooling = false + }.ToString()); + connection.Open(); + var writer = new DbWriter(connection); + var chunkCount = DbReader.FindChunkPageSize + 1; + var totalLines = chunkCount * 2 + 1; + var id = writer.UpsertFile(new FileRecord { Path = "source.txt", Lang = "text", Lines = totalLines, Size = 4096 }); + writer.InsertChunks(Enumerable.Range(0, chunkCount).Reverse().Select(index => new ChunkRecord + { + FileId = id, + ChunkIndex = 1000 + index, + StartLine = index * 2 + 1, + EndLine = index * 2 + 3, + Content = "needle\r\ncontext\r\nneedle" + }).Append(new ChunkRecord + { + FileId = id, + ChunkIndex = 999, + StartLine = 1, + EndLine = 3, + Content = "needle needle\r\nowned\r\nneedle" + }).ToList()); + // A missing indexed source must not fall back to the live file. + // 索引本文が欠けた場合も実ファイルへフォールバックしない。 + writer.UpsertFile(new FileRecord { Path = "missing.txt", Lang = "text", Lines = 1, Size = 6 }); + File.WriteAllText(Path.Combine(project.Root, "missing.txt"), "needle"); + File.WriteAllText(Path.Combine(project.Root, "source.txt"), "unindexed live changes"); + if (legacy) + { + using var drop = connection.CreateCommand(); + drop.CommandText = $"DROP INDEX {DbReader.BoundedResourceReadChunkIndexName}; DROP INDEX {DbReader.BoundedResourceReadChunkEndIndexName}"; + drop.ExecuteNonQuery(); + } + using (var readOnly = connection.CreateCommand()) + { + readOnly.CommandText = "PRAGMA query_only = ON"; + readOnly.ExecuteNonQuery(); + } + + foreach (var resumed in new[] { false, true }) + { + using var command = connection.CreateCommand(); + command.CommandText = DbReader.FindLineChunkPageSql(!legacy, resumed); + command.Parameters.AddWithValue("@fileId", id); + command.Parameters.AddWithValue("@startLine", 1); + command.Parameters.AddWithValue("@chunkIndex", 999); + using (var rows = command.ExecuteReader()) + { + Assert.Equal(4, rows.FieldCount); + Assert.Equal(new[] { "id", "start_line", "end_line", "chunk_index" }, + Enumerable.Range(0, rows.FieldCount).Select(rows.GetName)); + var count = 0; + while (rows.Read()) count++; + Assert.Equal(DbReader.FindChunkPageSize, count); + } + command.CommandText = "EXPLAIN QUERY PLAN " + command.CommandText; + using var plan = command.ExecuteReader(); + var details = new List(); + while (plan.Read()) details.Add(plan.GetString(3)); + if (legacy) + Assert.Contains(details, detail => detail.Contains("TEMP B-TREE", StringComparison.OrdinalIgnoreCase)); + else + { + Assert.Contains(details, detail => detail.Contains(DbReader.BoundedResourceReadChunkIndexName, StringComparison.Ordinal)); + Assert.DoesNotContain(details, detail => detail.Contains("TEMP B-TREE", StringComparison.OrdinalIgnoreCase)); + } + } + + using var reader = new DbReader(connection); + foreach (var (query, regex, exact, candidates) in new[] + { + ("needle", false, false, false), ("needle", false, false, true), + ("needle", false, true, false), ("needle", true, false, false) + }) + { + var full = reader.FindInFiles(query, 200, regex: regex, exact: exact, useIndexedLiteralCandidates: candidates, + before: 1, after: 1); + Assert.Equal(chunkCount + 2, full.Count); + Assert.Equal(new[] { 1 }.Concat(Enumerable.Range(0, chunkCount + 1).Select(index => index * 2 + 1)), + full.Select(row => row.Line)); + Assert.Equal(new[] { 1, 8 }, full.Take(2).Select(row => row.Column)); + Assert.Contains("owned", full[0].Snippet, StringComparison.Ordinal); + Assert.All(full, row => Assert.Equal("source.txt", row.Path)); + Assert.False(full.Scan.Truncated); + Assert.Equal(full.Count, reader.CountFindInFiles(query, regex: regex, exact: exact, + useIndexedLiteralCandidates: candidates).Count); + Assert.Empty(reader.FindInFiles(query, 200, regex: regex, pathPatterns: ["missing.txt"])); + + var paged = new List(); + FindScanSummary scan = default; + for (var pageIndex = 0; pageIndex < full.Count; pageIndex++) + { + var page = reader.FindInFiles(query, 5, regex: regex, exact: exact, useIndexedLiteralCandidates: candidates, + before: 1, after: 1, captureContinuation: true, resumePath: scan.NextPath, + resumeLine: scan.NextLine, resumeFileOrdinal: scan.NextFileOrdinal, + resumeMatchOrdinal: scan.NextMatchOrdinal, resumeByteOffset: scan.NextByteOffset); + paged.AddRange(page); + scan = page.Scan; + if (scan.NextPath is null) break; + } + Assert.Null(scan.NextPath); + Assert.Equal(full.Select(Position), paged.Select(Position)); + + var countSum = 0; + scan = default; + for (var pageIndex = 0; pageIndex < totalLines; pageIndex++) + { + var page = reader.CountFindInFiles(query, regex: regex, exact: exact, useIndexedLiteralCandidates: candidates, + maxLinesScanned: 7, resumePath: scan.NextPath, resumeLine: scan.NextLine, + resumeFileOrdinal: scan.NextFileOrdinal); + Assert.InRange(page.Scan.LinesScanned, 0, 7); + countSum += page.Count; + scan = page.Scan; + if (scan.NextPath is null) break; + Assert.Equal("line_scan_limit", scan.TruncationReason); + } + Assert.Null(scan.NextPath); + Assert.Equal(full.Count, countSum); + } + } + + [Theory] + [InlineData(false)] + [InlineData(true)] + public void NullSource_PreservesOrderedFailureAndEarlyStop_Issue5415(bool legacy) + { + using var project = TestProjectHelper.CreateTempProjectScope("cdidx_find_null_5415"); + var db = TestProjectHelper.CreateProjectDb(project.Root); + using var connection = new SqliteConnection(new SqliteConnectionStringBuilder + { + DataSource = db, + Pooling = false + }.ToString()); + connection.Open(); + var writer = new DbWriter(connection); + var id = writer.UpsertFile(new FileRecord { Path = "null.txt", Lang = "text", Lines = 2, Size = 6 }); + writer.InsertChunks([new ChunkRecord { FileId = id, ChunkIndex = 1, StartLine = 1, EndLine = 1, Content = "needle" }]); + using (var insert = connection.CreateCommand()) + { + insert.CommandText = "INSERT INTO chunks(file_id, chunk_index, start_line, end_line, content) VALUES (@fileId, 0, 2, 2, NULL)"; + insert.Parameters.AddWithValue("@fileId", id); + insert.ExecuteNonQuery(); + if (legacy) + { + insert.CommandText = $"DROP INDEX {DbReader.BoundedResourceReadChunkIndexName}"; + insert.ExecuteNonQuery(); + } + } + using var reader = new DbReader(connection); + Assert.Equal(1, Assert.Single(reader.FindInFiles("needle", 1)).Line); + Assert.Throws(() => reader.CountFindInFiles("needle")); + Assert.Throws(() => reader.FindInFiles("missing", 10, regex: true)); + } + + private static string Position(FileFindResult row) + => $"{row.Path}:{row.Line}:{row.Column}:{row.Length}:{row.StartLine}:{row.EndLine}:{row.Snippet}"; +}